Handle CAS writes with built-in SYNC and without retries in the client
This commit is contained in:
@@ -558,6 +558,24 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_kv_stress_imm:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: IMMEDIATE_COMMIT=1 /root/vitastor/tests/test_kv_stress.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_splitbrain:
|
test_splitbrain:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
|
|||||||
@@ -765,8 +765,13 @@ void cluster_client_t::execute_internal(cluster_op_t *op)
|
|||||||
{
|
{
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
if ((op->opcode == OSD_OP_WRITE || op->opcode == OSD_OP_DELETE) && enable_writeback && !(op->flags & OP_FLUSH_BUFFER) &&
|
// CAS writes are simplified: they're not cached, not resliced, not retried, and not part of the regular write queue at all
|
||||||
!op->version /* no CAS writeback */)
|
if ((op->opcode == OSD_OP_WRITE || op->opcode == OSD_OP_DELETE) && op->version)
|
||||||
|
{
|
||||||
|
execute_cas(op);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if ((op->opcode == OSD_OP_WRITE || op->opcode == OSD_OP_DELETE) && enable_writeback && !(op->flags & OP_FLUSH_BUFFER))
|
||||||
{
|
{
|
||||||
if (wb->writebacks_active >= client_max_writeback_iodepth)
|
if (wb->writebacks_active >= client_max_writeback_iodepth)
|
||||||
{
|
{
|
||||||
@@ -788,7 +793,7 @@ void cluster_client_t::execute_internal(cluster_op_t *op)
|
|||||||
}
|
}
|
||||||
if ((op->opcode == OSD_OP_WRITE || op->opcode == OSD_OP_DELETE) && !(op->flags & OP_IMMEDIATE_COMMIT))
|
if ((op->opcode == OSD_OP_WRITE || op->opcode == OSD_OP_DELETE) && !(op->flags & OP_IMMEDIATE_COMMIT))
|
||||||
{
|
{
|
||||||
if (!(op->flags & OP_FLUSH_BUFFER) && !op->version /* no CAS write-repeat */)
|
if (!(op->flags & OP_FLUSH_BUFFER))
|
||||||
{
|
{
|
||||||
uint64_t flush_id = ++wb->last_flush_id;
|
uint64_t flush_id = ++wb->last_flush_id;
|
||||||
wb->copy_write(op, CACHE_REPEATING, flush_id);
|
wb->copy_write(op, CACHE_REPEATING, flush_id);
|
||||||
@@ -847,6 +852,72 @@ void cluster_client_t::execute_internal(cluster_op_t *op)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void cluster_client_t::execute_cas(cluster_op_t *op)
|
||||||
|
{
|
||||||
|
slice_rw(op);
|
||||||
|
op->needs_reslice = false;
|
||||||
|
if ((op->opcode == OSD_OP_WRITE || op->opcode == OSD_OP_DELETE) && op->version && op->parts.size() > 1)
|
||||||
|
{
|
||||||
|
// Atomic writes to multiple stripes are unsupported
|
||||||
|
op->retval = -EINVAL;
|
||||||
|
auto cb = std::move(op->callback);
|
||||||
|
cb(op);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
int res = try_send(op, 0, [this, op](osd_op_t *part)
|
||||||
|
{
|
||||||
|
int expected = part->req.hdr.opcode == OSD_OP_DELETE ? 0 : part->req.rw.len;
|
||||||
|
op->retval = part->reply.hdr.retval;
|
||||||
|
op->retval = op->retval == expected ? 0 : (op->retval >= 0 ? -EIO : op->retval);
|
||||||
|
op->retval = op->retval == -EPIPE ? -EINTR : op->retval;
|
||||||
|
auto peer_it = msgr.osd_peer_fds.find(op->parts[0].osd_num);
|
||||||
|
if (op->retval != 0 || (op->flags & OP_IMMEDIATE_COMMIT))
|
||||||
|
{
|
||||||
|
auto cb = std::move(op->callback);
|
||||||
|
cb(op);
|
||||||
|
}
|
||||||
|
else if (peer_it == msgr.osd_peer_fds.end())
|
||||||
|
{
|
||||||
|
// Care must be taken to make sure that the client doesn't reconnect to the OSD
|
||||||
|
// before executing the previously completed operation callback (!)
|
||||||
|
op->retval = -EINTR;
|
||||||
|
auto cb = std::move(op->callback);
|
||||||
|
cb(op);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
// CAS writes have a built-in sync
|
||||||
|
auto peer_fd = peer_it->second;
|
||||||
|
*part = (osd_op_t){
|
||||||
|
.op_type = OSD_OP_OUT,
|
||||||
|
.peer_fd = peer_fd,
|
||||||
|
.req = {
|
||||||
|
.hdr = {
|
||||||
|
.magic = SECONDARY_OSD_OP_MAGIC,
|
||||||
|
.opcode = OSD_OP_SYNC,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
.callback = [this, op](osd_op_t *part)
|
||||||
|
{
|
||||||
|
op->retval = part->reply.hdr.retval;
|
||||||
|
op->retval = op->retval == -EPIPE ? -EINTR : op->retval;
|
||||||
|
auto cb = std::move(op->callback);
|
||||||
|
cb(op);
|
||||||
|
},
|
||||||
|
};
|
||||||
|
msgr.outbox_push(part);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
if (res == TRY_SEND_CONNECTING || res == TRY_SEND_OFFLINE)
|
||||||
|
{
|
||||||
|
// In theory, CAS writes could wait for the PG to come up, but it's easier to just fail it
|
||||||
|
op->retval = -EINTR;
|
||||||
|
auto cb = std::move(op->callback);
|
||||||
|
cb(op);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
bool cluster_client_t::check_rw(cluster_op_t *op)
|
bool cluster_client_t::check_rw(cluster_op_t *op)
|
||||||
{
|
{
|
||||||
if (op->opcode == OSD_OP_SYNC)
|
if (op->opcode == OSD_OP_SYNC)
|
||||||
@@ -956,13 +1027,6 @@ resume_0:
|
|||||||
// Slice the operation into parts
|
// Slice the operation into parts
|
||||||
slice_rw(op);
|
slice_rw(op);
|
||||||
op->needs_reslice = false;
|
op->needs_reslice = false;
|
||||||
if ((op->opcode == OSD_OP_WRITE || op->opcode == OSD_OP_DELETE) && op->version && op->parts.size() > 1)
|
|
||||||
{
|
|
||||||
// Atomic writes to multiple stripes are unsupported
|
|
||||||
op->retval = -EINVAL;
|
|
||||||
erase_op(op);
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
resume_1:
|
resume_1:
|
||||||
// Send unsent parts, if they're not subject to change
|
// Send unsent parts, if they're not subject to change
|
||||||
op->state = 2;
|
op->state = 2;
|
||||||
@@ -1307,7 +1371,7 @@ bool cluster_client_t::affects_osd(uint64_t inode, uint64_t offset, uint64_t len
|
|||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
|
||||||
int cluster_client_t::try_send(cluster_op_t *op, int i)
|
int cluster_client_t::try_send(cluster_op_t *op, int i, std::function<void(osd_op_t *op_part)> cb)
|
||||||
{
|
{
|
||||||
if (!msgr_initialized)
|
if (!msgr_initialized)
|
||||||
{
|
{
|
||||||
@@ -1367,7 +1431,7 @@ int cluster_client_t::try_send(cluster_op_t *op, int i)
|
|||||||
? (uint8_t*)op->part_bitmaps + pg_bitmap_size*i : NULL),
|
? (uint8_t*)op->part_bitmaps + pg_bitmap_size*i : NULL),
|
||||||
.bitmap_len = (unsigned)(op->opcode == OSD_OP_READ || op->opcode == OSD_OP_READ_BITMAP || op->opcode == OSD_OP_READ_CHAIN_BITMAP
|
.bitmap_len = (unsigned)(op->opcode == OSD_OP_READ || op->opcode == OSD_OP_READ_BITMAP || op->opcode == OSD_OP_READ_CHAIN_BITMAP
|
||||||
? pg_bitmap_size : 0),
|
? pg_bitmap_size : 0),
|
||||||
.callback = [this, part](osd_op_t *op_part)
|
.callback = cb ? cb : [this, part](osd_op_t *op_part)
|
||||||
{
|
{
|
||||||
handle_op_part(part);
|
handle_op_part(part);
|
||||||
},
|
},
|
||||||
|
|||||||
@@ -175,12 +175,13 @@ protected:
|
|||||||
void on_change_node_placement_hook();
|
void on_change_node_placement_hook();
|
||||||
|
|
||||||
void execute_internal(cluster_op_t *op);
|
void execute_internal(cluster_op_t *op);
|
||||||
|
void execute_cas(cluster_op_t *op);
|
||||||
void unshift_op(cluster_op_t *op);
|
void unshift_op(cluster_op_t *op);
|
||||||
int continue_rw(cluster_op_t *op);
|
int continue_rw(cluster_op_t *op);
|
||||||
bool check_rw(cluster_op_t *op);
|
bool check_rw(cluster_op_t *op);
|
||||||
void slice_rw(cluster_op_t *op);
|
void slice_rw(cluster_op_t *op);
|
||||||
void reset_retry_timer(int new_duration);
|
void reset_retry_timer(int new_duration);
|
||||||
int try_send(cluster_op_t *op, int i);
|
int try_send(cluster_op_t *op, int i, std::function<void(osd_op_t *op_part)> cb = nullptr);
|
||||||
int continue_sync(cluster_op_t *op);
|
int continue_sync(cluster_op_t *op);
|
||||||
void send_sync(cluster_op_t *op, cluster_op_part_t *part);
|
void send_sync(cluster_op_t *op, cluster_op_part_t *part);
|
||||||
void handle_op_part(cluster_op_part_t *part);
|
void handle_op_part(cluster_op_part_t *part);
|
||||||
|
|||||||
@@ -262,7 +262,10 @@ struct __attribute__((__packed__)) osd_reply_del_t
|
|||||||
uint32_t left_on_dead_count;
|
uint32_t left_on_dead_count;
|
||||||
};
|
};
|
||||||
|
|
||||||
// sync to the primary OSD
|
// sync to the primary OSD. semantics:
|
||||||
|
// 1) any non-synced write may disappear on OSD restart. even if it's a CAS write.
|
||||||
|
// 2) sync only guarantees to commit completed writes, not in-progress ones.
|
||||||
|
// 3) sync is a no-op when immediate_commit is active.
|
||||||
struct __attribute__((__packed__)) osd_op_sync_t
|
struct __attribute__((__packed__)) osd_op_sync_t
|
||||||
{
|
{
|
||||||
osd_op_header_t header;
|
osd_op_header_t header;
|
||||||
|
|||||||
+1
-25
@@ -139,7 +139,6 @@ struct kv_db_t
|
|||||||
uint64_t next_free = 0;
|
uint64_t next_free = 0;
|
||||||
uint32_t kv_block_size = 0;
|
uint32_t kv_block_size = 0;
|
||||||
uint32_t ino_block_size = 0;
|
uint32_t ino_block_size = 0;
|
||||||
bool immediate_commit = false;
|
|
||||||
uint64_t memory_limit = 128*1024*1024;
|
uint64_t memory_limit = 128*1024*1024;
|
||||||
uint64_t evict_unused_age = 1000;
|
uint64_t evict_unused_age = 1000;
|
||||||
uint64_t evict_max_misses = 10;
|
uint64_t evict_max_misses = 10;
|
||||||
@@ -529,13 +528,6 @@ void kv_db_t::open(inode_t inode_id, json11::Json cfg, std::function<void(int)>
|
|||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
this->inode_id = inode_id;
|
this->inode_id = inode_id;
|
||||||
this->immediate_commit = cli->get_immediate_commit(inode_id);
|
|
||||||
if (!this->immediate_commit)
|
|
||||||
{
|
|
||||||
// FIXME: CAS is unusable without immediate_commit at the moment. Fix it
|
|
||||||
cb(-EINVAL);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
this->ino_block_size = pool_cfg.data_block_size * pg_data_size;
|
this->ino_block_size = pool_cfg.data_block_size * pg_data_size;
|
||||||
this->kv_block_size = kv_block_size;
|
this->kv_block_size = kv_block_size;
|
||||||
this->next_free = 0;
|
this->next_free = 0;
|
||||||
@@ -695,7 +687,6 @@ void kv_db_t::close(std::function<void()> cb)
|
|||||||
next_free = 0;
|
next_free = 0;
|
||||||
kv_block_size = 0;
|
kv_block_size = 0;
|
||||||
ino_block_size = 0;
|
ino_block_size = 0;
|
||||||
immediate_commit = false;
|
|
||||||
block_cache.clear();
|
block_cache.clear();
|
||||||
known_versions.clear();
|
known_versions.clear();
|
||||||
cb();
|
cb();
|
||||||
@@ -1404,22 +1395,7 @@ static void write_block(kv_db_t *db, kv_block_t *blk, std::function<void(int)> c
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
delete op;
|
delete op;
|
||||||
if (res < 0 || db->immediate_commit)
|
cb(res);
|
||||||
{
|
|
||||||
cb(res);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
op = new cluster_op_t;
|
|
||||||
op->opcode = OSD_OP_SYNC;
|
|
||||||
op->callback = [cb](cluster_op_t *op)
|
|
||||||
{
|
|
||||||
auto res = op->retval;
|
|
||||||
delete op;
|
|
||||||
cb(res);
|
|
||||||
};
|
|
||||||
db->cli->execute(op);
|
|
||||||
}
|
|
||||||
};
|
};
|
||||||
db->cli->execute(op);
|
db->cli->execute(op);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -165,7 +165,6 @@ resume_3:
|
|||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
// Check CAS version
|
// Check CAS version
|
||||||
// FIXME: Handle CAS writes as "immediate" in non-immediate_commit pools, otherwise CAS doesn't make sense
|
|
||||||
if (cur_op->req.rw.version && op_data->fact_ver != (cur_op->req.rw.version-1))
|
if (cur_op->req.rw.version && op_data->fact_ver != (cur_op->req.rw.version-1))
|
||||||
{
|
{
|
||||||
deref_object_state(pg, &op_data->object_state, true);
|
deref_object_state(pg, &op_data->object_state, true);
|
||||||
|
|||||||
@@ -49,6 +49,7 @@ SCHEME=ec ./test_snapshot_chain.sh
|
|||||||
SCHEME=ec ./test_snapshot_down.sh
|
SCHEME=ec ./test_snapshot_down.sh
|
||||||
|
|
||||||
./test_kv_stress.sh
|
./test_kv_stress.sh
|
||||||
|
IMMEDIATE_COMMIT=1 ./test_kv_stress.sh
|
||||||
|
|
||||||
./test_splitbrain.sh
|
./test_splitbrain.sh
|
||||||
|
|
||||||
|
|||||||
@@ -1,6 +1,5 @@
|
|||||||
#!/bin/bash -ex
|
#!/bin/bash -ex
|
||||||
|
|
||||||
IMMEDIATE_COMMIT=1
|
|
||||||
PG_COUNT=16
|
PG_COUNT=16
|
||||||
. `dirname $0`/run_3osds.sh
|
. `dirname $0`/run_3osds.sh
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user