Repository navigation
SDSTOR-25714 craft: checkpoint-trigger design cleanup (PR #181 follow-ups) #182
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Changes from 4 commits
9e8e1eb
bf34d4f
46078e1
a544d21
712e752
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -365,25 +365,17 @@ unique< CraftJournalBackend > make_homestore_journal_backend(shared< homestore:: | |
| return std::make_unique< HomeStoreCraftJournalBackend >(std::move(logstore), vol_ordinal, lba_size); | ||
| } | ||
|
|
||
| // ─── HomeStoreCraftCheckpointTrigger (SDSTOR-22888) ────────────────────────── | ||
| // | ||
| // Thin wrapper over homestore::cp_mgr(). One instance is shared by every volume's CraftReplDev. | ||
|
|
||
| class HomeStoreCraftCheckpointTrigger : public CraftCheckpointTrigger { | ||
| public: | ||
| async_status trigger_cp_flush(bool force) override { | ||
| checkpoint_trigger_fn_t make_homestore_checkpoint_trigger_fn() { | ||
| return [](bool force) -> async_status { | ||
| if (co_await homestore::cp_mgr().trigger_cp_flush(force)) co_return ok(); | ||
| // cp_mgr().trigger_cp_flush() returns false, synchronously, when a flush is already in | ||
| // progress (cp_mgr.cpp: m_in_flush_phase). That's expected and harmless when force=false | ||
| // (this trigger's only current caller, fire_checkpoint_trigger, coalesces with any in-flight | ||
| // flush on purpose). Only a force=true false return is a real failure worth surfacing. | ||
| if (!force) co_return ok(); | ||
| co_return std::unexpected(make_error_condition(volume_error::INTERNAL_ERROR)); | ||
| } | ||
| }; | ||
|
|
||
| unique< CraftCheckpointTrigger > make_homestore_checkpoint_trigger() { | ||
| return std::make_unique< HomeStoreCraftCheckpointTrigger >(); | ||
| }; | ||
| } | ||
|
|
||
| // ─── constructor ────────────────────────────────────────────────────────────── | ||
|
|
@@ -787,18 +779,18 @@ bool CraftReplDev::checkpoint_interval_crossed_locked(int64_t commit_lsn_snapsho | |
| } | ||
|
|
||
| void CraftReplDev::fire_checkpoint_trigger(int64_t commit_lsn_snapshot) { | ||
| if (checkpoint_trigger_ == nullptr) { | ||
| if (!checkpoint_trigger_) { | ||
| LOGW("commit_lsn={} crossed checkpoint interval but no checkpoint_trigger_ wired -- skipping", | ||
| commit_lsn_snapshot); | ||
| return; | ||
| } | ||
| // force=false: let this coalesce with any checkpoint already in flight rather than forcing | ||
| // back-to-back flushes under high commit throughput (see CraftCheckpointTrigger's doc comment). | ||
| // back-to-back flushes under high commit throughput (see checkpoint_trigger_fn_t's doc comment). | ||
| // Detached (fire-and-forget): nothing here depends on the flush completing. A failure is logged, | ||
| // not propagated, same posture as catch-up/fetch failures elsewhere in this class. | ||
| auto self = shared_from_this(); | ||
| detail::detach([self, commit_lsn_snapshot]() -> async_status { | ||
| if (auto cp = co_await self->checkpoint_trigger_->trigger_cp_flush(false); !cp) | ||
| if (auto cp = co_await self->checkpoint_trigger_(false); !cp) | ||
| LOGE("checkpoint trigger failed at commit_lsn={}: {}", commit_lsn_snapshot, cp.error().message()); | ||
| co_return ok(); | ||
| }()); | ||
|
|
@@ -812,8 +804,8 @@ void CraftReplDev::fire_checkpoint_trigger(int64_t commit_lsn_snapshot) { | |
| // every subsequent write()/keep_alive() retries the advance. Never holds a lock across the co_await | ||
| // read_slot() suspension point below (same rule fetch_data's doc comment already establishes). | ||
|
|
||
| async_result< int64_t > CraftReplDev::commit_impl(int64_t upto_lsn, write_index_fn_t const& write_fn, | ||
| delete_index_fn_t const& delete_fn) { | ||
| async_result< int64_t > CraftReplDev::commit_impl(int64_t upto_lsn, write_index_fn_t write_fn, | ||
| delete_index_fn_t delete_fn) { | ||
| int64_t commit_lsn, last_append_lsn; | ||
| { | ||
| std::lock_guard lk{state_mu_}; | ||
|
|
@@ -823,12 +815,23 @@ async_result< int64_t > CraftReplDev::commit_impl(int64_t upto_lsn, write_index_ | |
| last_append_lsn = state_.last_append_lsn; | ||
| } | ||
| // Guaranteed reset on every exit path (stall, success, or error): a local RAII object's destructor | ||
| // runs when the coroutine frame unwinds, exactly like a plain function's locals on return. | ||
| // runs when the coroutine frame unwinds, exactly like a plain function's locals on return. Also | ||
| // runs the checkpoint-interval check here rather than only on the happy-path tail: folding it into | ||
| // this same critical section means every exit -- including an early error return mid-loop -- still | ||
| // gets a chance to fire the checkpoint (a slot that keeps failing on retry no longer permanently | ||
| // starves it), and merges what would otherwise be two separate state_mu_ acquisitions into one. | ||
| struct RunningGuard { | ||
| CraftReplDev* self; | ||
| ~RunningGuard() { | ||
| std::lock_guard lk{self->state_mu_}; | ||
| self->commit_running_ = false; | ||
| int64_t final_commit_lsn; | ||
| bool should_checkpoint; | ||
| { | ||
| std::lock_guard lk{self->state_mu_}; | ||
| self->commit_running_ = false; | ||
| final_commit_lsn = self->state_.commit_lsn; | ||
| should_checkpoint = self->checkpoint_interval_crossed_locked(final_commit_lsn); | ||
| } | ||
| if (should_checkpoint) self->fire_checkpoint_trigger(final_commit_lsn); | ||
|
|
||
| } | ||
| } guard{this}; | ||
|
|
||
|
|
@@ -946,15 +949,8 @@ async_result< int64_t > CraftReplDev::commit_impl(int64_t upto_lsn, write_index_ | |
| state_.commit_lsn = lsn; | ||
|
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. you can maintain a local variable Update that variable without holding any lock at the end of each loop (line 749)
Contributor
Author
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. ACK |
||
| } | ||
|
|
||
| int64_t final_commit_lsn; | ||
| bool should_checkpoint; | ||
| { | ||
| std::lock_guard lk{state_mu_}; | ||
| final_commit_lsn = state_.commit_lsn; | ||
| should_checkpoint = checkpoint_interval_crossed_locked(final_commit_lsn); | ||
| } | ||
| if (should_checkpoint) fire_checkpoint_trigger(final_commit_lsn); | ||
| co_return final_commit_lsn; | ||
| std::lock_guard lk{state_mu_}; | ||
| co_return state_.commit_lsn; | ||
| } | ||
|
|
||
| async_result< int64_t > CraftReplDev::commit(int64_t upto_lsn) { | ||
|
|
@@ -969,14 +965,14 @@ async_result< int64_t > CraftReplDev::commit(int64_t upto_lsn) { | |
| delete_index_fn_t delete_fn = [this](lba_t s, lba_t e, std::vector< homestore::blk_id >& freed) { | ||
| return indx_tbl_->delete_lba_range(s, e, freed); | ||
| }; | ||
| auto r = co_await commit_impl(upto_lsn, write_fn, delete_fn); | ||
| auto r = co_await commit_impl(upto_lsn, std::move(write_fn), std::move(delete_fn)); | ||
| co_return r; | ||
| } | ||
|
|
||
| #ifdef _PRERELEASE | ||
| async_result< int64_t > CraftReplDev::commit_with(int64_t upto_lsn, write_index_fn_t write_fn, | ||
| delete_index_fn_t delete_fn) { | ||
| auto r = co_await commit_impl(upto_lsn, write_fn, delete_fn); | ||
| auto r = co_await commit_impl(upto_lsn, std::move(write_fn), std::move(delete_fn)); | ||
| co_return r; | ||
| } | ||
| #endif | ||
|
|
||
Uh oh!
There was an error while loading. Please reload this page.