From c0e112dbc148460e7407cf595826527ae372e207 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 01:27:10 +0200 Subject: [PATCH 01/31] Add an unbounded retry policy with attempt spacing to `Retry` `Retry::untilDefinitive` has no window and no lease bound; `bind` saturates, so it ends only on an answer, the fence or liveness. `attempt_spacing_ms` carries the spacing the engine will apply to its reissues. Co-Authored-By: Claude Sonnet 5.5 --- .../ContentAddressed/Backend/CasRetry.cpp | 16 ++ .../ContentAddressed/Backend/CasRetry.h | 19 +++ src/Disks/tests/gtest_cas_requests.cpp | 149 ++++++++++++++++++ 3 files changed, 184 insertions(+) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRetry.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRetry.cpp index c7b4c27215b5..cfbfab930c56 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRetry.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRetry.cpp @@ -17,6 +17,22 @@ uint64_t Retry::backoff(uint32_t attempt) return thread_local_rng() % (ceiling + 1); /// full jitter: uniform(0, ceiling) } +uint64_t Retry::spacedPause(uint64_t spacing_draw_ms, uint64_t request_started_ms, uint64_t now_ms) +{ + const uint64_t taken_ms = now_ms > request_started_ms ? now_ms - request_started_ms : 0; + return taken_ms >= spacing_draw_ms ? 0 : spacing_draw_ms - taken_ms; +} + +uint64_t Retry::drawSpacing(uint64_t attempt_spacing_ms_) +{ + const uint64_t spread = attempt_spacing_ms_ / 5; + const uint64_t low = attempt_spacing_ms_ - spread; + const uint64_t high = attempt_spacing_ms_ > std::numeric_limits::max() - spread + ? std::numeric_limits::max() + : attempt_spacing_ms_ + spread; + return low + thread_local_rng() % (high - low + 1); +} + Retry::Bound Retry::bind(uint64_t now_ms) const { const uint64_t own_deadline_ms = policy_deadline_ms diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRetry.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRetry.h index de235e8fd3fa..7534b0810962 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRetry.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRetry.h @@ -1,5 +1,6 @@ #pragma once #include +#include #include namespace DB::Cas @@ -23,6 +24,10 @@ struct Retry /// iterations a fresh window. Empty for a single verb, which gets its window from where it is /// called. std::optional policy_deadline_ms = std::nullopt; + /// When set, a reissue in `CasOperation`'s write and read loops waits until this long (drawn in + /// [0.8, 1.2] of it) has passed since the failed request started, instead of the engine's backoff. + /// A request that took longer is reissued at once. The first-attempt fuse is reissued at once either way. + std::optional attempt_spacing_ms = std::nullopt; /// Full jitter: uniform(0, min(5000, 200 << (attempt-1))) milliseconds. `attempt` is 1-based; /// `attempt == 0` returns 0. @@ -54,6 +59,20 @@ struct Retry /// The standard policy, but at most one attempt is ever sent. static Retry once() { return {.window_ms = 90'000, .lease_deadline_ms = std::nullopt, .single_attempt = true}; } + /// No window and no lease bound: retried until a definitive answer, or until the fence or the + /// caller's liveness ends it. `bind` saturates, so the only horizon is the range of the clock. + static Retry untilDefinitive(uint64_t attempt_spacing_ms_) + { + return {.window_ms = std::numeric_limits::max(), .lease_deadline_ms = std::nullopt, + .single_attempt = false, .attempt_spacing_ms = attempt_spacing_ms_}; + } + + /// The pause before a retry under `attempt_spacing_ms`: `spacing_draw_ms` minus what the failed + /// request took, not below 0. A clock sample before the start counts as no time taken. + static uint64_t spacedPause(uint64_t spacing_draw_ms, uint64_t request_started_ms, uint64_t now_ms); + /// A uniform draw in [0.8, 1.2] of `attempt_spacing_ms_`, saturating at the top of the range. + static uint64_t drawSpacing(uint64_t attempt_spacing_ms_); + /// This policy, made single-attempt. A frozen loop policy keeps its absolute deadline through it, /// which is what lets a loop send an unrepeatable request under the same bound as the rest. Retry asSingleAttempt() const diff --git a/src/Disks/tests/gtest_cas_requests.cpp b/src/Disks/tests/gtest_cas_requests.cpp index 81086f27f433..92cdbc7ea489 100644 --- a/src/Disks/tests/gtest_cas_requests.cpp +++ b/src/Disks/tests/gtest_cas_requests.cpp @@ -34,6 +34,7 @@ #include +#include #include #include #include @@ -579,6 +580,58 @@ TEST(CASRetry, BindSaturatesAndLeavesAnEqualLeaseOffTheLeaseSource) EXPECT_TRUE(lease.lease_bound); } +TEST(CASRetrySpacing, SpacedPauseWaitsOutTheRestOfTheDraw) +{ + EXPECT_EQ(Retry::spacedPause(1'000, 5'000, 5'000), 1'000u); /// failed at once + EXPECT_EQ(Retry::spacedPause(1'000, 5'000, 5'300), 700u); + EXPECT_EQ(Retry::spacedPause(1'000, 5'000, 6'000), 0u); /// took exactly the draw + EXPECT_EQ(Retry::spacedPause(1'000, 5'000, 10'000), 0u); /// took longer: retried at once + /// A sample before the start counts as no time taken, so the wait is never shortened by it. + EXPECT_EQ(Retry::spacedPause(1'000, 5'000, 4'000), 1'000u); +} + +TEST(CASRetrySpacing, DrawIsWithinAFifthOfTheSpacing) +{ + bool low = false; + bool high = false; + for (int i = 0; i < 2'000; ++i) + { + const uint64_t draw = Retry::drawSpacing(1'000); + ASSERT_GE(draw, 800u); + ASSERT_LE(draw, 1'200u); + low = low || draw < 850; + high = high || draw > 1'150; + } + /// The bounds alone cannot tell a draw from a constant, so both ends must be reached. + EXPECT_TRUE(low); + EXPECT_TRUE(high); + EXPECT_EQ(Retry::drawSpacing(0), 0u); + constexpr uint64_t largest = std::numeric_limits::max(); + EXPECT_GE(Retry::drawSpacing(largest), largest - largest / 5) << "the top of the range must not wrap"; +} + +TEST(CASRetrySpacing, UntilDefinitiveHasNoWindowNoLeaseAndASpacing) +{ + constexpr uint64_t largest = std::numeric_limits::max(); + const Retry policy = Retry::untilDefinitive(1'000); + EXPECT_EQ(policy.window_ms, largest); + EXPECT_FALSE(policy.lease_deadline_ms.has_value()); + EXPECT_FALSE(policy.single_attempt); + EXPECT_FALSE(policy.policy_deadline_ms.has_value()); + EXPECT_EQ(policy.attempt_spacing_ms, std::optional(1'000)); + for (const uint64_t now : {uint64_t{0}, uint64_t{1}, uint64_t{1'000'000}, largest - 1, largest}) + { + const Retry::Bound bound = policy.bind(now); + EXPECT_EQ(bound.deadline_ms, largest) << "now " << now; + EXPECT_FALSE(bound.lease_bound) << "now " << now; + } + /// Every other policy keeps the engine's own backoff. + EXPECT_FALSE(Retry::standard().attempt_spacing_ms.has_value()); + EXPECT_FALSE(Retry::within(1'000).attempt_spacing_ms.has_value()); + EXPECT_FALSE(Retry::once().attempt_spacing_ms.has_value()); + EXPECT_FALSE(Retry::untilLeaseSafe(2'000'000, 2'000).attempt_spacing_ms.has_value()); +} + TEST(CASRequests, CreateThenReplaceThenRemove) { FakeClock clock; @@ -2947,6 +3000,102 @@ TEST(CASRequestsFuse, ReadRefreshedCredentialTextDoesNotDoubleCountTheFuse) EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASRequestFirstAttemptFuse].load() - fuses_before, 0u); } +namespace +{ + +/// A store whose requests each take `duration_ms` of the injected clock and fail as `fault` says. +/// Every request is logged with the instant it started. +class TimedFaultBackend : public InMemoryBackend +{ +public: + enum class Verb : uint8_t { Put, Get }; + struct Sent + { + Verb verb; + uint64_t started_ms; + }; + /// The exception the request fails with, or null to serve it. `nth` counts requests of `verb` from 1. + using Fault = std::function; + + explicit TimedFaultBackend(FakeClock & clock_) : clock(clock_) {} + + std::optional read(const String & key, DB::Cas::TransportAccess & access) override + { + begin(Verb::Get); + return InMemoryBackend::read(key, access); + } + + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + DB::Cas::TransportAccess & access) override + { + begin(Verb::Put); + return InMemoryBackend::write(key, bytes, expected_value, access); + } + + std::vector startsOf(Verb verb) const + { + std::vector starts; + for (const Sent & request : sent) + if (request.verb == verb) + starts.push_back(request.started_ms); + return starts; + } + + Fault fault; + uint64_t duration_ms = 0; + std::vector sent; + +private: + void begin(Verb verb) + { + const uint64_t started = clock.now.load(); + const size_t nth = 1 + static_cast(std::count_if(sent.begin(), sent.end(), + [&](const Sent & request) { return request.verb == verb; })); + sent.push_back({verb, started}); + clock.now.fetch_add(duration_ms); + if (auto error = fault ? fault(verb, nth, started) : nullptr) + std::rethrow_exception(error); + } + + FakeClock & clock; +}; + +using Verb = TimedFaultBackend::Verb; + +constexpr uint64_t kSpacingMs = 1'000; + +} + +/// Spacing is measured on the injected clock, so a bind that wrapped at the top of the range would +/// refuse the first request; the only refusal is two envelopes no longer fitting before the clock ends. +TEST(CASRequestsSpacing, UntilDefinitiveRefusesNothingBeforeTheEndOfTheClock) +{ + constexpr uint64_t largest = std::numeric_limits::max(); + FakeClock clock; + clock.now = largest - 20'000; + auto backend = std::make_shared(clock); + backend->fault = [](Verb verb, size_t, uint64_t) -> std::exception_ptr + { + return verb == Verb::Put ? connectHint() : nullptr; + }; + auto requests = makeRequests(backend, clock); + requests.setAttemptReservationForTest(7'000); + auto op = requests.admit(); + + const WriteResult result = op.create("k", "v", Retry::untilDefinitive(kSpacingMs)); + + const auto * gave_up = std::get_if(&result); + ASSERT_NE(gave_up, nullptr); + EXPECT_EQ(gave_up->why, GaveUp::Why::Deadline); + EXPECT_EQ(gave_up->deadline_source, GaveUp::Source::Policy); + EXPECT_TRUE(gave_up->sent_any); + const auto puts = backend->startsOf(Verb::Put); + ASSERT_GE(puts.size(), 2u); + EXPECT_EQ(puts.front(), largest - 20'000); + EXPECT_LE(puts.back(), largest - 14'000) << "every PUT was admitted with room for its two envelopes"; +} + #endif TEST(CASRequestBudget, EnvelopeIsValidatedNotTheBareAttempt) From 703b9ab01b00903a4e02bd801c1592b35ae9faf1 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 01:27:13 +0200 Subject: [PATCH 02/31] Watch mount tokens through one TokenWatch helper `claimMountAwaitingExpiry` and `computeHeartbeatFloor` each kept their own token and first-seen pair. Both now use `TokenWatch`, whose stability test is false for a sample older than the sighting instead of wrapping. `computeHeartbeatFloor` takes the observer clock as a function; it still samples it once per call. Co-Authored-By: Claude Sonnet 5.5 --- .../ContentAddressed/Gc/CasGc.cpp | 4 +- .../ContentAddressed/Pool/CasServerRoot.cpp | 38 ++++++----- .../ContentAddressed/Pool/CasServerRoot.h | 27 ++++---- src/Disks/tests/gtest_cas_mount.cpp | 63 ++++++++++++++----- 4 files changed, 88 insertions(+), 44 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGc.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGc.cpp index f871182c8edb..918c9b10177b 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGc.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGc.cpp @@ -690,7 +690,7 @@ RoundReport Gc::runRegularRound(std::function on_lease_acquired, bool al /// PUT per newly-fenced mount. { GcPhaseTimer t(phase_sink, "heartbeat_floor"); - const HeartbeatFloor floor = computeHeartbeatFloor(op, layout, now_ms_fn(), mono_ms_fn(), + const HeartbeatFloor floor = computeHeartbeatFloor(op, layout, now_ms_fn(), mono_ms_fn, stable_threshold_ms, mount_obs); report.fence_outs = floor.fenced_now; if (floor.fenced_now > 0) @@ -4704,7 +4704,7 @@ RebuildReport Gc::rebuildBaseline(bool force) /// `mountObservationThresholdMs` -- see its doc comment (CasServerRoot.h). const uint64_t stable_threshold_ms = mountObservationThresholdMs( ttl_ms, static_cast(store->poolConfig().mount_renew_period.count())); - computeHeartbeatFloor(op, layout, now_ms_fn(), mono_ms_fn(), stable_threshold_ms, mount_obs); + computeHeartbeatFloor(op, layout, now_ms_fn(), mono_ms_fn, stable_threshold_ms, mount_obs); /// Retired-in-snapshot: the rebuilt seal's `condemned_summary` must be TOTAL over gc_shards so a /// subsequent regular round reads graduation/carry decisions zero-I/O off it (and its `carryParentRefs` diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp index 10de8e758b3c..f3c428177d2a 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp @@ -932,6 +932,16 @@ String mountDoubleStartMessage(const String & srid, const std::optional= first_seen_mono_ms && sample_before_read_ms - first_seen_mono_ms >= threshold_ms; +} + uint64_t mountObservationThresholdMs(uint64_t ttl_ms, uint64_t cadence_ms) { return ttl_ms + ttl_ms / 20 + cadence_ms; @@ -957,15 +967,15 @@ MountClaimResult claimMountAwaitingExpiry( /// threshold, which passes the full renewal period into the same shared helper. const uint64_t threshold_ms = mountObservationThresholdMs(ttl_ms, poll); - std::optional observed; - uint64_t observed_since = 0; + std::optional watch; size_t restarts = 0; while (true) { - const bool threshold_met = observed && mono_ms_fn() - observed_since >= threshold_ms; + const bool threshold_met = watch && watch->stableFor(threshold_ms, mono_ms_fn()); MountClaimResult r = claimMount(op, l, srid, our_uuid, our_epoch, now_ms_fn(), ttl_ms, - threshold_met ? observed : std::nullopt, sink, /*unsafe_reclaim_authorization=*/{}); + threshold_met ? std::optional(watch->token) : std::nullopt, sink, + /*unsafe_reclaim_authorization=*/{}); if (r.kind != MountClaimResult::LiveDoubleStart) return r; @@ -997,14 +1007,13 @@ MountClaimResult claimMountAwaitingExpiry( r.body = decodeMountLease(got->bytes); } - if (!observed || *observed != *current_etag) + if (!watch || watch->token != *current_etag) { - if (observed && ++restarts > kMaxObservationRestarts) + if (watch && ++restarts > kMaxObservationRestarts) /// The incarnation kept changing across bounded restarts — the holder is genuinely alive /// (actively renewing), not a dead predecessor. Report it rather than waiting forever. return r; - observed = *current_etag; - observed_since = mono_ms_fn(); + watch = TokenWatch::sighted(*current_etag, mono_ms_fn()); if (on_wait_start && r.body) on_wait_start(*r.body, threshold_ms); LOG_INFO(getLogger("CasMountLease"), @@ -1018,10 +1027,11 @@ MountClaimResult claimMountAwaitingExpiry( } HeartbeatFloor computeHeartbeatFloor(CasOperation & op, const Layout & l, uint64_t now_ms, - uint64_t mono_now_ms, uint64_t stable_threshold_ms, + const std::function & mono_ms_fn, uint64_t stable_threshold_ms, MountObservationMap & obs) { HeartbeatFloor floor; + const uint64_t round_start_ms = mono_ms_fn(); /// `obs` is keyed by every srid this leader has EVER observed, but a /// srid removed from the LIST entirely (its `/mount` key gone -- e.g. `SYSTEM CAS @@ -1082,13 +1092,11 @@ HeartbeatFloor computeHeartbeatFloor(CasOperation & op, const Layout & l, uint64 /// raced against our own fence-out attempt) — (re)starts the observation window and /// counts as `live` this call. const auto it = obs.find(srid); - const bool stable = it != obs.end() && it->second.etag == observed->etag - && mono_now_ms - it->second.first_seen_mono_ms >= stable_threshold_ms; - - if (!stable) + const bool watched = it != obs.end() && it->second.token == observed->etag; + if (!watched || !it->second.stableFor(stable_threshold_ms, round_start_ms)) { - if (it == obs.end() || it->second.etag != observed->etag) - obs.insert_or_assign(srid, MountIncarnationObservation{observed->etag, mono_now_ms}); + if (!watched) + obs.insert_or_assign(srid, TokenWatch::sighted(observed->etag, round_start_ms)); ++floor.live; return std::nullopt; } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h index 1c1fd8928f97..6d221d279ee6 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h @@ -344,21 +344,25 @@ MountClaimResult claimMountAwaitingExpiry( const std::function & on_wait_start = {}, const CasEventSink & sink = {}); -/// One `server_root_id`'s cross-round incarnation-stability observation, -/// owned by the GC leader instance (`Cas::Gc::mount_obs`) and threaded through consecutive -/// `computeHeartbeatFloor` calls — one GC round is one observation tick. Mirrors -/// `claimMountAwaitingExpiry`'s observation loop, but at heartbeat-gate granularity rather than a -/// tight poll loop. -struct MountIncarnationObservation +/// One observed token and the instant it was first seen, on the observer's own monotonic clock. +struct TokenWatch { - Etag etag; + Etag token; uint64_t first_seen_mono_ms = 0; + + /// `sample_after_read_ms` is a clock sample taken after the read that returned `token_`. An earlier + /// sample would count the time the read took as time watched. + static TokenWatch sighted(Etag token_, uint64_t sample_after_read_ms); + + /// True when the token has been watched for at least `threshold_ms`. `sample_before_read_ms` is a + /// clock sample taken before the read that confirmed the token; false when it precedes the sighting. + bool stableFor(uint64_t threshold_ms, uint64_t sample_before_read_ms) const; }; /// Keyed by `server_root_id`. In-memory only: a fresh leader (after a steal, or a process restart) /// starts with an empty map, which only delays fencing an already-dead mount by one extra round while /// it (re)establishes the observation — safe (never fences early), never unsafe. -using MountObservationMap = std::map; +using MountObservationMap = std::map; /// GC heartbeat gate (GC round protocol step 1). Run by the GC leader at the top of a round: LIST /// `gc/server-roots/` (O(servers), single-digit counts), GET each mount body, and classify + fence out @@ -371,8 +375,7 @@ using MountObservationMap = std::map; /// terminated marker; /// - otherwise, observation-based liveness (the same /// principle `claimMountAwaitingExpiry` uses for a mount's OWN reopen, applied here to the GC's -/// fence-out): `obs` remembers, per srid, the incarnation last seen and the leader's OWN -/// monotonic clock reading (`mono_now_ms`) at the moment it first saw it. A body whose +/// fence-out): `obs` holds, per srid, a `TokenWatch` of the incarnation last seen. A body whose /// CURRENT incarnation differs from (or is absent from) `obs` is (re)started fresh — counted `live`, /// never fenced this call, regardless of what its stamped `expires_at_ms` claims (a bare /// wall-clock stamp is never trusted — see `claimMount`'s "certificate of death" doc). Only once @@ -386,7 +389,7 @@ using MountObservationMap = std::map; /// /// `now_ms` is WALL clock, used only for the audit/diagnostic log line — it never participates in the /// fence decision (mirrors `claimMountAwaitingExpiry`'s `now_ms_fn` vs `mono_ms_fn` split). -/// `mono_now_ms` is the OBSERVATION clock: the caller's OWN monotonic reading, never compared against +/// `mono_ms_fn` is the OBSERVATION clock: the caller's OWN monotonic clock, never compared against /// any other node's clock. `obs` is owned by the caller and threaded across consecutive calls (one GC /// leader instance, `Cas::Gc::mount_obs`) — a fresh leader starts with an empty map (safe: delays /// fencing one round, never fences early). @@ -407,7 +410,7 @@ struct HeartbeatFloor }; HeartbeatFloor computeHeartbeatFloor(CasOperation & op, const Layout & l, uint64_t now_ms, - uint64_t mono_now_ms, uint64_t stable_threshold_ms, + const std::function & mono_ms_fn, uint64_t stable_threshold_ms, MountObservationMap & obs); /// One `gc/server-roots//mount` slot whose holder is not provably finished with the prefix. diff --git a/src/Disks/tests/gtest_cas_mount.cpp b/src/Disks/tests/gtest_cas_mount.cpp index a9472994237f..23af9f244460 100644 --- a/src/Disks/tests/gtest_cas_mount.cpp +++ b/src/Disks/tests/gtest_cas_mount.cpp @@ -1501,7 +1501,7 @@ namespace /// Rev.6 §token-stability observation removed the wall clock from the fence DECISION; `kNowMs` below /// is threaded through only as `computeHeartbeatFloor`'s audit-only `now_ms`. constexpr uint64_t kNowMs = 1'000'000; -/// The fence-out threshold measured on the LEADER's OWN monotonic clock (`mono_now_ms`), independent +/// The fence-out threshold measured on the LEADER's OWN monotonic clock (`mono_ms_fn`), independent /// of any lease's stamped `expires_at_ms`. constexpr uint64_t kStableThresholdMs = 10'000; @@ -1539,6 +1539,39 @@ void renewMount(CasOperation & op, const Layout & l, const String & srid) mustCommit(op.replace(l.mountKey(srid), encodeMountLease(m), got->etag, Retry::standard()), "renewed mount " + srid); } + +/// An observer clock that does not move during a call: a sighting and the round start read the same value. +std::function frozenClock(uint64_t ms) +{ + return [ms] { return ms; }; +} +} + +TEST(CASTokenWatch, CountsFromAfterTheRead) +{ + auto b = std::make_shared(); + Ops ops(b); + const Etag t1 = std::get(ops.op.create("k", "v1", Retry::standard())).etag; + const Etag t2 = std::get(ops.op.replace("k", "v2", t1, Retry::standard())).etag; + constexpr uint64_t threshold_ms = 10'000; + + /// The read that returned `t1` ended at 5000: the watch counts from there. + const TokenWatch watch = TokenWatch::sighted(t1, /*sample_after_read_ms=*/ 5'000); + EXPECT_EQ(watch.token, t1); + EXPECT_EQ(watch.first_seen_mono_ms, 5'000u); + + EXPECT_FALSE(watch.stableFor(threshold_ms, 14'999)); + EXPECT_TRUE(watch.stableFor(threshold_ms, 15'000)); + + /// A sample older than the sighting proves nothing about how long the token held. + EXPECT_FALSE(watch.stableFor(threshold_ms, 4'000)); + EXPECT_FALSE(watch.stableFor(0, 4'000)); + + /// A changed token is a new watch, counted from its own read. + const TokenWatch renewed = TokenWatch::sighted(t2, 15'000); + EXPECT_NE(renewed.token, watch.token); + EXPECT_FALSE(renewed.stableFor(threshold_ms, 15'000)); + EXPECT_TRUE(renewed.stableFor(threshold_ms, 25'000)); } TEST(CASHeartbeatFloor, FirstSightNeverFencesEvenIfStampLooksExpired) @@ -1552,7 +1585,7 @@ TEST(CASHeartbeatFloor, FirstSightNeverFencesEvenIfStampLooksExpired) seedMount(ops.op, l, "s1", /*expires*/ 10, /*fenced*/ false, /*min_active_build_sequence*/ 0); MountObservationMap obs; - const HeartbeatFloor floor = computeHeartbeatFloor(ops.op, l, /*now_ms*/ kNowMs, /*mono_now_ms*/ 0, + const HeartbeatFloor floor = computeHeartbeatFloor(ops.op, l, /*now_ms*/ kNowMs, frozenClock(0), kStableThresholdMs, obs); EXPECT_EQ(floor.fenced_now, 0u); @@ -1569,14 +1602,14 @@ TEST(CASHeartbeatFloor, StableIncarnationPastThresholdIsFenced) seedMount(ops.op, l, "s1", /*expires*/ 10, /*fenced*/ false, /*min_active_build_sequence*/ 0); MountObservationMap obs; - const HeartbeatFloor floor_before = computeHeartbeatFloor(ops.op, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); + const HeartbeatFloor floor_before = computeHeartbeatFloor(ops.op, l, kNowMs, frozenClock(0), kStableThresholdMs, obs); EXPECT_EQ(floor_before.fenced_now, 0u); const MountLease before = decodeMountLease(ops.op.read(l.mountKey("s1"), Retry::standard())->bytes); /// No renewal in between: the SAME incarnation, observed since mono 0, is now stable for the full /// threshold on the leader's own clock. - const HeartbeatFloor floor2 = computeHeartbeatFloor(ops.op, l, kNowMs, /*mono*/ kStableThresholdMs, + const HeartbeatFloor floor2 = computeHeartbeatFloor(ops.op, l, kNowMs, frozenClock(kStableThresholdMs), kStableThresholdMs, obs); EXPECT_EQ(floor2.fenced_now, 1u); @@ -1594,20 +1627,20 @@ TEST(CASHeartbeatFloor, RenewalBetweenRoundsRestartsObservation) seedMount(ops.op, l, "s1", /*expires*/ 10, /*fenced*/ false, /*min_active_build_sequence*/ 0); MountObservationMap obs; - computeHeartbeatFloor(ops.op, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); + computeHeartbeatFloor(ops.op, l, kNowMs, frozenClock(0), kStableThresholdMs, obs); ASSERT_TRUE(obs.contains("s1")); - const Etag first_etag = obs.at("s1").etag; + const Etag first_etag = obs.at("s1").token; renewMount(ops.op, l, "s1"); const Etag renewed_etag = currentEtag(ops.op, l.mountKey("s1")); EXPECT_NE(renewed_etag, first_etag); - const HeartbeatFloor floor2 = computeHeartbeatFloor(ops.op, l, kNowMs, /*mono*/ kStableThresholdMs, + const HeartbeatFloor floor2 = computeHeartbeatFloor(ops.op, l, kNowMs, frozenClock(kStableThresholdMs), kStableThresholdMs, obs); EXPECT_EQ(floor2.fenced_now, 0u); ASSERT_TRUE(obs.contains("s1")); - EXPECT_EQ(obs.at("s1").etag, renewed_etag); + EXPECT_EQ(obs.at("s1").token, renewed_etag); EXPECT_EQ(obs.at("s1").first_seen_mono_ms, kStableThresholdMs); } @@ -1625,7 +1658,7 @@ TEST(CASHeartbeatFloor, UnseenSridPrunedFromObservationMap) seedMount(ops.op, l, "s2", /*expires*/ 10, /*fenced*/ false, /*min_active_build_sequence*/ 0); MountObservationMap obs; - computeHeartbeatFloor(ops.op, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); + computeHeartbeatFloor(ops.op, l, kNowMs, frozenClock(0), kStableThresholdMs, obs); ASSERT_TRUE(obs.contains("s1")); ASSERT_TRUE(obs.contains("s2")); @@ -1637,7 +1670,7 @@ TEST(CASHeartbeatFloor, UnseenSridPrunedFromObservationMap) renewMount(ops.op, l, "s1"); ASSERT_EQ(ops.op.removeCurrent(l.mountKey("s2"), Retry::standard()), Removal::Removed); - computeHeartbeatFloor(ops.op, l, kNowMs, /*mono*/ kStableThresholdMs, kStableThresholdMs, obs); + computeHeartbeatFloor(ops.op, l, kNowMs, frozenClock(kStableThresholdMs), kStableThresholdMs, obs); EXPECT_TRUE(obs.contains("s1")); EXPECT_FALSE(obs.contains("s2")) << "a srid removed from the LIST entirely must be pruned from obs, not linger forever"; @@ -1664,7 +1697,7 @@ TEST(CASHeartbeatFloor, ClassifiesAndFencesOut) MountObservationMap obs; /// Round 1 (mono 0): first sight of every non-terminal mount — nothing is fence-eligible yet. - const HeartbeatFloor floor_before = computeHeartbeatFloor(ops.op, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); + const HeartbeatFloor floor_before = computeHeartbeatFloor(ops.op, l, kNowMs, frozenClock(0), kStableThresholdMs, obs); EXPECT_EQ(floor_before.live, 3u); // s1, s2, s3: observation just started EXPECT_EQ(floor_before.terminated, 1u); // s5 EXPECT_EQ(floor_before.fenced_now, 0u); @@ -1681,7 +1714,7 @@ TEST(CASHeartbeatFloor, ClassifiesAndFencesOut) /// Round 2 (mono == threshold): s1/s2's renewed incarnations restart their observation (still /// live); s3's original incarnation has now held stable for the full threshold -> fenced. - const HeartbeatFloor floor2 = computeHeartbeatFloor(ops.op, l, kNowMs, /*mono*/ kStableThresholdMs, + const HeartbeatFloor floor2 = computeHeartbeatFloor(ops.op, l, kNowMs, frozenClock(kStableThresholdMs), kStableThresholdMs, obs); EXPECT_EQ(floor2.live, 2u); // s1, s2: renewed, observation restarted @@ -1759,14 +1792,14 @@ TEST(CASHeartbeatFloor, FenceOutLosesTheIncarnationRaceAndReclassifiesLive) MountObservationMap obs; /// Round 1: first sight, observation starts — never reaches the fence-out path (the race /// decorator stays armed for round 2). - const HeartbeatFloor floor_before = computeHeartbeatFloor(ops.op, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); + const HeartbeatFloor floor_before = computeHeartbeatFloor(ops.op, l, kNowMs, frozenClock(0), kStableThresholdMs, obs); EXPECT_EQ(floor_before.fenced_now, 0u); /// Round 2: the incarnation has been stable past threshold, so the function attempts the /// fence-out. The decorator renews concurrently under the real incarnation, the write is refused, /// and the re-decision reclassifies the slot as live (observation restarted on the new /// incarnation) — never fenced. - const HeartbeatFloor floor2 = computeHeartbeatFloor(ops.op, l, kNowMs, /*mono*/ kStableThresholdMs, + const HeartbeatFloor floor2 = computeHeartbeatFloor(ops.op, l, kNowMs, frozenClock(kStableThresholdMs), kStableThresholdMs, obs); EXPECT_EQ(floor2.fenced_now, 0u); @@ -1784,7 +1817,7 @@ TEST(CASHeartbeatFloor, EmptyPrefixYieldsNoLiveMounts) Ops ops(b); MountObservationMap obs; - const HeartbeatFloor floor = computeHeartbeatFloor(ops.op, l, kNowMs, /*mono*/ 0, kStableThresholdMs, obs); + const HeartbeatFloor floor = computeHeartbeatFloor(ops.op, l, kNowMs, frozenClock(0), kStableThresholdMs, obs); EXPECT_EQ(floor.live, 0u); EXPECT_EQ(floor.terminated, 0u); From 0ec8214d15b4488ab0142ba77715a05543211afa Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 01:35:19 +0200 Subject: [PATCH 03/31] Keep memory-limit exceptions off the CAS mount-lease thread The tracker still counts the renewal thread's allocations but no longer throws on it. An allocation failure outside the request used to end the thread; one inside it already was, and stays, a retried transient failure. Related: https://github.com/Altinity/ClickHouse/issues/2403 Co-Authored-By: Claude Sonnet 5.5 --- .../ContentAddressed/Pool/CasMountRuntime.cpp | 4 + src/Disks/tests/gtest_cas_pool.cpp | 138 ++++++++++++++++++ 2 files changed, 142 insertions(+) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index e3f52fd7957a..6e5b5080749a 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -1,6 +1,7 @@ #include #include #include +#include #include #include #include @@ -643,6 +644,9 @@ void CasMountRuntime::startBackgroundWorkers(std::chrono::milliseconds period) void CasMountRuntime::renewalLoop() { setThreadName(ThreadName::CAS_LEASE_RENEWER); + /// The tracker still counts this thread's allocations but never throws on it: a renewal failed by a + /// memory limit costs the mount, and an exception outside the request ends this thread. + LockMemoryExceptionInThread memory_exception_lock(VariableContext::Global); { std::unique_lock lock(driver_mutex); driver_cv.wait(lock, [this] { return worker_loops_released; }); diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp index 452511f5f761..788c418b3c2b 100644 --- a/src/Disks/tests/gtest_cas_pool.cpp +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -10,6 +10,10 @@ #include #include #include +#include +#include +#include +#include #include #include #include @@ -37,6 +41,7 @@ extern const int UNKNOWN_FORMAT_VERSION; extern const int FILE_DOESNT_EXIST; extern const int UNKNOWN_EXCEPTION; extern const int NETWORK_ERROR; +extern const int MEMORY_LIMIT_EXCEEDED; } namespace ProfileEvents @@ -2004,6 +2009,7 @@ class RuntimeRenewBackend final : public DB::Cas::tests::CountingBackend LandThenThrow, BlockThenDelegate, BlockThenThrow, + ThrowMemoryLimitExceeded, }; Fault fault = Fault::None; @@ -2029,6 +2035,8 @@ class RuntimeRenewBackend final : public DB::Cas::tests::CountingBackend throw DB::Exception(DB::ErrorCodes::CORRUPTED_DATA, "runtime renewal barrier is absent"); barrier->arriveAndWait(); } + if (current == Fault::ThrowMemoryLimitExceeded) + throw DB::Exception(DB::ErrorCodes::MEMORY_LIMIT_EXCEEDED, "injected memory limit exceeded on the renewal request"); if (current == Fault::ThrowBefore || current == Fault::BlockThenThrow) { if (before_throw) @@ -3568,6 +3576,136 @@ TEST(CASPoolRemount, TeardownJoinsBothWorkersBeforeRelease) std::numeric_limits::max()); } +namespace +{ + +/// Above the 4 MiB a thread batches before it reaches the tracker, below the 16 MiB at which an +/// allocation over an ignored limit also sends a trace. +constexpr Int64 kOverLimitAllocationBytes = 8 * 1024 * 1024; + +struct OverLimitAllocation +{ + /// What the tracker threw; empty when it did not throw. + String failure; + /// How much the global tracker grew by the allocation. + Int64 counted = 0; +}; + +/// One accounted allocation made while the global tracker is over its hard limit. The limit is restored +/// before this returns, whatever the allocation did. +OverLimitAllocation allocateOverTheGlobalLimit() +{ + DB::CurrentThread::flushUntrackedMemory(); + const Int64 saved_limit = total_memory_tracker.getHardLimit(); + SCOPE_EXIT({ total_memory_tracker.setHardLimit(saved_limit); }); + total_memory_tracker.setHardLimit(1); + + OverLimitAllocation result; + const Int64 before = total_memory_tracker.get(); + try + { + std::ignore = CurrentMemoryTracker::alloc(kOverLimitAllocationBytes); + } + catch (...) + { + result.failure = DB::getCurrentExceptionMessage(/*with_stacktrace=*/false); + return result; + } + result.counted = total_memory_tracker.get() - before; + std::ignore = CurrentMemoryTracker::free(kOverLimitAllocationBytes); + return result; +} + +} + +/// With the global tracker over its limit, an allocation on the lease thread before the request and +/// another while the result is consumed do not throw and are still counted; a memory-limit exception +/// raised inside the request is retried, does not trip the fence, and the thread goes on renewing. +TEST(CASMountRuntime, MemoryLimitDoesNotEndTheLeaseThread) +{ + auto backend = std::make_shared(); + const Layout layout("runtime-memory-limit"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + const Int64 hard_limit_before = total_memory_tracker.getHardLimit(); + + std::atomic admissions{0}; + OverLimitAllocation before_request; + DB::Cas::tests::ManualBarrier second_admission; + + std::optional while_consumed; + String outcome; + String attempts_sent; + String classification; + DB::Cas::tests::ManualBarrier reported; + CasEventSink sink = [&](CasEvent event) + { + if (event.type != CasEventType::WatermarkRenew || while_consumed) + return; + while_consumed = allocateOverTheGlobalLimit(); + outcome = event.outcome; + attempts_sent = event.detail["attempts_sent"]; + classification = event.detail["classification"]; + reported.arriveAndWait(); + }; + + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{ + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }, + .renewal_admitted_hook_for_test = [&] + { + const uint32_t admission = ++admissions; + if (admission == 1) + before_request = allocateOverTheGlobalLimit(); + else if (admission == 2) + second_admission.arriveAndWait(); + }}, + "test", sink, runtimeRenewBudget(), [] { return false; }); + /// Declared after the runtime so it runs first: a failed expectation must not leave the lease thread + /// parked on a barrier while the runtime's destructor joins it. + SCOPE_EXIT({ + reported.release(); + second_admission.release(); + }); + CasMountRuntime & runtime = *runtime_holder; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + runtime.armMountFence(uuid, 1, anchor + 1000); + const uint64_t leases_lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); + + backend->fault = RuntimeRenewBackend::Fault::ThrowMemoryLimitExceeded; + runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + + reported.waitUntilArrived(); + const String reported_outcome = outcome; + reported.release(); + ASSERT_EQ(reported_outcome, "recovered") << "a memory-limit exception inside the request must be retried"; + EXPECT_EQ(attempts_sent, "2"); + EXPECT_EQ(classification, "committed_after_retry"); + + /// The worker is admitted for its next renewal: the thread outlived the injected failure. + second_admission.waitUntilArrived(); + EXPECT_TRUE(runtime.mayMutate()); + EXPECT_EQ(runtime.lifecycle(), PoolLifecycle::Live); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(), leases_lost_before); + + EXPECT_EQ(before_request.failure, "") << "the allocation before the request threw"; + EXPECT_GE(before_request.counted, kOverLimitAllocationBytes); + ASSERT_TRUE(while_consumed.has_value()); + EXPECT_EQ(while_consumed->failure, "") << "the allocation while the result was consumed threw"; + EXPECT_GE(while_consumed->counted, kOverLimitAllocationBytes); + EXPECT_EQ(total_memory_tracker.getHardLimit(), hard_limit_before); + + second_admission.release(); + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); +} + TEST(CASPoolRemount, NaturalTerminalTransitionMakesBothPersistentWorkersSelfExit) { for (PoolLifecycle terminal : {PoolLifecycle::IdentityLost, PoolLifecycle::VanishedReplaced}) From 26de2d51e18e669a984094824867cbcabf678f34 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 01:39:46 +0200 Subject: [PATCH 04/31] Space engine reissues under `Retry::attempt_spacing_ms` A spaced policy reissues a `PUT` (ordinary, connect-hint and credential-refresh), a conflict and a read no sooner than a spacing draw after the failed request started; a request that took longer is reissued at once. The first-attempt fuse and the read after an unclear `PUT` stay immediate. Policies without spacing keep their pauses and their clock reads. Co-Authored-By: Claude Sonnet 5.5 --- .../ContentAddressed/Backend/CasRequests.cpp | 34 ++- .../ContentAddressed/Backend/CasRequests.h | 23 +- src/Disks/tests/gtest_cas_requests.cpp | 262 ++++++++++++++++++ 3 files changed, 302 insertions(+), 17 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.cpp index 898c0c8c1504..dde360a710c5 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.cpp @@ -453,6 +453,13 @@ bool CasOperation::fits(uint64_t needed_ms, const Retry::Bound & bound) const return needed_ms <= bound.deadline_ms - now; } +uint64_t CasOperation::reissuePause(const Retry & policy, uint64_t request_started_ms, uint64_t unspaced_ms) const +{ + if (!policy.attempt_spacing_ms) + return unspaced_ms; + return Retry::spacedPause(Retry::drawSpacing(*policy.attempt_spacing_ms), request_started_ms, owner.now_ms()); +} + bool CasOperation::refreshAndClassifyReadFault(const std::exception & e, bool & refresh_attempted, bool & refreshed) { if (const auto * db_e = dynamic_cast(&e); db_e && isDeterministicLocalFailure(db_e->code())) @@ -865,22 +872,25 @@ std::optional CasOperation::gatedPause(uint64_t pause_ms, uint32_t return std::nullopt; } -std::optional CasOperation::pauseAndReissue(WriteState & state, const Retry::Bound & bound) +std::optional CasOperation::pauseAndReissue(WriteState & state, const Retry & policy, const Retry::Bound & bound) { - return gatedPause(Retry::backoff(++state.reissues), 2, state, bound, detail::recordReissue, /*should_sleep=*/true); + const uint64_t pause_ms = reissuePause(policy, state.attempt_started_ms, Retry::backoff(++state.reissues)); + return gatedPause(pause_ms, 2, state, bound, detail::recordReissue, /*should_sleep=*/true); } -std::optional CasOperation::pauseForConflict(WriteState & state, const Retry::Bound & bound) +std::optional CasOperation::pauseForConflict(WriteState & state, const Retry & policy, const Retry::Bound & bound) { - return gatedPause(Retry::conflictBackoff(), 2, state, bound, detail::recordConflictPause, /*should_sleep=*/true); + const uint64_t pause_ms = reissuePause(policy, state.attempt_started_ms, Retry::conflictBackoff()); + return gatedPause(pause_ms, 2, state, bound, detail::recordConflictPause, /*should_sleep=*/true); } /// A flat pause before reissuing an attempt whose failure text named a failed connection. static constexpr uint64_t kConnectHintPauseMs = 50; -std::optional CasOperation::pauseFlat(WriteState & state, const Retry::Bound & bound) +std::optional CasOperation::pauseFlat(WriteState & state, const Retry & policy, const Retry::Bound & bound) { - return gatedPause(kConnectHintPauseMs, 2, state, bound, detail::recordReissue, /*should_sleep=*/true); + const uint64_t pause_ms = reissuePause(policy, state.attempt_started_ms, kConnectHintPauseMs); + return gatedPause(pause_ms, 2, state, bound, detail::recordReissue, /*should_sleep=*/true); } std::optional CasOperation::reissueAtOnce(WriteState & state, const Retry::Bound & bound) @@ -915,6 +925,8 @@ WriteResult CasOperation::writeLoop(const String & key, const String & bytes, co if (!fits(reservation, bound)) return gaveUp(GaveUp::Why::Deadline, sourceFor(bound), state); + if (policy.attempt_spacing_ms) + state.attempt_started_ms = owner.now_ms(); detail::recordAttempt(); ++state.attempts_sent; state.sent_any = true; @@ -1019,7 +1031,7 @@ WriteResult CasOperation::writeLoop(const String & key, const String & bytes, co /// inner write is unresolved either. Re-send it under the credentials the refresh installed. if (refresh_owns_reissue) { - if (auto given_up = pauseAndReissue(state, bound)) + if (auto given_up = pauseAndReissue(state, policy, bound)) return *given_up; continue; } @@ -1031,7 +1043,7 @@ WriteResult CasOperation::writeLoop(const String & key, const String & bytes, co /// the read below settles it. if (connect_hint && !policy.single_attempt) { - if (auto given_up = pauseFlat(state, bound)) + if (auto given_up = pauseFlat(state, policy, bound)) return *given_up; continue; } @@ -1087,7 +1099,7 @@ WriteResult CasOperation::writeLoop(const String & key, const String & bytes, co return *given_up; continue; } - if (auto given_up = pauseAndReissue(state, bound)) + if (auto given_up = pauseAndReissue(state, policy, bound)) return *given_up; } } @@ -1143,7 +1155,7 @@ WriteResult CasOperation::readModifyWrite(const String & key, const DecideOnObje /// A clean lost race is settled: the resolve read holds the fresh object and the next /// iteration decides on it. Only a conflict that settled a transport fault is paced by the /// growing schedule. - if (auto given_up = state.any_ambiguous ? pauseAndReissue(state, bound) : pauseForConflict(state, bound)) + if (auto given_up = state.any_ambiguous ? pauseAndReissue(state, policy, bound) : pauseForConflict(state, policy, bound)) return *given_up; /// Only when the resolve settled nothing is a fresh read owed; otherwise `current` already is @@ -1200,7 +1212,7 @@ WriteResult CasOperation::readModifyWriteOnPresence(const String & key, const De /// A clean lost race is settled: the resolve read holds the fresh object and the next /// iteration decides on it. Only a conflict that settled a transport fault is paced by the /// growing schedule. - if (auto given_up = state.any_ambiguous ? pauseAndReissue(state, bound) : pauseForConflict(state, bound)) + if (auto given_up = state.any_ambiguous ? pauseAndReissue(state, policy, bound) : pauseForConflict(state, policy, bound)) return *given_up; if (std::holds_alternative(state.last_seen)) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.h index dd79f1a9db4b..055bfc55952d 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.h @@ -340,6 +340,8 @@ class CasOperation Observation last_seen = NotObserved{}; uint32_t reissues = 0; bool refresh_attempted = false; + /// When the latest attempt was sent; sampled only under `Retry::attempt_spacing_ms`. + uint64_t attempt_started_ms = 0; }; /// Why a read-class request stopped without an answer. Every give-up below throws the same @@ -358,7 +360,8 @@ class CasOperation std::optional stop; }; - /// One read-class request under the policy: admission, attempt, classification, jittered reissue. + /// One read-class request under the policy: admission, attempt, classification, then a jittered + /// reissue, or a spaced one under `Retry::attempt_spacing_ms`. /// Returns whatever `once` returns, or throws -- the read surface reports failure by exception. template auto readLoop(std::string_view verb, const String & subject, const Retry & policy, @@ -405,20 +408,25 @@ class CasOperation std::optional gatedPause(uint64_t pause_ms, uint32_t envelopes, WriteState & state, const Retry::Bound & bound, void (*record)(), bool should_sleep); /// Admission, then the jittered sleep. A value means the call ended during it; nullopt means the - /// caller may send another attempt. - std::optional pauseAndReissue(WriteState & state, const Retry::Bound & bound); + /// caller may send another attempt. This and the two siblings below sleep the spaced pause instead + /// under `Retry::attempt_spacing_ms`. + std::optional pauseAndReissue(WriteState & state, const Retry & policy, const Retry::Bound & bound); /// The sibling for a clean lost race: the same admission and the same reservation, a flat /// `Retry::conflictBackoff` sleep, and `state.reissues` untouched, so a transport fault that follows /// starts its own schedule at the beginning. - std::optional pauseForConflict(WriteState & state, const Retry::Bound & bound); + std::optional pauseForConflict(WriteState & state, const Retry & policy, const Retry::Bound & bound); /// The sibling for a failure text that named a failed connection. The same admission and the same /// reservation, a flat `kConnectHintPauseMs` sleep, and `state.reissues` untouched. - std::optional pauseFlat(WriteState & state, const Retry::Bound & bound); + std::optional pauseFlat(WriteState & state, const Retry & policy, const Retry::Bound & bound); /// The sibling for a first-attempt fuse timeout: the same admission and the same reservation, NO /// sleep at all, and `state.reissues` untouched -- the fuse is a connection-quality answer about a /// fresh connection, not a store fault, so nothing here is paced against it. std::optional reissueAtOnce(WriteState & state, const Retry::Bound & bound); + /// The pause before a reissue: `unspaced_ms`, or under `Retry::attempt_spacing_ms` what is left of + /// a fresh draw since `request_started_ms`. + uint64_t reissuePause(const Retry & policy, uint64_t request_started_ms, uint64_t unspaced_ms) const; + /// `sleep_ms` plus `envelopes` attempt reservations, saturating. uint64_t reservedFor(uint64_t sleep_ms, uint32_t envelopes) const; /// Is there room to START something needing `needed_ms` before the bound? The guarantee is on the @@ -452,6 +460,7 @@ auto CasOperation::readLoop(std::string_view verb, const String & subject, const const Retry::Bound & bound, Fn && once) { bool refresh_attempted = false; + uint64_t attempt_started_ms = 0; /// Two counters, deliberately kept separate: `attempt_no` is the PHYSICAL attempt count handed to /// the transport (so a reissue is seen as attempt >= 2); `ordinary_reissues` is the /// exponential-backoff index. They advance together on an ordinary failure, but the first-attempt @@ -469,6 +478,8 @@ auto CasOperation::readLoop(std::string_view verb, const String & subject, const if (!fits(reservation, bound)) giveUpReadDeadline(verb, subject, bound, attempt_no - 1); + if (policy.attempt_spacing_ms) + attempt_started_ms = owner.now_ms(); detail::recordAttempt(); try { @@ -509,7 +520,7 @@ auto CasOperation::readLoop(std::string_view verb, const String & subject, const } } - const uint64_t pause_ms = Retry::backoff(++ordinary_reissues); + const uint64_t pause_ms = reissuePause(policy, attempt_started_ms, Retry::backoff(++ordinary_reissues)); const uint64_t needed = reservedFor(pause_ms, 1); switch (gate(needed)) { diff --git a/src/Disks/tests/gtest_cas_requests.cpp b/src/Disks/tests/gtest_cas_requests.cpp index 92cdbc7ea489..328368f0786a 100644 --- a/src/Disks/tests/gtest_cas_requests.cpp +++ b/src/Disks/tests/gtest_cas_requests.cpp @@ -3063,8 +3063,23 @@ class TimedFaultBackend : public InMemoryBackend using Verb = TimedFaultBackend::Verb; +/// A transport failure that is neither a connect-failure hint nor a first-attempt fuse. +std::exception_ptr ordinaryFault() +{ + return std::make_exception_ptr(DB::S3Exception( + "Poco::Exception. Code: 1000, e.code() = 104, Connection reset by peer", Aws::S3::S3Errors::NETWORK_CONNECTION)); +} + constexpr uint64_t kSpacingMs = 1'000; +/// Creates `k` and forgets the requests that did it. +Etag seedK(TimedFaultBackend & backend, CasOperation & op) +{ + const Etag seen = *orThrow(op.create("k", "v1", Retry::standard()), "seed"); + backend.sent.clear(); + return seen; +} + } /// Spacing is measured on the injected clock, so a bind that wrapped at the top of the range would @@ -3096,6 +3111,253 @@ TEST(CASRequestsSpacing, UntilDefinitiveRefusesNothingBeforeTheEndOfTheClock) EXPECT_LE(puts.back(), largest - 14'000) << "every PUT was admitted with room for its two envelopes"; } +/// Spec test 10, first case: a hinted PUT is reissued without a read, every reissue starts one draw +/// after the previous one started, and the call is still retrying after 30 s. +TEST(CASRequestsSpacing, FastConnectFailuresAreSpacedFromTheirStart) +{ + FakeClock clock; + auto backend = std::make_shared(clock); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + const Etag seen = seedK(*backend, op); + const uint64_t t0 = clock.now; + backend->fault = [t0](Verb verb, size_t nth, uint64_t started) -> std::exception_ptr + { + if (verb != Verb::Put) + return nullptr; + if (nth == 1) + return fuseTimeout(); + return started < t0 + 30'000 ? connectHint() : nullptr; + }; + + const WriteResult result = op.replace("k", "v2", seen, Retry::untilDefinitive(kSpacingMs)); + + const auto * committed = std::get_if(&result); + ASSERT_NE(committed, nullptr); + const auto puts = backend->startsOf(Verb::Put); + ASSERT_GE(puts.size(), 3u); + EXPECT_EQ(committed->attempts_sent, puts.size()); + EXPECT_GE(puts.back(), t0 + 30'000) << "the call must still be retrying after 30 s"; + /// The fuse: its settle read and its reissue both follow at once. + EXPECT_EQ(backend->sent[1].verb, Verb::Get); + EXPECT_EQ(backend->sent[1].started_ms, t0); + EXPECT_EQ(puts[1], t0); + EXPECT_EQ(backend->startsOf(Verb::Get).size(), 1u) << "a hinted PUT is reissued without a read"; + uint64_t min_gap = std::numeric_limits::max(); + uint64_t max_gap = 0; + for (size_t i = 2; i < puts.size(); ++i) + { + min_gap = std::min(min_gap, puts[i] - puts[i - 1]); + max_gap = std::max(max_gap, puts[i] - puts[i - 1]); + } + EXPECT_GE(min_gap, 800u); + EXPECT_LE(max_gap, 1'200u); + EXPECT_EQ(clock.sleeps.size(), puts.size() - 2) << "one sleep per hinted reissue, none for the fuse"; +} + +/// Spec test 10, second case: the read after an unclear PUT is sent at once, the read's own fuse is +/// reissued at once, and the retries of the read and of the PUT are spaced. +TEST(CASRequestsSpacing, TheResolveReadFollowsAtOnceAndItsRetriesAreSpaced) +{ + FakeClock clock; + auto backend = std::make_shared(clock); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + const Etag seen = seedK(*backend, op); + const uint64_t t0 = clock.now; + backend->fault = [t0](Verb verb, size_t nth, uint64_t started) -> std::exception_ptr + { + if (verb == Verb::Put) + return started < t0 + 30'000 ? ordinaryFault() : nullptr; + /// Every resolve read: a fuse, an ordinary failure, then the answer. + switch (nth % 3) + { + case 1: return fuseTimeout(); + case 2: return ordinaryFault(); + default: return nullptr; + } + }; + + const WriteResult result = op.replace("k", "v2", seen, Retry::untilDefinitive(kSpacingMs)); + + ASSERT_TRUE(std::holds_alternative(result)); + const auto & sent = backend->sent; + const auto puts = backend->startsOf(Verb::Put); + ASSERT_GE(puts.size(), 2u); + EXPECT_GE(puts.back(), t0 + 30'000) << "the call must still be retrying after 30 s"; + ASSERT_EQ(sent.size(), puts.size() + 3 * (puts.size() - 1)) << "three reads per failed PUT"; + bool at_once = true; + uint64_t min_spaced = std::numeric_limits::max(); + uint64_t max_spaced = 0; + for (size_t p = 0; p + 1 < puts.size(); ++p) + { + const size_t i = 4 * p; /// this PUT, then its three reads, then the next PUT + ASSERT_EQ(sent[i].verb, Verb::Put); + ASSERT_EQ(sent[i + 1].verb, Verb::Get); + ASSERT_EQ(sent[i + 2].verb, Verb::Get); + ASSERT_EQ(sent[i + 3].verb, Verb::Get); + at_once = at_once && sent[i + 1].started_ms == sent[i].started_ms + && sent[i + 2].started_ms == sent[i + 1].started_ms; + for (const uint64_t gap : {sent[i + 3].started_ms - sent[i + 2].started_ms, sent[i + 4].started_ms - sent[i].started_ms}) + { + min_spaced = std::min(min_spaced, gap); + max_spaced = std::max(max_spaced, gap); + } + } + EXPECT_TRUE(at_once) << "the read follows its PUT at once, and the read's fuse is reissued at once"; + EXPECT_GE(min_spaced, 800u); + EXPECT_LE(max_spaced, 1'200u); +} + +/// Spec test 10, third case: a request that takes 5 s on the injected clock is followed by the next one +/// with no wait, for the read and for the PUT; the PUT's own fuse reissue does not sleep at all. +TEST(CASRequestsSpacing, ARequestThatTookLongerThanTheSpacingIsRetriedAtOnce) +{ + FakeClock clock; + auto backend = std::make_shared(clock); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + const Etag seen = seedK(*backend, op); + backend->duration_ms = 5'000; + const uint64_t t0 = clock.now; + backend->fault = [t0](Verb verb, size_t nth, uint64_t started) -> std::exception_ptr + { + if (verb == Verb::Put) + { + if (nth == 1) + return fuseTimeout(); + return started < t0 + 60'000 ? ordinaryFault() : nullptr; + } + return nth % 2 == 1 ? ordinaryFault() : nullptr; + }; + + const WriteResult result = op.replace("k", "v2", seen, Retry::untilDefinitive(kSpacingMs)); + + ASSERT_TRUE(std::holds_alternative(result)); + const auto & sent = backend->sent; + bool back_to_back = true; + for (size_t i = 1; i < sent.size(); ++i) + back_to_back = back_to_back && sent[i].started_ms == sent[i - 1].started_ms + 5'000; + EXPECT_TRUE(back_to_back) << "every request starts when the previous one ends"; + EXPECT_EQ(backend->startsOf(Verb::Put), + (std::vector{t0, t0 + 15'000, t0 + 30'000, t0 + 45'000, t0 + 60'000})); + /// Four read retries and three PUT reissues, each a zero-length sleep; the fuse reissue none. + EXPECT_EQ(clock.sleeps, std::vector(7, 0)); +} + +/// The worst case for request rate: every PUT is unclear and every resolve read hits its fuse and then +/// a failure before it answers. The fuse and the read after a PUT are immediate, so the bound comes from +/// the spacing alone: at most two PUT periods of 4 requests start in any second. +TEST(CASRequestsSpacing, AnUnclearPutWithAFailingReadStaysUnderEightRequestsPerSecond) +{ + FakeClock clock; + auto backend = std::make_shared(clock); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + const Etag seen = seedK(*backend, op); + const uint64_t t0 = clock.now; + constexpr uint64_t outage_ms = 60'000; + backend->fault = [t0](Verb verb, size_t nth, uint64_t started) -> std::exception_ptr + { + if (verb == Verb::Put) + return started < t0 + outage_ms ? fuseTimeout() : nullptr; /// a fuse on attempt 1, an ordinary timeout after + switch (nth % 3) + { + case 1: return fuseTimeout(); + case 2: return ordinaryFault(); + default: return nullptr; + } + }; + + const WriteResult result = op.replace("k", "v2", seen, Retry::untilDefinitive(kSpacingMs)); + + ASSERT_TRUE(std::holds_alternative(result)); + const auto & sent = backend->sent; + ASSERT_FALSE(sent.empty()); + EXPECT_GE(sent.back().started_ms, t0 + outage_ms); + std::vector per_second((sent.back().started_ms - t0) / 1'000 + 1, 0); + for (const auto & request : sent) + ++per_second[(request.started_ms - t0) / 1'000]; + EXPECT_LE(*std::max_element(per_second.begin(), per_second.end()), 8u); + EXPECT_LE(sent.size(), 4 * (outage_ms / 800 + 2)) << "at most 4 requests per 800 ms period"; +} + +TEST(CASRequestsSpacing, ACredentialRefreshReissueIsSpaced) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->setRefreshCredentialsResult(true); + backend->failNextWriteWith("k", s3Error(Aws::S3::S3Errors::INVALID_CLIENT_TOKEN_ID, "ExpiredToken")); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + + const WriteResult result = op.create("k", "v", Retry::untilDefinitive(kSpacingMs)); + + const auto * committed = std::get_if(&result); + ASSERT_NE(committed, nullptr); + EXPECT_EQ(committed->attempts_sent, 2u); + EXPECT_EQ(backend->getTotal(), 0u) << "the credential answer owes no read"; + EXPECT_EQ(backend->refreshCredentialsCalls(), 1u); + ASSERT_EQ(clock.sleeps.size(), 1u); + EXPECT_GE(clock.sleeps[0], 800u); + EXPECT_LE(clock.sleeps[0], 1'200u); +} + +/// A spaced policy has no window, so the spacing is the only bound on a conflict loop's rate too. +TEST(CASRequestsSpacing, CleanConflictPausesAreSpacedUnderASpacedPolicy) +{ + FakeClock clock; + auto backend = std::make_shared(); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + (void)orThrow(op.create("k", "v", Retry::standard()), "seed"); + constexpr int K = 3; + RaceMaker races(backend, clock, "k", K, /*ambiguous=*/false); + + const WriteResult result = op.readModifyWrite("k", appendX(), Retry::untilDefinitive(kSpacingMs)); + + ASSERT_TRUE(std::holds_alternative(result)); + ASSERT_EQ(clock.sleeps.size(), static_cast(K)); + for (const uint64_t pause : clock.sleeps) + { + EXPECT_GE(pause, 800u); + EXPECT_LE(pause, 1'200u); + } +} + +/// Spacing must not leak into a policy without it: per failed PUT, the read's `backoff(1)` and then the +/// PUT's growing `backoff(n)`. +TEST(CASRequestsSpacing, AnUnspacedPolicyKeepsTheGrowingBackoff) +{ + FakeClock clock; + auto backend = std::make_shared(clock); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + const Etag seen = seedK(*backend, op); + backend->fault = [](Verb verb, size_t nth, uint64_t) -> std::exception_ptr + { + if (verb == Verb::Put) + return nth <= 3 ? ordinaryFault() : nullptr; + return nth % 2 == 1 ? ordinaryFault() : nullptr; + }; + + const WriteResult result = op.replace("k", "v2", seen, Retry::standard()); + + const auto * committed = std::get_if(&result); + ASSERT_NE(committed, nullptr); + EXPECT_EQ(committed->attempts_sent, 4u); + const std::vector ceilings{200, 200, 200, 400, 200, 800}; + ASSERT_EQ(clock.sleeps.size(), ceilings.size()); + uint64_t total = 0; + for (size_t i = 0; i < ceilings.size(); ++i) + { + EXPECT_LE(clock.sleeps[i], ceilings[i]) << "sleep " << i; + total += clock.sleeps[i]; + } + /// Six full-jitter draws that are all zero have a probability below 1e-13; a zero-wait leak does not. + EXPECT_GT(total, 0u); +} + #endif TEST(CASRequestBudget, EnvelopeIsValidatedNotTheBareAttempt) From 38c5ac5fb659290b655568894dd8c271abcfe796 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 01:45:06 +0200 Subject: [PATCH 05/31] Date a GC token sighting after the read that produced it `computeHeartbeatFloor` dated a first sighting with the clock sample taken before the `LIST`, so a slow walk counted as time the token was watched. A second round could then fence a slot whose holder still had authority. The sighting is now sampled after the read, in the first decision and in the re-decision after a refused fence-out; the stability test keeps the round-start sample. Co-Authored-By: Claude Sonnet 5.5 --- .../ContentAddressed/Pool/CasServerRoot.cpp | 4 +- .../ContentAddressed/Pool/CasServerRoot.h | 3 +- src/Disks/tests/gtest_cas_mount.cpp | 145 ++++++++++++++++++ 3 files changed, 150 insertions(+), 2 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp index f3c428177d2a..b2f4dc011871 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp @@ -1031,6 +1031,7 @@ HeartbeatFloor computeHeartbeatFloor(CasOperation & op, const Layout & l, uint64 MountObservationMap & obs) { HeartbeatFloor floor; + /// Taken before any read, so it may confirm that a token held, but never date a sighting. const uint64_t round_start_ms = mono_ms_fn(); /// `obs` is keyed by every srid this leader has EVER observed, but a @@ -1095,8 +1096,9 @@ HeartbeatFloor computeHeartbeatFloor(CasOperation & op, const Layout & l, uint64 const bool watched = it != obs.end() && it->second.token == observed->etag; if (!watched || !it->second.stableFor(stable_threshold_ms, round_start_ms)) { + /// A sample from before this read would count the walk to the slot as time watched. if (!watched) - obs.insert_or_assign(srid, TokenWatch::sighted(observed->etag, round_start_ms)); + obs.insert_or_assign(srid, TokenWatch::sighted(observed->etag, mono_ms_fn())); ++floor.live; return std::nullopt; } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h index 6d221d279ee6..e3ef3ee434da 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h @@ -390,7 +390,8 @@ using MountObservationMap = std::map; /// `now_ms` is WALL clock, used only for the audit/diagnostic log line — it never participates in the /// fence decision (mirrors `claimMountAwaitingExpiry`'s `now_ms_fn` vs `mono_ms_fn` split). /// `mono_ms_fn` is the OBSERVATION clock: the caller's OWN monotonic clock, never compared against -/// any other node's clock. `obs` is owned by the caller and threaded across consecutive calls (one GC +/// any other node's clock. It is sampled once at entry for the stability test and again after the read +/// for every sighting. `obs` is owned by the caller and threaded across consecutive calls (one GC /// leader instance, `Cas::Gc::mount_obs`) — a fresh leader starts with an empty map (safe: delays /// fencing one round, never fences early). /// diff --git a/src/Disks/tests/gtest_cas_mount.cpp b/src/Disks/tests/gtest_cas_mount.cpp index 23af9f244460..f4a7b671b63f 100644 --- a/src/Disks/tests/gtest_cas_mount.cpp +++ b/src/Disks/tests/gtest_cas_mount.cpp @@ -1545,6 +1545,135 @@ std::function frozenClock(uint64_t ms) { return [ms] { return ms; }; } + +/// Moves the observer clock on every read of a mount slot, so a round spends time between its first +/// clock sample and the read it decides on. With `renew_before_next_fence` set, the holder renews once +/// just before the next guarded write of a mount slot lands, so a fence-out is refused and decided +/// again on the holder's new token. +class SightingClockBackend : public InMemoryBackend +{ +public: + explicit SightingClockBackend(uint64_t & mono_) : mono(mono_) {} + + uint64_t read_cost_ms = 0; + bool renew_before_next_fence = false; + + std::optional read(const String & key, TransportAccess & access) override + { + if (key.ends_with("/mount")) + mono += read_cost_ms; + return InMemoryBackend::read(key, access); + } + + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, + TransportAccess & access) override + { + if (renew_before_next_fence && expected_value && key.ends_with("/mount")) + { + renew_before_next_fence = false; + const auto got = InMemoryBackend::read(key, access); + MountLease m = decodeMountLease(got->bytes); + m.seq += 1; + EXPECT_TRUE(InMemoryBackend::write(key, encodeMountLease(m), got->value, access).has_value()); + } + return InMemoryBackend::write(key, bytes, expected_value, access); + } + +private: + uint64_t & mono; +}; + +constexpr uint64_t kWalkMs = 4'000; + +void firstDecisionCountsFromAfterTheRead() +{ + uint64_t mono = 0; + auto b = std::make_shared(mono); + Layout l("p"); + Ops ops(b); + seedMount(ops.op, l, "s1", /*expires*/ 10, /*fenced*/ false, /*min_active_build_sequence*/ 0); + const auto clock = [&mono] { return mono; }; + MountObservationMap obs; + + /// The round starts at 0 and reaches the slot at `kWalkMs`. + b->read_cost_ms = kWalkMs; + ASSERT_EQ(computeHeartbeatFloor(ops.op, l, kNowMs, clock, kStableThresholdMs, obs).fenced_now, 0u); + ASSERT_TRUE(obs.contains("s1")); + EXPECT_EQ(obs.at("s1").first_seen_mono_ms, kWalkMs); + + /// One threshold after the first round started, less than one after its read. + b->read_cost_ms = 0; + mono = kStableThresholdMs; + const HeartbeatFloor early = computeHeartbeatFloor(ops.op, l, kNowMs, clock, kStableThresholdMs, obs); + ASSERT_EQ(early.fenced_now, 0u) << "fenced a token watched for less than the threshold"; + EXPECT_EQ(early.live, 1u); + + mono = kWalkMs + kStableThresholdMs; + EXPECT_EQ(computeHeartbeatFloor(ops.op, l, kNowMs, clock, kStableThresholdMs, obs).fenced_now, 1u); +} + +void reDecisionCountsFromAfterItsRead() +{ + uint64_t mono = 0; + auto b = std::make_shared(mono); + Layout l("p"); + Ops ops(b); + seedMount(ops.op, l, "s1", /*expires*/ 10, /*fenced*/ false, /*min_active_build_sequence*/ 0); + const auto clock = [&mono] { return mono; }; + MountObservationMap obs; + + computeHeartbeatFloor(ops.op, l, kNowMs, clock, kStableThresholdMs, obs); + ASSERT_TRUE(obs.contains("s1")); + const Etag first = obs.at("s1").token; + + /// Stable at this round's start, so it tries the fence-out. The holder renews first, the write is + /// refused, and the round decides again on the new token after reading it. + mono = kStableThresholdMs; + b->read_cost_ms = kWalkMs; + b->renew_before_next_fence = true; + const HeartbeatFloor refused = computeHeartbeatFloor(ops.op, l, kNowMs, clock, kStableThresholdMs, obs); + ASSERT_FALSE(b->renew_before_next_fence) << "the round never attempted the fence-out"; + ASSERT_EQ(refused.fenced_now, 0u); + ASSERT_EQ(refused.live, 1u); + ASSERT_NE(obs.at("s1").token, first); + + /// Nothing reads after the re-decision, so the clock still holds the value of its last read. + const uint64_t reread_at = mono; + ASSERT_GT(reread_at, kStableThresholdMs); + EXPECT_EQ(obs.at("s1").first_seen_mono_ms, reread_at); + + /// One threshold after that round started, less than one after its re-read. + b->read_cost_ms = 0; + mono = kStableThresholdMs + kStableThresholdMs; + ASSERT_EQ(computeHeartbeatFloor(ops.op, l, kNowMs, clock, kStableThresholdMs, obs).fenced_now, 0u) + << "fenced the renewed token before it was watched for the threshold"; + + mono = reread_at + kStableThresholdMs; + EXPECT_EQ(computeHeartbeatFloor(ops.op, l, kNowMs, clock, kStableThresholdMs, obs).fenced_now, 1u); +} + +void slowWalkDoesNotMoveTheStabilitySample() +{ + uint64_t mono = 0; + auto b = std::make_shared(mono); + Layout l("p"); + Ops ops(b); + seedMount(ops.op, l, "s1", /*expires*/ 10, /*fenced*/ false, /*min_active_build_sequence*/ 0); + const auto clock = [&mono] { return mono; }; + MountObservationMap obs; + + computeHeartbeatFloor(ops.op, l, kNowMs, clock, kStableThresholdMs, obs); + ASSERT_EQ(obs.at("s1").first_seen_mono_ms, 0u); + + /// The round starts one millisecond short of the threshold; its walk passes the threshold many times. + mono = kStableThresholdMs - 1; + b->read_cost_ms = 10 * kStableThresholdMs; + const HeartbeatFloor slow = computeHeartbeatFloor(ops.op, l, kNowMs, clock, kStableThresholdMs, obs); + EXPECT_EQ(slow.fenced_now, 0u); + EXPECT_EQ(slow.live, 1u); + EXPECT_EQ(obs.at("s1").first_seen_mono_ms, 0u) << "an unchanged token keeps its first sighting"; +} } TEST(CASTokenWatch, CountsFromAfterTheRead) @@ -1810,6 +1939,22 @@ TEST(CASHeartbeatFloor, FenceOutLosesTheIncarnationRaceAndReclassifiesLive) EXPECT_FALSE(decodeMountLease(after->bytes).gc_fenced); } +TEST(CASHeartbeat, GcCountsASightingFromAfterItsRead) +{ + { + SCOPED_TRACE("first decision"); + firstDecisionCountsFromAfterTheRead(); + } + { + SCOPED_TRACE("re-decision after a refused fence-out"); + reDecisionCountsFromAfterItsRead(); + } + { + SCOPED_TRACE("slow walk"); + slowWalkDoesNotMoveTheStabilitySample(); + } +} + TEST(CASHeartbeatFloor, EmptyPrefixYieldsNoLiveMounts) { auto b = std::make_shared(); From 817426de55cb1fed8c128507bf530019c2d25115 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 01:50:02 +0200 Subject: [PATCH 06/31] Run CAS conditional PUTs on the calling thread with a body-sized buffer A conditional PUT ran on the remote-FS writer pool, outside the memory guard of the mount-lease thread, with a 1 MiB buffer for an object of a few hundred bytes. It now uploads inline and its buffer starts at the body size, one byte for an empty body. Related: https://github.com/Altinity/ClickHouse/issues/2403 Co-Authored-By: Claude Sonnet 5.5 --- .../Backend/CasObjectStorageBackend.cpp | 7 +- .../Backend/CasObjectStorageBackend.h | 4 +- .../gtest_cas_s3_single_attempt_client.cpp | 114 +++++++++++++++++- 3 files changed, 120 insertions(+), 5 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.cpp index 0589929d2912..976866143026 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.cpp @@ -309,8 +309,10 @@ static detail::ConditionalWriteOutcome finalizeConditionalWriteInstrumented(Writ std::expected ObjectStorageBackend::nativeConditionalPut( const String & key, const String & bytes, const WriteSettings & ws) { + /// Sized to the body rather than 1 MiB; `WriteBufferFromS3` grows it if needed. At least one byte, + /// because `WriteBuffer::write` requires a non-empty buffer even for an empty body. auto buf = object_storage->writeObject( - StoredObject(key), WriteMode::Rewrite, /*attributes=*/std::nullopt, DBMS_DEFAULT_BUFFER_SIZE, ws); + StoredObject(key), WriteMode::Rewrite, /*attributes=*/std::nullopt, std::max(1, bytes.size()), ws); buf->write(bytes.data(), bytes.size()); if (finalizeConditionalWriteInstrumented(*buf) == detail::ConditionalWriteOutcome::PreconditionLost) return std::unexpected(RawConflict{}); @@ -809,6 +811,9 @@ WriteSettings ObjectStorageBackend::conditionalWriteSettings(size_t attempt_no) if (native_token_type == Dialect::Generation) ws.s3_force_single_part_upload = true; ws.s3_check_objects_after_upload_override = false; + /// Upload on the calling thread: the caller waits for the write anyway, and a pool thread would run + /// outside the memory guard the mount-lease thread holds. + ws.s3_allow_parallel_part_upload = false; /// Exactly one attempt at the WriteBufferFromS3 layer too: makeSinglepartUpload/ /// completeMultipartUpload run their OWN retry loop above the S3 client, reissuing the identical /// (conditional!) request on NO_SUCH_KEY — a client-level override alone does not bound it. Plain diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.h index f9f51396efbb..6281841ff087 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasObjectStorageBackend.h @@ -196,8 +196,8 @@ class ObjectStorageBackend final : public Backend /// Settings for a Native COMPARE/CREATE write (create-if-absent, compare-and-set): mark the request /// conditional, make exactly one attempt at every retry layer, skip the racy post-upload - /// existence/size check, and force a single PUT on generation stores because GCS does not - /// enforce the condition on multipart completion. `attempt_no` is the engine's own 1-based + /// existence/size check, upload on the calling thread, and force a single PUT on generation stores + /// because GCS does not enforce the condition on multipart completion. `attempt_no` is the engine's own 1-based /// physical-attempt count (see `TransportAccess::attemptNo`), carried into /// `object_storage_attempt_number` so the HTTP client sees a reissue as attempt >= 2. WriteSettings conditionalWriteSettings(size_t attempt_no) const; diff --git a/src/Disks/tests/gtest_cas_s3_single_attempt_client.cpp b/src/Disks/tests/gtest_cas_s3_single_attempt_client.cpp index 30ee3192bc54..27aa62612f2f 100644 --- a/src/Disks/tests/gtest_cas_s3_single_attempt_client.cpp +++ b/src/Disks/tests/gtest_cas_s3_single_attempt_client.cpp @@ -14,18 +14,24 @@ #include #include #include +#include +#include +#include #include #include #include #include #include +#include #include #include #include #include #include #include +#include +#include #include @@ -49,6 +55,11 @@ /// The single-attempt client clone must cap its connect timeout at the value the mount froze at open, /// never at the disk's (possibly wider, possibly reloaded, possibly unbounded) own connect timeout. +namespace ProfileEvents +{ +extern const Event S3PutObject; +} + namespace { @@ -196,7 +207,8 @@ class DelayedResponseServer /// succeeds. No SDK-level retry (`RetryStrategy{.max_retries = 0}`, /// `s3_slow_all_threads_after_retryable_error = false`): a retry would blur "the single-attempt clone /// made exactly one request" into "the SDK also tried again". -std::shared_ptr makeDispatchStorageForTest(const std::string & endpoint, long base_request_timeout_ms) +template +std::shared_ptr makeDispatchStorageForTest(const std::string & endpoint, long base_request_timeout_ms) { DB::RemoteHostFilter remote_host_filter; DB::S3::PocoHTTPClientConfiguration cfg = DB::S3::ClientFactory::instance().createClientConfiguration( @@ -228,7 +240,7 @@ std::shared_ptr makeDispatchStorageForTest(const std::strin cfg.http_keep_alive_timeout = 0; auto client = DB::S3::ClientFactory::instance().create( cfg, clientSettingsForTest(), "ACCESS_KEY_ID", "SECRET_ACCESS_KEY", "", {}, {}, DB::S3::CredentialsConfiguration{}); - return std::make_shared( + return std::make_shared( std::move(client), std::make_unique(), DB::S3::URI(endpoint + "/test-bucket/"), DB::S3Capabilities{}, DB::ObjectStorageKeyGeneratorPtr{}, "disk"); @@ -239,6 +251,43 @@ DB::ContextPtr contextForTest() return getContext().context; } +/// A genuine `S3ObjectStorage` that remembers the buffer size each `writeObject` was opened with. +class BufferSizeRecordingS3ObjectStorage final : public DB::S3ObjectStorage +{ +public: + using DB::S3ObjectStorage::S3ObjectStorage; + + std::unique_ptr writeObject( + const DB::StoredObject & object, + DB::WriteMode mode, + std::optional attributes, + size_t buf_size, + const DB::WriteSettings & write_settings) override + { + opened_buffer_sizes.push_back(buf_size); + return DB::S3ObjectStorage::writeObject(object, mode, attributes, buf_size, write_settings); + } + + std::vector opened_buffer_sizes; +}; + +/// A server that accepts every `PUT` with a fixed `ETag`. +void acceptPut(Poco::Net::HTTPServerResponse & response) +{ + response.set("ETag", "\"put-etag\""); + response.setContentLength(0); + response.setStatus(Poco::Net::HTTPResponse::HTTP_OK); + response.send(); +} + +/// A writable Native backend as `openPoolView` builds one: conditional writes are single-attempt. +std::shared_ptr conditionalBackendForTest(DB::ObjectStoragePtr storage) +{ + return std::make_shared( + std::move(storage), DB::Cas::ObjectStorageBackend::Mode::Native, + /*single_attempt_control_plane_=*/true, /*attempt_timeout_ms_=*/5000, /*connect_timeout_cap_ms_=*/5000); +} + } /// Test 6c of the spec: the clone's connect cap is the MIN of the base client's own connect timeout @@ -728,4 +777,65 @@ TEST(CASEnvelopeWiring, FreezeConnectTimeoutCapReachesTheBackendOverProductionDi } } +/// A conditional `PUT` is sent by the thread that asked for it, not by the remote-FS writer pool, so a +/// memory guard the caller holds covers it. `WriteBufferFromS3` counts `S3PutObject` on the thread that +/// calls `PutObject`, and a pool thread has counters of its own. +TEST(CASBackend, ConditionalPutRunsOnTheCallingThread) +{ + (void)contextForTest(); + + DelayedResponseServer server(std::chrono::milliseconds(0), acceptPut); + auto backend = conditionalBackendForTest(makeDispatchStorageForTest(server.getUrl(), 10000)); + EXPECT_FALSE(backend->conditionalWriteSettingsForTest().s3_allow_parallel_part_upload); + + DB::Cas::CasRequests requests(DB::Cas::BackendPtr(backend), DB::Cas::Fence::open()); + bool committed = false; + uint64_t puts_on_calling_thread = 0; + std::exception_ptr failure; + ThreadFromGlobalPool caller([&] + { + try + { + const uint64_t before = DB::CurrentThread::getProfileEvents()[ProfileEvents::S3PutObject].load(); + auto op = requests.admit(); + committed = std::holds_alternative(op.create("put-key", "body", DB::Cas::Retry::once())); + puts_on_calling_thread = DB::CurrentThread::getProfileEvents()[ProfileEvents::S3PutObject].load() - before; + } + catch (...) + { + failure = std::current_exception(); + } + }); + caller.join(); + if (failure) + std::rethrow_exception(failure); + + EXPECT_TRUE(committed); + EXPECT_EQ(server.requestsSeen(), 1u); + EXPECT_EQ(puts_on_calling_thread, 1u) << "the conditional PUT was sent by another thread"; +} + +/// A conditional `PUT` opens its buffer at the body size instead of 1 MiB, and an empty body at one +/// byte, which is what `WriteBuffer::write` needs to accept zero bytes. +TEST(CASBackend, ConditionalPutBufferStartsAtTheBodySize) +{ + (void)contextForTest(); + + DelayedResponseServer server(std::chrono::milliseconds(0), acceptPut); + auto storage = makeDispatchStorageForTest(server.getUrl(), 10000); + auto backend = conditionalBackendForTest(storage); + DB::Cas::CasRequests requests(DB::Cas::BackendPtr(backend), DB::Cas::Fence::open()); + + const std::string body(37, 'b'); + { + auto op = requests.admit(); + EXPECT_TRUE(std::holds_alternative(op.create("small-key", body, DB::Cas::Retry::once()))); + } + { + auto op = requests.admit(); + EXPECT_TRUE(std::holds_alternative(op.create("empty-key", "", DB::Cas::Retry::once()))); + } + EXPECT_EQ(storage->opened_buffer_sizes, (std::vector{body.size(), 1})); +} + #endif From 4804c478979ad63ad27cf53eb343a9cb679b7304 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 02:04:31 +0200 Subject: [PATCH 07/31] Report each CAS write attempt to an optional observer A caller that renews the mount lease needs to see requests and failures while a write is still retrying. `CasOperation::setRequestObserver` is told of every physical `PUT` when sent, of every `PUT` that throws, and of every failed resolve read. A refused precondition, a successful read and the caller's own reads are not reported, and an observer that throws is ignored. Co-Authored-By: Claude Opus 5.5 --- .../ContentAddressed/Backend/CasRequests.cpp | 28 +++- .../ContentAddressed/Backend/CasRequests.h | 17 +++ src/Disks/tests/gtest_cas_requests.cpp | 128 ++++++++++++++++++ 3 files changed, 170 insertions(+), 3 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.cpp index dde360a710c5..f60f373b7785 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.cpp @@ -6,6 +6,7 @@ #include #include #include +#include #include #include "config.h" @@ -898,6 +899,19 @@ std::optional CasOperation::reissueAtOnce(WriteState & state, const return gatedPause(0, 2, state, bound, detail::recordReissue, /*should_sleep=*/false); } +void CasOperation::notifyRequest(uint32_t attempt_no, const std::exception * failure) const noexcept +{ + if (!request_observer) + return; + try + { + request_observer(attempt_no, failure); + } + catch (...) // NOLINT(bugprone-empty-catch) + { + } +} + WriteResult CasOperation::writeLoop(const String & key, const String & bytes, const std::optional & expected, const Retry & policy, const Retry::Bound & bound, WriteState & state, ResolveWith resolve_refusal_with) @@ -930,6 +944,7 @@ WriteResult CasOperation::writeLoop(const String & key, const String & bytes, co detail::recordAttempt(); ++state.attempts_sent; state.sent_any = true; + notifyRequest(state.attempts_sent, nullptr); /// Disengaged means the attempt threw: its fate is unproven, and nothing may be read out of it. std::optional> outcome; @@ -956,6 +971,7 @@ WriteResult CasOperation::writeLoop(const String & key, const String & bytes, co } catch (const Exception & e) { + notifyRequest(state.attempts_sent, &e); if (isDeterministicLocalFailure(e.code())) throw; /// ONE refresh per call, only for the class a credential could explain, and only when a @@ -1007,6 +1023,7 @@ WriteResult CasOperation::writeLoop(const String & key, const String & bytes, co } catch (const std::exception & e) { + notifyRequest(state.attempts_sent, &e); /// Only the transport can have landed anything, and every exception it raises is a /// `Poco::Exception`. A local fault -- a bad allocation, a logic error raised inside the /// attempt -- is not a store answer, and settling it by a read would bury the bug behind an @@ -1056,9 +1073,14 @@ WriteResult CasOperation::writeLoop(const String & key, const String & bytes, co /// caller settles it with a HEAD; proving an ambiguous attempt landed needs the bytes, and there /// the body read is unavoidable. ProfileEvents::increment(ProfileEvents::CASRequestResolveRead); - const Resolved resolved = resolve_refusal_with == ResolveWith::Presence && !state.any_ambiguous - ? observePresence(key, policy, bound) - : observe(key, policy, bound); + resolve_read_put_no = state.attempts_sent; + Resolved resolved; + { + SCOPE_EXIT({ resolve_read_put_no = 0; }); + resolved = resolve_refusal_with == ResolveWith::Presence && !state.any_ambiguous + ? observePresence(key, policy, bound) + : observe(key, policy, bound); + } state.last_seen = resolved.seen; /// A bound refused the resolve, so say WHICH. Erasing it here is what let a lost fence be /// reported as an ordinary conflict and a lease refusal as a policy deadline. diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.h index 055bfc55952d..d09a37221520 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequests.h @@ -260,6 +260,14 @@ class CasOperation /// The write lane for keys several writers of this pool share. CasHotKeys & hotKeys() const { return *owner.hot_keys; } + /// Told of the requests of this operation's writes, on the issuing thread: each `PUT` when it is + /// sent (null `failure`) and again if it throws, and each failed resolve read. `attempt_no` is the + /// count of `PUT`s sent so far. A refused precondition is an answer, not a failure, and is not + /// reported; nor is a read that succeeds or one the caller issued itself. What the observer throws + /// is ignored. + using RequestObserver = std::function; + void setRequestObserver(RequestObserver observer) { request_observer = std::move(observer); } + /// `policy` with its window turned into an absolute deadline on this operation's clock, taken NOW. /// A hand-written loop freezes its policy once before it starts and passes the frozen value to /// every call it makes, so the loop ends when the window it was given ends -- rather than granting @@ -400,6 +408,9 @@ class CasOperation /// state the read never saw, which is how a lease refusal used to be reported as a policy deadline. WriteResult gaveUpAfterFailedObservation(std::optional stop, WriteState & state, const Retry::Bound & bound) const; + /// Calls `request_observer` if one is set. A report must not change a verdict, so whatever the + /// observer throws is swallowed here. + void notifyRequest(uint32_t attempt_no, const std::exception * failure) const noexcept; /// The shared shape behind every gated pause below: admission for `envelopes` attempt reservations /// plus `pause_ms`, the deadline check, the counter this pause records itself under, then the sleep /// -- called even with a zero `pause_ms` UNLESS `should_sleep` is false, which is reserved for the @@ -453,6 +464,10 @@ class CasOperation /// Written immediately before a read-class give-up throws, cleared and read only by the resolve /// read that swallows it. Every other caller lets the exception carry the verdict. std::optional last_read_stop; + RequestObserver request_observer; + /// The `PUT` count while a write's resolve read runs, 0 otherwise. A read failure is reported only + /// when it is set, so the caller's own reads stay silent. + uint32_t resolve_read_put_no = 0; }; template @@ -487,6 +502,8 @@ auto CasOperation::readLoop(std::string_view verb, const String & subject, const } catch (const std::exception & e) { + if (resolve_read_put_no != 0) + notifyRequest(resolve_read_put_no, &e); bool refreshed = false; if (refreshAndClassifyReadFault(e, refresh_attempted, refreshed)) throw; diff --git a/src/Disks/tests/gtest_cas_requests.cpp b/src/Disks/tests/gtest_cas_requests.cpp index 328368f0786a..edcfcbbc5126 100644 --- a/src/Disks/tests/gtest_cas_requests.cpp +++ b/src/Disks/tests/gtest_cas_requests.cpp @@ -745,6 +745,134 @@ TEST(CASRequests, AmbiguousCreateThatNeverLandedIsReissued) EXPECT_EQ(clock.sleeps.size(), 1u); } +/// The observer hears of each physical write attempt when it is sent, and of each one that throws. +TEST(CASRequests, RequestObserverSeesEachPutAndEachFailure) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->injectAmbiguousWrite("k"); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + std::vector> seen; + op.setRequestObserver([&](uint32_t attempt_no, const std::exception * failure) + { + seen.emplace_back(attempt_no, failure != nullptr); + }); + + WriteResult result = op.create("k", "v", Retry::standard()); + + ASSERT_TRUE(std::holds_alternative(result)); + const std::vector> expected{{1, false}, {1, true}, {2, false}}; + EXPECT_EQ(seen, expected); +} + +/// A deterministic local failure is reported before it propagates unchanged. +TEST(CASRequests, RequestObserverSeesADeterministicFailureBeforeItPropagates) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->failNextWriteWith("k", std::make_exception_ptr(DB::Exception( + DB::ErrorCodes::CORRUPTED_DATA, "injected deterministic failure"))); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + std::vector> seen; + op.setRequestObserver([&](uint32_t attempt_no, const std::exception * failure) + { + const auto * db_failure = dynamic_cast(failure); + seen.emplace_back(attempt_no, failure == nullptr ? 0 : (db_failure ? db_failure->code() : -1)); + }); + + expectThrowsCode(DB::ErrorCodes::CORRUPTED_DATA, [&] { (void)op.create("k", "v", Retry::standard()); }); + + const std::vector> expected{{1, 0}, {1, DB::ErrorCodes::CORRUPTED_DATA}}; + EXPECT_EQ(seen, expected); +} + +/// A refused precondition is the store's answer, not a failed request. +TEST(CASRequests, RequestObserverIsNotToldOfARefusedPrecondition) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->refuseNextWrite("k"); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + std::vector> seen; + op.setRequestObserver([&](uint32_t attempt_no, const std::exception * failure) + { + seen.emplace_back(attempt_no, failure != nullptr); + }); + + WriteResult result = op.create("k", "v", Retry::standard()); + + EXPECT_TRUE(std::holds_alternative(result)); + const std::vector> expected{{1, false}}; + EXPECT_EQ(seen, expected); +} + +/// Each failed resolve read is reported with the count of `PUT`s sent so far; the read that succeeds is not. +TEST(CASRequests, RequestObserverSeesEachFailedResolveRead) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->injectAmbiguousWrite("k"); + backend->failNextReadWith("k", std::make_exception_ptr(Poco::TimeoutException("injected read failure"))); + backend->failNextReadWith("k", std::make_exception_ptr(Poco::TimeoutException("injected read failure"))); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + std::vector> seen; + op.setRequestObserver([&](uint32_t attempt_no, const std::exception * failure) + { + seen.emplace_back(attempt_no, failure != nullptr); + }); + + WriteResult result = op.create("k", "v", Retry::standard()); + + ASSERT_TRUE(std::holds_alternative(result)); + /// PUT 1 sent, PUT 1 failed, two failed reads, PUT 2 sent. + const std::vector> expected{{1, false}, {1, true}, {1, true}, {1, true}, {2, false}}; + EXPECT_EQ(seen, expected); +} + +/// A read the caller issues itself is not a resolve read and is not reported. +TEST(CASRequests, RequestObserverIsNotToldOfACallersOwnRead) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->failNextReadWith("k", std::make_exception_ptr(Poco::TimeoutException("injected read failure"))); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + uint32_t calls = 0; + op.setRequestObserver([&](uint32_t, const std::exception *) { ++calls; }); + + EXPECT_FALSE(op.read("k", Retry::standard()).has_value()); + + EXPECT_EQ(calls, 0u); +} + +/// An observer that throws on every call changes neither the verdict nor the requests sent. +TEST(CASRequests, ThrowingRequestObserverChangesNothing) +{ + FakeClock clock; + auto backend = std::make_shared(); + backend->injectAmbiguousWrite("k"); + auto requests = makeRequests(backend, clock); + auto op = requests.admit(); + uint32_t calls = 0; + op.setRequestObserver([&](uint32_t, const std::exception *) + { + ++calls; + throw std::runtime_error("injected observer failure"); + }); + + WriteResult result = op.create("k", "v", Retry::standard()); + + const auto * committed = std::get_if(&result); + ASSERT_NE(committed, nullptr); + EXPECT_EQ(committed->attempts_sent, 2u); + EXPECT_EQ(backend->getTotal(), 1u); + EXPECT_EQ(calls, 3u); +} + /// The engine's own attempt number reaches the transport through `TransportAccess::attemptNo()`, for /// every primitive -- write, read (the resolve read is its own call, with its own attempt count) and /// list. From 2b6ab36ae75f7fe6eec9dbc9c93dc4aa9e15c898 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 02:13:50 +0200 Subject: [PATCH 08/31] Give the CAS mount renewer a lease plane with no fence `Pool` owns `lease_requests` (open fence, the mount plane's interruptible sleep) and hands it through `CasMountRuntime` to `MountLeaseRenewer`, which routes an `UntilDefinitive` renewal there. Adds `MountRenewPolicy`, `kMountRenewRetrySpacingMs`, `MountRenewRequestEvent` and the `policy`/`on_request` environment fields. Nothing sets `UntilDefinitive` yet, so no renewal changes plane or policy. Co-Authored-By: Claude Opus 5.5 --- .../ContentAddressed/Pool/CasMountRuntime.cpp | 6 +- .../ContentAddressed/Pool/CasMountRuntime.h | 7 +- .../ContentAddressed/Pool/CasPool.cpp | 25 +++--- .../ContentAddressed/Pool/CasPool.h | 21 ++--- .../ContentAddressed/Pool/CasServerRoot.cpp | 7 +- .../ContentAddressed/Pool/CasServerRoot.h | 28 ++++++- .../Tools/CasDecommission.cpp | 2 +- src/Disks/tests/gtest_cas_event_log.cpp | 1 + src/Disks/tests/gtest_cas_gc_ack_floor.cpp | 4 +- src/Disks/tests/gtest_cas_heartbeat.cpp | 76 ++++++++++--------- src/Disks/tests/gtest_cas_mount.cpp | 44 ++++++----- .../tests/gtest_cas_mount_claim_conflicts.cpp | 3 +- src/Disks/tests/gtest_cas_mount_runtime.cpp | 4 +- src/Disks/tests/gtest_cas_pool.cpp | 9 ++- 14 files changed, 148 insertions(+), 89 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index 6e5b5080749a..62f7c045c705 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -55,6 +55,7 @@ CasMountRuntime::CasMountRuntime( BackendPtr backend_ptr_, CasRequests & mount_requests_, CasRequests & farewell_requests_, + CasRequests & lease_requests_, const Layout & layout_, MountConfig config_, String server_root_id_, @@ -64,6 +65,7 @@ CasMountRuntime::CasMountRuntime( : backend_ptr(std::move(backend_ptr_)) , mount_requests(mount_requests_) , farewell_requests(farewell_requests_) + , lease_requests(lease_requests_) , layout(layout_) , config(std::move(config_)) , server_root_id(std::move(server_root_id_)) @@ -368,7 +370,7 @@ void CasMountRuntime::installRenewer( const std::function & now_ms) { auto replacement = std::make_unique( - mount_requests, farewell_requests, layout, server_root_id, our_uuid, writer_epoch, + mount_requests, farewell_requests, lease_requests, layout, server_root_id, our_uuid, writer_epoch, config.mount_lease_ttl_ms, now_ms, [this] { return minActive(); }, [this](CasEvent e) { emitEvent(std::move(e)); }, @@ -429,6 +431,8 @@ MountRenewOperationEnvironment CasMountRuntime::renewalEnvironment(bool worker_c return config.renewal_live_for_test ? config.renewal_live_for_test() : renewalLive(worker_call); }, .cancelled = [this] { return renewalCancelled(); }, + .policy = MountRenewPolicy::LeaseBound, + .on_request = {}, }; } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index 7f3ff1d76f36..c0175613c678 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -146,10 +146,12 @@ class CasMountRuntime public: CasMountRuntime( BackendPtr backend_ptr_, - /// The two planes the `MountLeaseRenewer` runs on: renewals under the mount fence, the farewell - /// on an open one. Owned by `Pool` and outliving this runtime. + /// The planes the `MountLeaseRenewer` runs on: a bounded renewal under the mount fence, the + /// claim and the farewell on an open one, and the worker's renewal on `lease_requests_`, which + /// has no lease budget and whose sleep a stop wakes. Owned by `Pool` and outliving this runtime. CasRequests & mount_requests_, CasRequests & farewell_requests_, + CasRequests & lease_requests_, const Layout & layout_, MountConfig config_, String server_root_id_, @@ -458,6 +460,7 @@ class CasMountRuntime BackendPtr backend_ptr; CasRequests & mount_requests; CasRequests & farewell_requests; + CasRequests & lease_requests; const Layout & layout; MountConfig config; String server_root_id; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp index 8ea206618492..98009fe3a20f 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp @@ -173,8 +173,8 @@ Pool::Pool(BackendPtr backend_, PoolConfig config_, PoolMeta meta_) , meta(std::move(meta_)) , hot_keys(config.hot_key_cache_bytes) /// The mount plane's fence reaches `mount_runtime`, declared far below: the closures capture - /// `this` and run only after construction, exactly like `ref_ledger`'s callbacks. All three planes - /// take the fence's own clock, so a policy bound to a mount-lease deadline and the fence that + /// `this` and run only after construction, exactly like `ref_ledger`'s callbacks. Every plane + /// takes the fence's own clock, so a policy bound to a mount-lease deadline and the fence that /// enforces it are read from the same source. , mount_requests(pool_backend, Fence{ [this] { return mount_runtime.fenceGeneration(); }, @@ -187,7 +187,7 @@ Pool::Pool(BackendPtr backend_, PoolConfig config_, PoolMeta meta_) /// The open plane's fence is the pool's teardown flag: generation 0 forever, exactly like /// `Fence::open`, but `admit` refuses once `beginTeardown` ran. A GC round, an FSCK or a probe in /// flight is then refused at its next request instead of running to completion under a disk that - /// is being torn down. The ref ledger and the farewell live on the other two planes, so + /// is being torn down. The ref ledger and the farewell live on other planes, so /// teardown's own I/O never meets this fence. A write already proven durable is admitted ONCE /// MORE (`postCommit`), so an armed teardown can turn a landed `gc/state` into a give-up rather /// than a commit. That is safe and not merely tolerable: the round is one-pass, so the next round @@ -201,6 +201,9 @@ Pool::Pool(BackendPtr backend_, PoolConfig config_, PoolMeta meta_) config.boot_ms_fn, config.retry_sleep_fn ? config.retry_sleep_fn : openPlaneSleepFn(), &hot_keys) + , lease_requests(pool_backend, Fence::open(), config.boot_ms_fn, + config.retry_sleep_fn ? config.retry_sleep_fn : mountPlaneSleepFn(), + &hot_keys) /// Seed the monotone admitted-algo cache from the pool state `createOrValidate` already /// established (fresh create, steady-state member, or a just-completed admission union) -- /// register-before-first-write means this Pool's own `writeAlgo()` is ALWAYS a @@ -242,13 +245,13 @@ Pool::Pool(BackendPtr backend_, PoolConfig config_, PoolMeta meta_) [this] (const RootNamespace & ns) { cancelInflightBuildsForNamespace(ns); }, config.recovery_pre_first_request_hook_for_test) /// Mount / write-fence / build-watermark / self-remount runtime. Injected with - /// backend/layout + the mount and farewell planes + the `MountConfig` slice + `server_root_id` + the event-sink reference + the pool + /// backend/layout + the mount, farewell and lease planes + the `MountConfig` slice + `server_root_id` + the event-sink reference + the pool /// `cas_request_budget` + the `remount_attempt` callback (== `Pool::tryRemountOnce`, whose claim/ /// recovery ORCHESTRATION stays on Pool). The callback captures `this`; it is invoked only at runtime /// (post-construction). Declared/constructed AFTER `ref_ledger`, preserving the original member order /// verbatim (mount destroyed first, ledger last; both orders proven safe -- see the header note). , mount_runtime( - pool_backend, mount_requests, farewell_requests, + pool_backend, mount_requests, farewell_requests, lease_requests, pool_layout, config.mountConfig(), config.server_root_id, event_sink_, config.cas_request_budget, [this] { return tryRemountOnce(); }) @@ -1946,14 +1949,15 @@ std::vector Pool::listMirroredChildren(const String & prefix) void Pool::setCasRetrySleepForTest(std::function sleep_fn) { - /// All three planes, not just the ledger's: a test that replaces the retry sleep must not be left - /// with a real one on the plane the site under test happens to use. + /// Every plane, not just the ledger's: a test that replaces the retry sleep must not be left with + /// a real one on the plane the site under test happens to use. farewell_requests.setSleepFnForTest(sleep_fn); ref_ledger.setCasRetrySleepForTest(sleep_fn); - /// `CasRequests` falls back to the engine's plain sleep for an empty argument -- which is neither - /// the mount plane's nor the open plane's default. Re-install both, so clearing the seam cannot - /// leave a parked renewal held for a whole capped backoff, or the open plane deaf to a teardown. + /// `CasRequests` falls back to the engine's plain sleep for an empty argument -- which is none of + /// the mount, lease or open plane's defaults. Re-install them, so clearing the seam cannot leave a + /// parked or stopping renewal held for a whole wait, or the open plane deaf to a teardown. gc_requests.setSleepFnForTest(sleep_fn ? sleep_fn : openPlaneSleepFn()); + lease_requests.setSleepFnForTest(sleep_fn ? sleep_fn : mountPlaneSleepFn()); mount_requests.setSleepFnForTest(sleep_fn ? std::move(sleep_fn) : mountPlaneSleepFn()); } @@ -1961,6 +1965,7 @@ void Pool::setCasRequestNowFnForTest(std::function now_fn) { mount_requests.setNowFnForTest(now_fn); farewell_requests.setNowFnForTest(now_fn); + lease_requests.setNowFnForTest(now_fn); gc_requests.setNowFnForTest(std::move(now_fn)); } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h index c566a9164f1a..7f1151341437 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h @@ -256,10 +256,10 @@ struct PoolConfig /// boot clock (`Pool::bootMs`); injected by tests to drive the fence deadline deterministically. std::function boot_ms_fn = {}; - /// The inter-attempt sleep for the mount, farewell and GC request planes (`mount_requests`, + /// The inter-attempt sleep for the request planes (`mount_requests`, `lease_requests`, /// `farewell_requests`, `gc_requests`), installed at their CONSTRUCTION -- before this `Pool` has /// claimed or read anything. Empty = each plane's own production default (an interruptible real - /// sleep for the mount and GC planes, `CasRequests`'s own real sleep for the farewell plane). A test + /// sleep for the mount, lease and GC planes, `CasRequests`'s own real sleep for the farewell plane). A test /// that also freezes `boot_ms_fn` must supply a matching sleep here: a retry loop bound to a clock /// that only moves when this function is called would otherwise retry forever against a REAL sleep /// that never calls it, because the deadline it measures against never appears to elapse. @@ -754,7 +754,7 @@ class Pool : public std::enable_shared_from_this const PoolMeta & poolMeta() const { return meta; } const Layout & layout() const { return pool_layout; } - /// ---- the three request planes ---- + /// ---- request planes ---- /// The mount plane: the durable writes whose right to land IS this node's mount lease. An /// operation admitted here is refused the moment the fence trips, is re-armed under a fresh lease /// incarnation, or runs out of room before the lease expires. @@ -1072,10 +1072,10 @@ class Pool : public std::enable_shared_from_this ref_ledger.setSnapshotBeforeCkptCasHookForTest(std::move(hook)); } - /// Test-only: replace the inter-attempt backoff sleep (e.g. with a clock-advancing no-op) on all - /// three request planes and on ref-table recovery, for tests that drive a persistent write fault to + /// Test-only: replace the inter-attempt backoff sleep (e.g. with a clock-advancing no-op) on every + /// request plane and on ref-table recovery, for tests that drive a persistent write fault to /// exhaustion through a fully wired Pool/disk and must not serve the production sleeps for real. - /// Call before driving traffic. On the three request planes an empty function restores each plane's + /// Call before driving traffic. On the request planes an empty function restores each plane's /// own default, the mount plane's interruptible sleep included. /// /// It does NOT bound a reissue the engine refuses to start: the engine's inter-attempt backoff is @@ -1084,7 +1084,7 @@ class Pool : public std::enable_shared_from_this /// clock, not the sleep. void setCasRetrySleepForTest(std::function sleep_fn); - /// Test-only: replace the request engine's clock on all three planes. A test driving a PERSISTENT + /// Test-only: replace the request engine's clock on every plane. A test driving a PERSISTENT /// transient fault must run the retry window on a clock it advances; the sleep seam alone cannot /// bound it, because a read the engine keeps reissuing is bounded by the policy deadline and the /// deadline is read from this clock. @@ -1198,17 +1198,20 @@ class Pool : public std::enable_shared_from_this PoolConfig config; PoolMeta meta; - /// The pool's write lane for keys several of its writers share, declared before the three planes + /// The pool's write lane for keys several of its writers share, declared before the planes /// that carry a pointer to it, so it outlives every operation they admit. `mutable` for the same /// reason the planes are. mutable CasHotKeys hot_keys; - /// The three planes' engines, declared before every component that is handed one and after the + /// The planes' engines, declared before every component that is handed one and after the /// config they take their clock from. `mutable` because issuing a request is not a change to the /// pool: a `const` observer still has to read the store. mutable CasRequests mount_requests; mutable CasRequests farewell_requests; mutable CasRequests gc_requests; + /// The worker renewal's plane: no lease budget, because the renewal keeps trying after the lease + /// expired; a stop, a park or a terminal lifecycle ends it through its liveness. + mutable CasRequests lease_requests; std::shared_ptr detached_work = std::make_shared(); diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp index b2f4dc011871..f60e7acdc115 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp @@ -1316,7 +1316,8 @@ constexpr uint64_t kFarewellBudgetMs = 10'000; constexpr uint64_t kFarewellSlackMs = 2'000; MountLeaseRenewer::MountLeaseRenewer( - CasRequests & mount_requests_, CasRequests & open_requests_, const Layout & layout_, + CasRequests & mount_requests_, CasRequests & open_requests_, CasRequests & worker_requests_, + const Layout & layout_, const String & srid_, UInt128 server_uuid_, uint64_t writer_epoch_, std::chrono::milliseconds ttl_, std::function now_ms_fn_, std::function min_active_build_sequence_fn_, @@ -1325,6 +1326,7 @@ MountLeaseRenewer::MountLeaseRenewer( std::function boot_ms_fn_) : mount_requests(mount_requests_) , open_requests(open_requests_) + , worker_requests(worker_requests_) , key(layout_.mountKey(srid_)) , srid(srid_) , server_uuid(server_uuid_) @@ -1570,7 +1572,8 @@ MountRenewResult MountLeaseRenewer::terminalResult(MountRenewResult result) MountRenewResult MountLeaseRenewer::renew(const MountRenewOperationEnvironment & environment) { - return renewOn(mount_requests, environment); + return renewOn( + environment.policy == MountRenewPolicy::UntilDefinitive ? worker_requests : mount_requests, environment); } MountRenewResult MountLeaseRenewer::renewForRemount(const MountRenewOperationEnvironment & environment) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h index e3ef3ee434da..ac3cc1167ef9 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h @@ -61,6 +61,25 @@ struct MountRenewResult std::exception_ptr failure; }; +/// How far a renewal may retry. +enum class MountRenewPolicy : uint8_t +{ + LeaseBound, /// startup, remount and direct renewals: `Retry::untilLeaseSafe` + UntilDefinitive, /// the background worker: `Retry::untilDefinitive(kMountRenewRetrySpacingMs)` +}; + +/// The spacing of an `UntilDefinitive` renewal's retries. +inline constexpr uint64_t kMountRenewRetrySpacingMs = 1000; + +/// One physical request of a renewal, reported as it happens: a `PUT` sent (`failed` false), a `PUT` +/// that failed, or a failed read that settles a `PUT` (both `failed` true). +struct MountRenewRequestEvent +{ + uint32_t request_no = 0; /// 1-based count of `PUT`s sent by this renewal + bool failed = false; + String failure_text; /// empty unless `failed` +}; + struct MountRenewOperationEnvironment { std::function boot_ms; @@ -71,6 +90,10 @@ struct MountRenewOperationEnvironment /// `NotAttempted` rather than terminal only when this node had already been asked to stop -- /// sampling it afterwards would read a flag that the refusal itself may have set. std::function cancelled; + MountRenewPolicy policy = MountRenewPolicy::LeaseBound; + /// Called on the renewing thread for every request sent and for every failure. May be empty. + /// What it throws is ignored. + std::function on_request; }; /// Validate a `server_root_id` — the explicit, configured identity of the content-addressed layout @@ -536,7 +559,8 @@ class MountLeaseRenewer { public: MountLeaseRenewer( - CasRequests & mount_requests_, CasRequests & open_requests_, const Layout & layout_, + CasRequests & mount_requests_, CasRequests & open_requests_, CasRequests & worker_requests_, + const Layout & layout_, const String & srid_, UInt128 server_uuid_, uint64_t writer_epoch_, std::chrono::milliseconds ttl_, std::function now_ms_fn_, std::function min_active_build_sequence_fn_, @@ -577,6 +601,8 @@ class MountLeaseRenewer CasRequests & mount_requests; CasRequests & open_requests; + /// The plane of an `UntilDefinitive` renewal: no lease budget, and a sleep a stop wakes. + CasRequests & worker_requests; String key; String srid; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasDecommission.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasDecommission.cpp index eef2ba04d9bf..6c8659d92881 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasDecommission.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Tools/CasDecommission.cpp @@ -173,7 +173,7 @@ DecommissionReport decommissionPoolMember(BackendPtr backend, PoolConfig config, if (drain_now_fn && drain_sleep_fn) { /// Re-affirms the same values `config` above already installed on `mount_requests`/ - /// `farewell_requests`/`gc_requests` at construction, and additionally wires `ref_ledger`'s own + /// `lease_requests`/`farewell_requests`/`gc_requests` at construction, and additionally wires `ref_ledger`'s own /// retry sleep, which has no construction-time seam of its own. `sweepNamespace` below issues /// its deletes on `admin`'s own GC plane. admin->setCasRequestNowFnForTest(drain_now_fn); diff --git a/src/Disks/tests/gtest_cas_event_log.cpp b/src/Disks/tests/gtest_cas_event_log.cpp index e55f775ef4f8..79cde4da3f8f 100644 --- a/src/Disks/tests/gtest_cas_event_log.cpp +++ b/src/Disks/tests/gtest_cas_event_log.cpp @@ -371,6 +371,7 @@ TEST(CASEvent, DeepReentrancyPreservesDeterministicPhysicalAttemptTruth) planes[index] = std::make_unique( backends[index], Fence::open(), [&] { return boot_ms; }, [&](uint64_t ms) { boot_ms += ms; }); renewers[index] = std::make_unique( + *planes[index], *planes[index], *planes[index], *layouts[index], diff --git a/src/Disks/tests/gtest_cas_gc_ack_floor.cpp b/src/Disks/tests/gtest_cas_gc_ack_floor.cpp index ea63268a8fe3..d0fff5bcff6b 100644 --- a/src/Disks/tests/gtest_cas_gc_ack_floor.cpp +++ b/src/Disks/tests/gtest_cas_gc_ack_floor.cpp @@ -933,7 +933,7 @@ void runExpiredMountFenceOutScenario(const PoolConfig & config) // never changes again. const String srid2 = "stale-server"; CasRequests renewer_requests = openRequestsForTest(backend); - MountLeaseRenewer srid2_renewer(renewer_requests, renewer_requests, layout, srid2, DB::UInt128(0x2222), + MountLeaseRenewer srid2_renewer(renewer_requests, renewer_requests, renewer_requests, layout, srid2, DB::UInt128(0x2222), /*writer_epoch=*/1, std::chrono::milliseconds(100), [] { return 1000u; }, [] { return 0u; }, {}, std::chrono::milliseconds(0), [] { return 0u; }); @@ -1061,7 +1061,7 @@ TEST(CASGCAckFloor, DefaultMonoClockTracksPoolsInjectedBootClockNotWallClock) // A stale mount, exactly as `ExpiredMountFencedOutAndExcluded`: one claim, never renewed again. const String srid2 = "stale-server"; CasRequests renewer_requests = openRequestsForTest(backend); - MountLeaseRenewer srid2_renewer(renewer_requests, renewer_requests, layout, srid2, DB::UInt128(0x2222), + MountLeaseRenewer srid2_renewer(renewer_requests, renewer_requests, renewer_requests, layout, srid2, DB::UInt128(0x2222), /*writer_epoch=*/1, std::chrono::milliseconds(100), [] { return 1000u; }, [fake_boot] diff --git a/src/Disks/tests/gtest_cas_heartbeat.cpp b/src/Disks/tests/gtest_cas_heartbeat.cpp index c39935d7ca7f..ea20ff3bada3 100644 --- a/src/Disks/tests/gtest_cas_heartbeat.cpp +++ b/src/Disks/tests/gtest_cas_heartbeat.cpp @@ -34,23 +34,24 @@ using namespace DB::Cas; namespace { -/// The two request planes this file's renewers run on. Both are open-fence -- the exclusivity these -/// tests exercise is the mount protocol's own, not a fence's -- on the same injected boot clock the -/// renewer's lease deadline is expressed on, so the two never disagree about how much budget is left. +/// The request planes this file's renewers run on. All are open-fence -- the exclusivity these tests +/// exercise is the mount protocol's own, not a fence's -- on the same injected boot clock the renewer's +/// lease deadline is expressed on, so they never disagree about how much budget is left. /// `sleep_step_ms`, when set, makes one inter-attempt pause jump the clock past the lease bound: that /// is how a test asks for exactly one physical attempt without a per-call attempt cap. It depends on /// the engine checking the bound, sleeping, then checking again -- a reissue that slept first would /// send a second attempt. `tests::OperationForTest` covers a fixture needing one operation, but -/// neither the two planes a renewer takes nor this clock, which is why this stays local. +/// neither the planes a renewer takes nor this clock, which is why this stays local. class Ops { public: Ops(std::shared_ptr backend, uint64_t * boot_ms, uint64_t sleep_step_ms = 0) : mount(openRequestsForTest(backend)) - , farewell(openRequestsForTest(std::move(backend))) + , farewell(openRequestsForTest(backend)) + , lease(openRequestsForTest(std::move(backend))) , op(mount.admit()) { - for (CasRequests * requests : {&mount, &farewell}) + for (CasRequests * requests : {&mount, &farewell, &lease}) { requests->setNowFnForTest([boot_ms] { return *boot_ms; }); requests->setSleepFnForTest( @@ -63,6 +64,7 @@ class Ops CasRequests mount; CasRequests farewell; + CasRequests lease; CasOperation op; }; @@ -181,6 +183,8 @@ MountRenewOperationEnvironment renewalEnvironment( .boot_ms = [&boot_ms] { return boot_ms; }, .live = live, .cancelled = cancelled, + .policy = MountRenewPolicy::LeaseBound, + .on_request = {}, }; } @@ -225,7 +229,7 @@ TEST(CASHeartbeat, AnchorCarriesFloor) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); - MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), [&] { return now_ms; }, [&] { return min_active_build_sequence_now; }, {}, std::chrono::milliseconds(0), [&] { return boot_ms; }); @@ -251,7 +255,7 @@ TEST(CASHeartbeat, RenewRereadsCallbackAndBumpsSeq) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); - MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), [&] { return now_ms; }, [&] { return min_active_build_sequence_now; }, {}, std::chrono::milliseconds(0), [&] { return boot_ms; }); @@ -279,7 +283,7 @@ TEST(CASHeartbeat, StopStampsExpiredAndFarewellSentinel) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); - MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(0), [&] { return boot_ms; }); @@ -335,7 +339,7 @@ TEST(CASHeartbeat, FarewellIsAdmittedUnderTheDefaultBudget) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/30000); - MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(30000), [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(2000), [&] { return boot_ms; }); @@ -368,7 +372,7 @@ TEST(CASHeartbeat, FarewellIsAdmittedUnderADifferentEnvelope) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/40000); - MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(40000), [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(2000), [&] { return boot_ms; }); @@ -402,7 +406,7 @@ TEST(CASHeartbeat, FarewellIsRefusedWhenTheLeaseExpiresBeforeItsDerivedWindow) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/5000); - MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(5000), [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(2000), [&] { return boot_ms; }); @@ -448,7 +452,7 @@ TEST(CASHeartbeat, ForeignIncarnationDuringFarewellLeavesTheSuccessorUntouchedAn Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); - MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(0), [&] { return boot_ms; }); @@ -505,7 +509,7 @@ TEST(CASHeartbeat, SameEpochUnfencedTouchIsUncertainNotFatal) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); - MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(0), [&] { return boot_ms; }); @@ -552,7 +556,7 @@ TEST(CASHeartbeat, SupersededTouchIsFailClosedNotFatal) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); - MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(0), [&] { return boot_ms; }); @@ -606,7 +610,7 @@ TEST(CASHeartbeat, ForeignUuidTouchFailsClosedWithoutAborting) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); - MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(0), [&] { return boot_ms; }); @@ -694,7 +698,7 @@ TEST(CASMountAudit, RenewerAdoptEmitsClaimAndTerminateEmitsRelease) std::vector seen; CasEventSink sink = [&](const CasEvent & e) { seen.push_back(e); }; - MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), [&] { return now_ms; }, [] { return uint64_t{5}; }, sink, std::chrono::milliseconds(0), [&] { return boot_ms; }); @@ -732,7 +736,7 @@ TEST(CASMountAudit, RenewerForeignConflictRefusesAndNamesHolder) ASSERT_EQ(claimMount(ops.op, layout, srid, uuid_x, /*our_epoch=*/1, now_ms, /*ttl_ms=*/100).kind, MountClaimResult::Claimed); - MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid_y, /*writer_epoch=*/1, + MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid_y, /*writer_epoch=*/1, std::chrono::milliseconds(100), [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(2000), [&] { return boot_ms; }); @@ -774,7 +778,7 @@ TEST(CASMountAudit, RenewerAdoptRefusesFencedSelfWithTypedError) std::vector seen; CasEventSink sink = [&](const CasEvent & e) { seen.push_back(e); }; /// A renewer for the SAME (uuid, epoch) tries to adopt the now-fenced slot. - MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), [&] { return now_ms; }, [] { return uint64_t{5}; }, sink, std::chrono::milliseconds(2000), [&] { return boot_ms; }); @@ -814,7 +818,7 @@ TEST(CASHeartbeat, RenewOverFencedOwnSlotIsClassifiedNotForeign) std::vector seen; CasEventSink sink = [&](const CasEvent & e) { seen.push_back(e); }; - MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), [&] { return now_ms; }, [] { return uint64_t{5}; }, sink, std::chrono::milliseconds(0), [&] { return boot_ms; }); @@ -869,7 +873,7 @@ TEST(CASHeartbeat, RenewerStateAllowsOnlyActiveReleaseOrTerminal) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, "released", uuid, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, layout, "released", uuid, 9, std::chrono::milliseconds(1000), + ops.mount, ops.farewell, ops.lease, layout, "released", uuid, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{7}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); EXPECT_EQ(renewer.state(), MountLeaseRenewerState::New); @@ -894,7 +898,7 @@ TEST(CASHeartbeat, RenewerStateAllowsOnlyActiveReleaseOrTerminal) Ops ops(backend, &boot_ms, /*sleep_step_ms=*/10'000); seedOwnClaim(ops.op, layout, "terminal", uuid, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, layout, "terminal", uuid, 9, std::chrono::milliseconds(1000), + ops.mount, ops.farewell, ops.lease, layout, "terminal", uuid, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{7}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); renewer.start(); @@ -922,7 +926,7 @@ TEST(CASHeartbeat, RenewalRetriesOneImmutableBodyAndAdoptsLostResponse) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, layout, srid, uuid, 9, std::chrono::milliseconds(1000), + ops.mount, ops.farewell, ops.lease, layout, srid, uuid, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{7}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); renewer.start(); @@ -960,7 +964,7 @@ TEST(CASHeartbeat, RenewalOverConnectFailuresRecoversWithoutASettleRead) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, 9, wall_ms, 30000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, layout, srid, uuid, 9, std::chrono::milliseconds(30000), + ops.mount, ops.farewell, ops.lease, layout, srid, uuid, 9, std::chrono::milliseconds(30000), [&] { return wall_ms; }, [] { return uint64_t{7}; }, {}, std::chrono::milliseconds(2000), [&] { return boot_ms; }); renewer.start(); @@ -991,7 +995,7 @@ TEST(CASHeartbeat, DeadlineBeforeSendTerminalizesWithTypedFailure) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, "test", UInt128{1}, 9, wall_ms, 100); MountLeaseRenewer renewer( - ops.mount, ops.farewell, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(100), + ops.mount, ops.farewell, ops.lease, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(100), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); renewer.start(); @@ -1019,7 +1023,7 @@ TEST(CASHeartbeat, CancellationBeforeSendIsNotAttemptedAndAllowsRelease) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, "test", UInt128{1}, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + ops.mount, ops.farewell, ops.lease, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); renewer.start(); @@ -1045,7 +1049,7 @@ TEST(CASHeartbeat, CancellationAfterSendIsTerminalAndForbidsRelease) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, "test", UInt128{1}, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + ops.mount, ops.farewell, ops.lease, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); renewer.start(); @@ -1074,7 +1078,7 @@ TEST(CASHeartbeat, SlowResolvedSuccessKeepsAttemptStartAnchor) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, "test", UInt128{1}, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + ops.mount, ops.farewell, ops.lease, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); renewer.start(); @@ -1099,7 +1103,7 @@ TEST(CASHeartbeat, SamePairTwinAndForeignOrSuccessorStayTerminal) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, "test", uuid, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, layout, "test", uuid, 9, std::chrono::milliseconds(1000), + ops.mount, ops.farewell, ops.lease, layout, "test", uuid, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); renewer.start(); @@ -1133,7 +1137,7 @@ TEST(CASHeartbeat, ExpectedPredecessorThenLateLandingIsAdoptedExactly) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, "test", UInt128{1}, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + ops.mount, ops.farewell, ops.lease, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); renewer.start(); @@ -1162,7 +1166,7 @@ TEST(CASHeartbeat, GcFenceAndVanishedMountStayTerminal) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, "test", UInt128{1}, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + ops.mount, ops.farewell, ops.lease, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); renewer.start(); @@ -1198,7 +1202,7 @@ TEST(CASHeartbeat, LateDeliveryAfterTerminalCannotRearmOrOverwriteSuccessor) Ops ops(backend, &boot_ms, /*sleep_step_ms=*/10'000); seedOwnClaim(ops.op, layout, "before-reclaim", UInt128{1}, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, layout, "before-reclaim", UInt128{1}, 9, std::chrono::milliseconds(1000), + ops.mount, ops.farewell, ops.lease, layout, "before-reclaim", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, CasEventSink{}, std::chrono::milliseconds(20), [&] { return boot_ms; }); renewer.start(); @@ -1220,7 +1224,7 @@ TEST(CASHeartbeat, LateDeliveryAfterTerminalCannotRearmOrOverwriteSuccessor) Ops ops(backend, &boot_ms, /*sleep_step_ms=*/10'000); seedOwnClaim(ops.op, layout, "after-successor", UInt128{1}, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, layout, "after-successor", UInt128{1}, 9, std::chrono::milliseconds(1000), + ops.mount, ops.farewell, ops.lease, layout, "after-successor", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); renewer.start(); @@ -1246,7 +1250,7 @@ TEST(CASHeartbeat, LateDeliveryAfterTerminalCannotRearmOrOverwriteSuccessor) ASSERT_EQ(claimMount(ops.op, layout, "after-successor", UInt128{1}, 10, wall_ms, 1000).kind, MountClaimResult::Claimed); MountLeaseRenewer successor( - ops.mount, ops.farewell, layout, "after-successor", UInt128{1}, 10, std::chrono::milliseconds(1000), + ops.mount, ops.farewell, ops.lease, layout, "after-successor", UInt128{1}, 10, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); successor.start(); @@ -1268,7 +1272,7 @@ TEST(CASHeartbeat, WallClockStepsAndBootSuspendCannotExtendAuthority) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, "test", UInt128{1}, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + ops.mount, ops.farewell, ops.lease, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); renewer.start(); @@ -1334,7 +1338,7 @@ TEST(CASHeartbeat, RenewalStopsBeforeTheCutoffWhenEveryAttemptConsumesTheEnvelop Layout layout("pool"); Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, "test", UInt128{0x1234}, 9, wall_ms, 1000); - MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, "test", UInt128{0x1234}, 9, std::chrono::milliseconds(1000), + MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, "test", UInt128{0x1234}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{7}; }, {}, std::chrono::milliseconds(100), [&] { return boot_ms; }); renewer.start(); diff --git a/src/Disks/tests/gtest_cas_mount.cpp b/src/Disks/tests/gtest_cas_mount.cpp index f4a7b671b63f..5a6231c9e4e3 100644 --- a/src/Disks/tests/gtest_cas_mount.cpp +++ b/src/Disks/tests/gtest_cas_mount.cpp @@ -59,13 +59,13 @@ void renewOrThrow(MountLeaseRenewer & renewer) ASSERT_EQ(result.outcome, MountRenewOutcome::Committed); } -/// The two request planes a renewer in this file runs on, plus one operation for the protocol calls -/// driven directly. Both planes are open-fence: these fixtures hold no mount lease, so nothing here -/// should be refused by a fence it does not have. The clock and the sleep are ALWAYS injected -- a -/// fixture that drives a lease deadline passes its own so a slow machine cannot run the bound out -/// mid-test, and one that does not still must not sleep for real when a fault sends the engine round -/// again. `tests::OperationForTest` covers the one-operation case but neither the two planes nor the -/// clock, which is why this stays local. +/// The request planes a renewer in this file runs on, plus one operation for the protocol calls driven +/// directly. All planes are open-fence: these fixtures hold no mount lease, so nothing here should be +/// refused by a fence it does not have. The clock and the sleep are ALWAYS injected -- a fixture that +/// drives a lease deadline passes its own so a slow machine cannot run the bound out mid-test, and one +/// that does not still must not sleep for real when a fault sends the engine round again. +/// `tests::OperationForTest` covers the one-operation case but neither the planes nor the clock, which +/// is why this stays local. class Ops { public: @@ -73,11 +73,12 @@ class Ops Ops(std::shared_ptr backend, uint64_t * boot_ms) : mount(openRequestsForTest(backend)) - , farewell(openRequestsForTest(std::move(backend))) + , farewell(openRequestsForTest(backend)) + , lease(openRequestsForTest(std::move(backend))) , op(mount.admit()) { uint64_t * clock = boot_ms ? boot_ms : &own_clock; - for (CasRequests * requests : {&mount, &farewell}) + for (CasRequests * requests : {&mount, &farewell, &lease}) { requests->setNowFnForTest([clock] { return *clock; }); requests->setSleepFnForTest([clock](uint64_t ms) { *clock += ms; }); @@ -89,6 +90,7 @@ class Ops CasRequests mount; CasRequests farewell; + CasRequests lease; CasOperation op; private: @@ -723,7 +725,7 @@ TEST(CASMountLease, AbsentClaimThenRenewBumpsSeq) Ops ops(b, &boot); auto r = claimMount(ops.op, l, "r", UInt128(1), /*epoch*/ 7, now, /*ttl*/ 100); EXPECT_EQ(r.kind, MountClaimResult::Claimed); - MountLeaseRenewer k(ops.mount, ops.farewell, l, "r", UInt128(1), 7, std::chrono::milliseconds(100), + MountLeaseRenewer k(ops.mount, ops.farewell, ops.lease, l, "r", UInt128(1), 7, std::chrono::milliseconds(100), [&] { return now; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(0), [&] { return boot; }); k.start(); @@ -743,7 +745,7 @@ TEST(CASMountLease, HolderBodiesMintFreshAttemptIdsAndFenceCopiesIt) const String key = layout.mountKey("r"); const MountLease claimed = decodeMountLease(ops.op.read(key, Retry::standard())->bytes); - MountLeaseRenewer renewer(ops.mount, ops.farewell, layout, "r", UInt128{1}, 7, std::chrono::milliseconds(100), + MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, "r", UInt128{1}, 7, std::chrono::milliseconds(100), [&] { return now; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(0), [&] { return boot; }); renewer.start(); @@ -799,7 +801,7 @@ TEST(CASMountLease, VanishedBackingStoreStopsRenewalWithoutLogicalError) uint64_t boot = 0; Ops ops(b, &boot); ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), /*epoch*/ 7, now, /*ttl*/ 100).kind, MountClaimResult::Claimed); - MountLeaseRenewer k(ops.mount, ops.farewell, l, "r", UInt128(1), 7, std::chrono::milliseconds(100), + MountLeaseRenewer k(ops.mount, ops.farewell, ops.lease, l, "r", UInt128(1), 7, std::chrono::milliseconds(100), [&] { return now; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(0), [&] { return boot; }); k.start(); @@ -842,7 +844,7 @@ TEST(CASMountLease, TerminateAfterVanishedBackingStoreIsNoOpRelease) uint64_t now = 1000; Ops ops(b); ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), /*epoch*/ 7, now, /*ttl*/ 100).kind, MountClaimResult::Claimed); - MountLeaseRenewer k(ops.mount, ops.farewell, l, "r", UInt128(1), 7, std::chrono::milliseconds(100), + MountLeaseRenewer k(ops.mount, ops.farewell, ops.lease, l, "r", UInt128(1), 7, std::chrono::milliseconds(100), [&] { return now; }, [] { return uint64_t{0}; }); k.start(); @@ -1154,7 +1156,7 @@ TEST(CASMountLease, RenewerStartAdoptsOurOwnClaimNotDoubleStart) Ops ops(b); // The normal flow: claimMount writes the live mount under (uuid=1, epoch=7), THEN renewer.start(). ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), /*epoch*/ 7, now, /*ttl*/ 100).kind, MountClaimResult::Claimed); - MountLeaseRenewer k(ops.mount, ops.farewell, l, "r", UInt128(1), /*epoch*/ 7, std::chrono::milliseconds(100), + MountLeaseRenewer k(ops.mount, ops.farewell, ops.lease, l, "r", UInt128(1), /*epoch*/ 7, std::chrono::milliseconds(100), [&] { return now; }, [] { return uint64_t{0}; }); EXPECT_NO_THROW(k.start()); // adopts our own live (uuid=1,epoch=7) mount — NOT a double-start EXPECT_EQ(decodeMountLease(ops.op.read(l.mountKey("r"), Retry::standard())->bytes).writer_epoch, 7u); @@ -2116,7 +2118,7 @@ TEST(CASMountObservation, RenewalDuringObservationRestartsIt) /// wrote (no seq bump, per the ADOPT RULE), then a synchronous renewal mints a new incarnation /// mid-observation. uint64_t renewer_wall = 500; - MountLeaseRenewer renewer(ops.mount, ops.farewell, l, "r", UInt128(1), 7, std::chrono::milliseconds(500), + MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, l, "r", UInt128(1), 7, std::chrono::milliseconds(500), [&] { return renewer_wall; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(0), [&] { return renewer_boot; }); renewer.start(); @@ -2305,7 +2307,7 @@ TEST(CASMountLease, ClaimAdoptIsTwoRequests) /// The absent-slot mint. backend->reads = backend->heads = backend->writes = 0; - MountLeaseRenewer minting(ops.mount, ops.farewell, l, "fresh", UInt128(1), 7, + MountLeaseRenewer minting(ops.mount, ops.farewell, ops.lease, l, "fresh", UInt128(1), 7, std::chrono::milliseconds(100), [&] { return now; }, [] { return uint64_t{0}; }); minting.start(); EXPECT_EQ(backend->reads, 1u); @@ -2316,7 +2318,7 @@ TEST(CASMountLease, ClaimAdoptIsTwoRequests) ASSERT_EQ(claimMount(ops.op, l, "adopted", UInt128(1), /*epoch*/ 7, now, /*ttl*/ 100).kind, MountClaimResult::Claimed); backend->reads = backend->heads = backend->writes = 0; - MountLeaseRenewer adopting(ops.mount, ops.farewell, l, "adopted", UInt128(1), 7, + MountLeaseRenewer adopting(ops.mount, ops.farewell, ops.lease, l, "adopted", UInt128(1), 7, std::chrono::milliseconds(100), [&] { return now; }, [] { return uint64_t{0}; }); adopting.start(); EXPECT_EQ(backend->reads, 1u); @@ -2355,10 +2357,10 @@ TEST(CASMountLease, FarewellRunsOnAnOpenFenceAfterTheMountFenceIsLost) ASSERT_EQ(claimMount(seed, l, "renewing", UInt128(1), 7, now, /*ttl*/ 1000).kind, MountClaimResult::Claimed); ASSERT_EQ(claimMount(seed, l, "departing", UInt128(1), 7, now, /*ttl*/ 1000).kind, MountClaimResult::Claimed); - MountLeaseRenewer renewing(mount_requests, open_requests, l, "renewing", UInt128(1), 7, + MountLeaseRenewer renewing(mount_requests, open_requests, open_requests, l, "renewing", UInt128(1), 7, std::chrono::milliseconds(1000), [&] { return now; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(0), [&] { return boot; }); - MountLeaseRenewer departing(mount_requests, open_requests, l, "departing", UInt128(1), 7, + MountLeaseRenewer departing(mount_requests, open_requests, open_requests, l, "departing", UInt128(1), 7, std::chrono::milliseconds(1000), [&] { return now; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(0), [&] { return boot; }); renewing.start(); @@ -2396,7 +2398,7 @@ TEST(CASMountLease, ClaimIsNotAdmittedUnderTheMountFence) open_requests.setNowFnForTest([&boot] { return boot; }); open_requests.setSleepFnForTest([&boot](uint64_t ms) { boot += ms; }); - MountLeaseRenewer renewer(mount_requests, open_requests, l, "r", UInt128(1), 7, + MountLeaseRenewer renewer(mount_requests, open_requests, open_requests, l, "r", UInt128(1), 7, std::chrono::milliseconds(1000), [&] { return now; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(0), [&] { return boot; }); EXPECT_NO_THROW(renewer.start()); @@ -2456,7 +2458,7 @@ TEST(CASMountLease, RemountRenewalIsAdmittedOffTheMountFence) open_requests.setNowFnForTest([&boot] { return boot; }); open_requests.setSleepFnForTest([&boot](uint64_t ms) { boot += ms; }); - MountLeaseRenewer renewer(mount_requests, open_requests, l, "r", UInt128(1), 7, + MountLeaseRenewer renewer(mount_requests, open_requests, open_requests, l, "r", UInt128(1), 7, std::chrono::milliseconds(1000), [&] { return now; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(0), [&] { return boot; }); renewer.start(); diff --git a/src/Disks/tests/gtest_cas_mount_claim_conflicts.cpp b/src/Disks/tests/gtest_cas_mount_claim_conflicts.cpp index 6eb8f9c9b7b8..f7c974073a49 100644 --- a/src/Disks/tests/gtest_cas_mount_claim_conflicts.cpp +++ b/src/Disks/tests/gtest_cas_mount_claim_conflicts.cpp @@ -15,7 +15,7 @@ using DB::Cas::tests::OperationForTest; namespace { -/// One renewer for the mount slot of server-root "r", under (uuid=1, epoch=7) unless overridden. Both +/// One renewer for the mount slot of server-root "r", under (uuid=1, epoch=7) unless overridden. All /// of its planes are the same open-fence one: what these tests exercise is the mount protocol's own /// exclusivity, not a fence's, and no test here renews, which is the only caller of the mount plane. MountLeaseRenewer makeRenewer( @@ -25,6 +25,7 @@ MountLeaseRenewer makeRenewer( uint64_t epoch = 7) { return MountLeaseRenewer( + requests, requests, requests, Layout("p"), diff --git a/src/Disks/tests/gtest_cas_mount_runtime.cpp b/src/Disks/tests/gtest_cas_mount_runtime.cpp index 0294fa178a35..8ff8a4e4af64 100644 --- a/src/Disks/tests/gtest_cas_mount_runtime.cpp +++ b/src/Disks/tests/gtest_cas_mount_runtime.cpp @@ -26,8 +26,9 @@ class RuntimeFixture [this](uint64_t g, uint64_t needed) { return runtime.admit(g, needed); }, [this](uint64_t g) { runtime.checkFenceOrThrow(g); }}) , farewell(backend, Fence::open()) + , lease(backend, Fence::open()) , runtime( - backend, mount, farewell, layout, + backend, mount, farewell, lease, layout, MountConfig{.boot_ms_fn = [this] { return boot_ms; }}, "test", sink, CasRequestBudget{.attempt_timeout_ms = attempt_timeout_ms, @@ -47,6 +48,7 @@ class RuntimeFixture CasEventSink sink; CasRequests mount; CasRequests farewell; + CasRequests lease; CasMountRuntime runtime; }; diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp index 788c418b3c2b..f4d3cb258dc5 100644 --- a/src/Disks/tests/gtest_cas_pool.cpp +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -2055,7 +2055,7 @@ class RuntimeRenewBackend final : public DB::Cas::tests::CountingBackend CasRequestBudget runtimeRenewBudget(); -/// A directly-constructed `CasMountRuntime` plus the two request planes it needs. `Pool` builds those +/// A directly-constructed `CasMountRuntime` plus the request planes it needs. `Pool` builds those /// from its own members; a test has no `Pool`, so the mount plane's fence reaches the runtime through /// this holder -- the closures run only once the runtime is issuing requests, well after construction. class RuntimeUnderTest @@ -2068,7 +2068,8 @@ class RuntimeUnderTest [this](uint64_t g, uint64_t needed) { return runtime.admit(g, needed); }, [this](uint64_t g) { runtime.checkFenceOrThrow(g); }}) , farewell(backend, DB::Cas::Fence::open()) - , runtime(backend, mount, farewell, std::forward(args)...) + , lease(backend, DB::Cas::Fence::open()) + , runtime(backend, mount, farewell, lease, std::forward(args)...) { /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the /// budget field alone; every construction of this holder pairs the two via `runtimeRenewBudget`, @@ -2080,6 +2081,9 @@ class RuntimeUnderTest /// deadline against real boottime, finds it long past, and refuses every request unsent. mount.setNowFnForTest([this] { return runtime.bootMsNow(); }); farewell.setNowFnForTest([this] { return runtime.bootMsNow(); }); + lease.setNowFnForTest([this] { return runtime.bootMsNow(); }); + /// As `Pool` wires it: a stop wakes the worker renewal's wait. + lease.setSleepFnForTest([this](uint64_t ms) { runtime.sleepInterruptibly(ms); }); } /// The workers are joined HERE, not only by the tests that assert on teardown: `CasMountRuntime` @@ -2102,6 +2106,7 @@ class RuntimeUnderTest private: DB::Cas::CasRequests mount; DB::Cas::CasRequests farewell; + DB::Cas::CasRequests lease; CasMountRuntime runtime; }; From ff39bb159f6ab390d2465886e50647b18f9f0ef2 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 02:22:41 +0200 Subject: [PATCH 09/31] Renew the CAS mount lease until a definitive answer under UntilDefinitive `MountLeaseRenewer::renew` with `MountRenewPolicy::UntilDefinitive` retries under `Retry::untilDefinitive(kMountRenewRetrySpacingMs)` on the worker plane, with no lease bound, and reports each `PUT` and each failure through `MountRenewOperationEnvironment::on_request`. Classification is unchanged. `LeaseBound` keeps `Retry::untilLeaseSafe`. Co-Authored-By: Claude Opus 5.5 --- .../ContentAddressed/Pool/CasServerRoot.cpp | 34 +- .../ContentAddressed/Pool/CasServerRoot.h | 18 +- src/Disks/tests/gtest_cas_heartbeat.cpp | 627 +++++++++++++++++- 3 files changed, 667 insertions(+), 12 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp index f60e7acdc115..902c569a2731 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp @@ -8,6 +8,7 @@ #include #include #include +#include #include #include #include @@ -1315,6 +1316,20 @@ constexpr uint64_t kFarewellBudgetMs = 10'000; /// "admission-time arithmetic needs room to actually run, not just to pass at t=0". constexpr uint64_t kFarewellSlackMs = 2'000; +namespace +{ +/// The text of a failed request as an operator should read it: a `DB::Exception`'s message, a Poco +/// exception's display text (its `what` is only the class name), any other exception's `what`. +String describeRequestFailure(const std::exception & failure) +{ + if (const auto * db_failure = dynamic_cast(&failure)) + return db_failure->message(); + if (const auto * poco_failure = dynamic_cast(&failure)) + return poco_failure->displayText(); + return failure.what(); +} +} + MountLeaseRenewer::MountLeaseRenewer( CasRequests & mount_requests_, CasRequests & open_requests_, CasRequests & worker_requests_, const Layout & layout_, @@ -1622,11 +1637,22 @@ MountRenewResult MountLeaseRenewer::renewOn( result.attempt_start_boot_ms = attempt_start_boot_ms; CasOperation op = plane.admit(environment.live); + if (environment.on_request) + op.setRequestObserver([&on_request = environment.on_request](uint32_t attempt_no, const std::exception * failure) + { + on_request(MountRenewRequestEvent{ + .request_no = attempt_no, + .failed = failure != nullptr, + .failure_text = failure ? describeRequestFailure(*failure) : String{}, + }); + }); + const Retry policy = environment.policy == MountRenewPolicy::UntilDefinitive + ? Retry::untilDefinitive(kMountRenewRetrySpacingMs) + : Retry::untilLeaseSafe(confirmed_deadline_boot_ms, static_cast(lease_safety_margin.count())); std::optional written; try { - written = op.replace(key, body, precondition(), - Retry::untilLeaseSafe(confirmed_deadline_boot_ms, static_cast(lease_safety_margin.count()))); + written = op.replace(key, body, precondition(), policy); } catch (...) { @@ -1761,8 +1787,8 @@ void MountLeaseRenewer::terminate(CasOperation & op) : doubled_reservation_ms + kFarewellSlackMs; const uint64_t farewell_window_ms = std::max(kFarewellBudgetMs, two_envelope_reservation_plus_slack_ms); /// The derived window alone is not enough: mount-control activity must also never run past the - /// point this node's own fence may already be gone (the same rule `renew` enforces via - /// `Retry::untilLeaseSafe` above). The precondition on this write already stops it from clobbering + /// point this node's own fence may already be gone (the rule a bounded renewal enforces with + /// `Retry::untilLeaseSafe`). The precondition on this write already stops it from clobbering /// a successor if it DOES land late, but a shutdown holding the process open to retry a write past /// its own lease-safe deadline serves no one -- the successor's own reclaim does not wait for it. /// `confirmed_deadline_boot_ms` is set at `start()` and kept current by every successful `renew`, diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h index ac3cc1167ef9..d972841dbe85 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h @@ -549,12 +549,14 @@ bool isCreatorFenceTerminal(CasOperation & op, const Layout & layout, const Stri /// - foreign uuid → fail closed; /// - absent → `create`; expired-our-uuid (any epoch) → `replace` reclaim. /// -/// PLANES. Only the RENEWAL is admitted under the mount fence, because only a renewal writes under -/// authority the fence is tracking. The claim and the farewell are admitted off it: a self-remount -/// claims with the fence already latched lost, so a claim gated on the fence could never reclaim, and -/// a farewell refused because the fence has run down would leave the slot looking live until GC -/// fences it out. Neither is unguarded: a claim's safety is its own conditional write, and a caller -/// that has shutdown facts hands them over as a `Liveness`. +/// PLANES. A bounded renewal is admitted under the mount fence, because it writes under the authority +/// the fence tracks. The worker's renewal (`UntilDefinitive`) runs on a plane with no lease budget: it +/// keeps trying after the lease expired, and a stop, a park or a terminal lifecycle reaches it through +/// its liveness. The claim and the farewell are admitted off the fence: a self-remount claims with the +/// fence already latched lost, so a claim gated on the fence could never reclaim, and a farewell +/// refused because the fence has run down would leave the slot looking live until GC fences it out. +/// Neither is unguarded: a claim's safety is its own conditional write, and a caller that has shutdown +/// facts hands them over as a `Liveness`. class MountLeaseRenewer { public: @@ -573,7 +575,9 @@ class MountLeaseRenewer /// Adopt the already-claimed mount. Returns the exact pre-I/O BOOTTIME anchor. `liveness` carries /// the caller's shutdown terms; the mount fence is deliberately not consulted here. uint64_t start(Liveness liveness = {}); - /// The steady-state renewal, admitted under the mount fence. + /// The steady-state renewal. `LeaseBound` runs under the mount fence with `Retry::untilLeaseSafe`; + /// `UntilDefinitive` runs on the worker plane with `Retry::untilDefinitive(kMountRenewRetrySpacingMs)` + /// and ends only on a definitive answer or when `environment.live` refuses. MountRenewResult renew(const MountRenewOperationEnvironment & environment); /// The remount's re-anchor, which is bootstrap control rather than steady state: a remount renews /// BEFORE it arms the fence for the new incarnation, so the fence is still latched lost and an diff --git a/src/Disks/tests/gtest_cas_heartbeat.cpp b/src/Disks/tests/gtest_cas_heartbeat.cpp index ea20ff3bada3..bcdaf1dcd565 100644 --- a/src/Disks/tests/gtest_cas_heartbeat.cpp +++ b/src/Disks/tests/gtest_cas_heartbeat.cpp @@ -12,7 +12,9 @@ #include #include +#include #include +#include #include #include #include @@ -21,6 +23,8 @@ namespace DB::ErrorCodes { extern const int NETWORK_ERROR; extern const int ABORTED; + extern const int CORRUPTED_DATA; + extern const int FILE_DOESNT_EXIST; } using namespace DB::Cas; @@ -98,6 +102,8 @@ class RenewalScriptBackend : public InMemoryBackend ReturnThenCancel, ThrowBeforeThenLandAfterResolve, ThrowConnectHint, + ThrowFirstAttemptFuse, + ThrowStoreRefusal, }; struct Attempt @@ -111,6 +117,17 @@ class RenewalScriptBackend : public InMemoryBackend std::vector attempts; std::function cancel_after_write; uint64_t read_calls = 0; + /// Consulted when `actions` is empty: TRUE fails the guarded write with `outage_action`. + std::function outage; + Action outage_action = Action::ThrowBefore; + /// Scripted answers for reads of a mount slot: `ThrowBefore` and `ThrowFirstAttemptFuse` fail the + /// read, `Delegate` serves it. Consulted before `read_outage`. + std::deque read_actions; + /// TRUE fails a read of a mount slot with a transport timeout. + std::function read_outage; + /// Called on every scripted write and on every read of a mount slot, before it is answered. + std::function on_attempt; + std::function on_read; /// Only a GUARDED write of a mount slot is scripted; the fixture's own seeding and every other /// key reach the store untouched. @@ -122,9 +139,18 @@ class RenewalScriptBackend : public InMemoryBackend return InMemoryBackend::write(key, bytes, expected_value, access); attempts.push_back({key, bytes, expected_value}); - const Action action = actions.empty() ? Action::Delegate : actions.front(); + if (on_attempt) + on_attempt(); + Action action = Action::Delegate; if (!actions.empty()) + { + action = actions.front(); actions.pop_front(); + } + else if (outage && outage()) + { + action = outage_action; + } if (action == Action::ThrowConnectHint) { @@ -135,6 +161,16 @@ class RenewalScriptBackend : public InMemoryBackend throw Poco::TimeoutException("connect timed out"); #endif } + if (action == Action::ThrowFirstAttemptFuse) + throwFirstAttemptFuse(); + if (action == Action::ThrowStoreRefusal) + { +#if USE_AWS_S3 + throw DB::S3Exception("the store answered MalformedXML", Aws::S3::S3Errors::UNKNOWN, "MalformedXML"); +#else + throw DB::Exception(DB::ErrorCodes::ABORTED, "a store refusal needs USE_AWS_S3"); +#endif + } if (action == Action::ThrowBefore || action == Action::ThrowBeforeThenLandAfterResolve) { @@ -158,6 +194,24 @@ class RenewalScriptBackend : public InMemoryBackend std::optional read(const String & key, TransportAccess & access) override { ++read_calls; + if (key.ends_with("/mount")) + { + if (on_read) + on_read(); + if (!read_actions.empty()) + { + const Action action = read_actions.front(); + read_actions.pop_front(); + if (action == Action::ThrowFirstAttemptFuse) + throwFirstAttemptFuse(); + if (action == Action::ThrowBefore) + throw Poco::TimeoutException("injected renewal read failure"); + } + else if (read_outage && read_outage()) + { + throw Poco::TimeoutException("injected renewal read outage"); + } + } std::optional result = InMemoryBackend::read(key, access); if (pending && pending->key == key) { @@ -171,6 +225,16 @@ class RenewalScriptBackend : public InMemoryBackend } private: + /// The adaptive first-attempt timeout the engine reissues at once. + [[noreturn]] static void throwFirstAttemptFuse() + { +#if USE_AWS_S3 + throw DB::S3Exception("Timeout", Aws::S3::S3Errors::NETWORK_CONNECTION); +#else + throw Poco::TimeoutException("first-attempt fuse"); +#endif + } + std::optional pending; }; @@ -1349,3 +1413,564 @@ TEST(CASHeartbeat, RenewalStopsBeforeTheCutoffWhenEveryAttemptConsumesTheEnvelop EXPECT_EQ(result.outcome, MountRenewOutcome::Terminal); EXPECT_LE(boot_ms, cutoff) << "the last attempt started inside the cutoff and the engine did not start one that could not finish"; } + +namespace +{ +/// The production shape at test scale: a 30 s lease, renewed one 10 s period after its anchor, with a +/// 2 s safety margin and the 7 s attempt envelope a bounded renewal reserves twice before each +/// request. That leaves a bounded renewal about 4 s of retries; an `UntilDefinitive` one has no bound. +class UnboundedRenewalFixture +{ +public: + static constexpr uint64_t ttl_ms = 30'000; + static constexpr uint64_t period_ms = 10'000; + static constexpr uint64_t margin_ms = 2'000; + + UnboundedRenewalFixture() + { + backend->setAttemptTimeoutMs(7'000); + ops = std::make_unique(backend, &boot_ms); + seedOwnClaim(ops->op, layout, "test", UInt128{1}, 9, wall_ms, ttl_ms); + renewer = std::make_unique( + ops->mount, ops->farewell, ops->lease, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(ttl_ms), + [this] { return wall_ms; }, [] { return uint64_t{0}; }, + [this](CasEvent event) { events.push_back(std::move(event)); }, + std::chrono::milliseconds(margin_ms), [this] { return boot_ms; }); + anchor = renewer->start(); + backend->attempts.clear(); + backend->read_calls = 0; + events.clear(); + boot_ms = anchor + period_ms; + renewal_start = boot_ms; + } + + UnboundedRenewalFixture(const UnboundedRenewalFixture &) = delete; + UnboundedRenewalFixture & operator=(const UnboundedRenewalFixture &) = delete; + + MountRenewResult renew( + MountRenewPolicy policy, + const std::function & live = {}, + const std::function & cancelled = {}, + std::function on_request = {}) + { + MountRenewOperationEnvironment environment = renewalEnvironment(boot_ms, live, cancelled); + environment.policy = policy; + environment.on_request = std::move(on_request); + return renewer->renew(environment); + } + + /// The `branch` of every `MountConflict` event, in order. + std::vector conflictBranches() const + { + std::vector branches; + for (const CasEvent & event : events) + if (event.type == CasEventType::MountConflict) + branches.push_back(event.detail.at("branch")); + return branches; + } + + String mountKey() const { return layout.mountKey("test"); } + + std::shared_ptr backend = std::make_shared(); + Layout layout{"pool"}; + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + std::vector events; + std::unique_ptr ops; + std::unique_ptr renewer; + uint64_t anchor = 0; + uint64_t renewal_start = 0; +}; + +/// A request start (`P`) or a resolve read (`R`), with the boot clock when it was issued. +using RequestLog = std::vector>; + +bool inSpacing(uint64_t gap_ms) +{ + return gap_ms >= kMountRenewRetrySpacingMs * 8 / 10 && gap_ms <= kMountRenewRetrySpacingMs * 12 / 10; +} +} + +/// An outage longer than the cutoff a bounded renewal reserves for is retried to its end. +TEST(CASHeartbeat, RenewalOutlivesTheReservationCutoff) +{ + UnboundedRenewalFixture f; + f.backend->outage = [&] { return f.boot_ms < f.renewal_start + 8'000; }; + + const MountRenewResult result = f.renew(MountRenewPolicy::UntilDefinitive); + + ASSERT_EQ(result.outcome, MountRenewOutcome::Committed); + EXPECT_EQ(f.renewer->state(), MountLeaseRenewerState::Active); + /// One request per spacing interval of 800 to 1200 ms until the outage ends, then the one that lands. + EXPECT_GE(f.backend->attempts.size(), 8u); + EXPECT_LE(f.backend->attempts.size(), 11u); + EXPECT_EQ(static_cast(result.attempts_sent), f.backend->attempts.size()); +} + +TEST(CASHeartbeat, RenewalSendsOneTupleOnEveryAttempt) +{ + UnboundedRenewalFixture f; + f.backend->outage = [&] { return f.boot_ms < f.renewal_start + 8'000; }; + + const MountRenewResult result = f.renew(MountRenewPolicy::UntilDefinitive); + + ASSERT_EQ(result.outcome, MountRenewOutcome::Committed); + const auto & attempts = f.backend->attempts; + ASSERT_GT(attempts.size(), 1u); + const UInt128 attempt_id = decodeMountLease(attempts.front().bytes).write_attempt_id; + for (const auto & attempt : attempts) + { + EXPECT_EQ(attempt.key, attempts.front().key); + EXPECT_EQ(attempt.bytes, attempts.front().bytes); + EXPECT_EQ(attempt.expected, attempts.front().expected); + EXPECT_EQ(decodeMountLease(attempt.bytes).write_attempt_id, attempt_id); + } + EXPECT_EQ(f.ops->op.read(f.mountKey(), Retry::standard())->bytes, attempts.front().bytes) + << "the body that landed is the one every attempt carried"; +} + +TEST(CASHeartbeat, RenewalSucceedsPastTheDeadlineWithItsFirstStart) +{ + UnboundedRenewalFixture f; + f.backend->outage = [&] { return f.boot_ms < f.anchor + UnboundedRenewalFixture::ttl_ms + 15'000; }; + + const MountRenewResult result = f.renew(MountRenewPolicy::UntilDefinitive); + + ASSERT_EQ(result.outcome, MountRenewOutcome::Committed); + EXPECT_EQ(result.attempt_start_boot_ms, f.renewal_start); + EXPECT_EQ(f.renewer->lastCommittedAttemptStartBootMs(), f.renewal_start); + EXPECT_LT(result.attempt_start_boot_ms + UnboundedRenewalFixture::ttl_ms, f.boot_ms) + << "the lease this success confirms is already over"; +} + +TEST(CASHeartbeat, LandedAttemptIsAdoptedAfterALongOutage) +{ + UnboundedRenewalFixture f; + f.backend->actions = {RenewalScriptBackend::Action::LandThenThrow}; + f.backend->read_outage = [&] { return f.boot_ms < f.anchor + UnboundedRenewalFixture::ttl_ms + 15'000; }; + std::vector reported; + + const MountRenewResult result = f.renew( + MountRenewPolicy::UntilDefinitive, {}, {}, [&](const MountRenewRequestEvent & event) { reported.push_back(event); }); + const uint64_t reads = f.backend->read_calls; + + ASSERT_EQ(result.outcome, MountRenewOutcome::Committed); + EXPECT_TRUE(result.resolved_by_read); + EXPECT_EQ(result.attempts_sent, 1u); + EXPECT_GT(f.boot_ms, f.anchor + UnboundedRenewalFixture::ttl_ms) << "the adopting read came after the deadline"; + EXPECT_TRUE(f.conflictBranches().empty()) << "an adopted attempt is not a conflict such as same_epoch_state_uncertain"; + EXPECT_EQ(decodeMountLease(f.ops->op.read(f.mountKey(), Retry::standard())->bytes).write_attempt_id, + decodeMountLease(f.backend->attempts.front().bytes).write_attempt_id); + + /// The one PUT sent, its failure, then one failed event per failed read; the last read succeeded. + ASSERT_GE(reads, 2u); + ASSERT_EQ(reported.size(), 2 + (reads - 1)); + EXPECT_FALSE(reported[0].failed); + EXPECT_TRUE(reported[1].failed); + EXPECT_NE(reported[1].failure_text.find("injected renewal response loss after commit"), String::npos) + << reported[1].failure_text; + for (size_t i = 2; i < reported.size(); ++i) + { + EXPECT_EQ(reported[i].request_no, 1u) << i; + EXPECT_TRUE(reported[i].failed) << i << ": no request but the first PUT was sent"; + EXPECT_NE(reported[i].failure_text.find("injected renewal read outage"), String::npos) + << i << ": " << reported[i].failure_text; + } +} + +/// Each definitive answer, met by a request sent past the deadline, ends the renewal with today's +/// classification. +TEST(CASHeartbeat, DefinitiveAnswersStayTerminalPastTheDeadline) +{ + enum class Answer : uint8_t { GcFenced, Foreign, NewerOwnEpoch, OwnEpochOtherBytes, Absent, StoreRefusal, LocalFailure }; + const auto run = [](Answer answer) + { + UnboundedRenewalFixture f; + const String key = f.mountKey(); + const auto replace_slot = [&](const std::function & change) + { + const auto got = f.ops->op.read(key, Retry::standard()); + ASSERT_TRUE(got.has_value()); + MountLease lease = decodeMountLease(got->bytes); + change(lease); + ++lease.seq; + mustCommit(f.ops->op.replace(key, encodeMountLease(lease), got->etag, Retry::standard()), "changed slot"); + }; + switch (answer) + { + case Answer::GcFenced: + replace_slot([](MountLease & lease) { lease.gc_fenced = true; }); + break; + case Answer::Foreign: + replace_slot([](MountLease & lease) { lease.server_uuid = UInt128{2}; }); + break; + case Answer::NewerOwnEpoch: + replace_slot([](MountLease & lease) { lease.writer_epoch = 10; }); + break; + case Answer::OwnEpochOtherBytes: + replace_slot([](MountLease & lease) { lease.write_attempt_id = UInt128{0xAAAA}; }); + break; + case Answer::Absent: + { + const auto got = f.ops->op.read(key, Retry::standard()); + ASSERT_TRUE(got.has_value()); + ASSERT_EQ(f.ops->op.remove(key, got->etag, Retry::standard()), Removal::Removed); + break; + } + case Answer::StoreRefusal: +#if USE_AWS_S3 + f.backend->failNextWriteWith(key, std::make_exception_ptr(DB::S3Exception( + "the store answered MalformedXML", Aws::S3::S3Errors::UNKNOWN, "MalformedXML"))); +#endif + break; + case Answer::LocalFailure: + f.backend->failNextWriteWith(key, std::make_exception_ptr(DB::Exception( + DB::ErrorCodes::CORRUPTED_DATA, "injected deterministic local failure"))); + break; + } + f.backend->attempts.clear(); + f.events.clear(); + /// Past the lease: a bounded renewal would send nothing here. + f.boot_ms = f.anchor + UnboundedRenewalFixture::ttl_ms + 1'000; + + const MountRenewResult result = f.renew(MountRenewPolicy::UntilDefinitive); + + const DB::Exception failure = terminalException(result); + EXPECT_EQ(f.renewer->state(), MountLeaseRenewerState::RenewalTerminal); + EXPECT_EQ(f.backend->attempts.size(), 1u) << "the answer came to a request sent past the deadline"; + const std::vector branches = f.conflictBranches(); + switch (answer) + { + case Answer::GcFenced: + EXPECT_EQ(branches, std::vector{"fenced_by_gc"}); + break; + case Answer::Foreign: + EXPECT_EQ(branches, std::vector{"foreign_writer"}); + break; + case Answer::NewerOwnEpoch: + EXPECT_EQ(branches, std::vector{"superseded"}); + break; + case Answer::OwnEpochOtherBytes: + EXPECT_EQ(branches, std::vector{"same_epoch_state_uncertain"}); + break; + case Answer::Absent: + EXPECT_EQ(branches, std::vector{"vanished"}); + EXPECT_EQ(failure.code(), DB::ErrorCodes::FILE_DOESNT_EXIST) << failure.message(); + break; + case Answer::StoreRefusal: + EXPECT_TRUE(branches.empty()); + EXPECT_NE(failure.message().find("the store refused the renewal"), String::npos) << failure.message(); + break; + case Answer::LocalFailure: + EXPECT_TRUE(branches.empty()); + EXPECT_EQ(failure.code(), DB::ErrorCodes::CORRUPTED_DATA) << failure.message(); + EXPECT_NE(failure.message().find("injected deterministic local failure"), String::npos); + break; + } + }; + for (Answer answer : {Answer::GcFenced, Answer::Foreign, Answer::NewerOwnEpoch, Answer::OwnEpochOtherBytes, + Answer::Absent, Answer::StoreRefusal, Answer::LocalFailure}) + { +#if !USE_AWS_S3 + if (answer == Answer::StoreRefusal) + continue; +#endif + SCOPED_TRACE(static_cast(answer)); + run(answer); + } +} + +/// The cut-off node's story: renewals fail past the deadline, GC on a healthy node fences the slot, +/// and the first request that reaches the store reads the fence. +TEST(CASHeartbeat, AFenceSeenAfterALongOutageEndsTheRenewal) +{ + UnboundedRenewalFixture f; + const uint64_t fenced_at = f.anchor + UnboundedRenewalFixture::ttl_ms + 15'000; + bool fenced = false; + f.backend->outage = [&] { return !fenced; }; + f.ops->lease.setSleepFnForTest([&](uint64_t ms) + { + f.boot_ms += ms; + if (fenced || f.boot_ms < fenced_at) + return; + fenced = true; + const auto got = f.ops->op.read(f.mountKey(), Retry::standard()); + ASSERT_TRUE(got.has_value()); + MountLease lease = decodeMountLease(got->bytes); + lease.gc_fenced = true; + ++lease.seq; + mustCommit(f.ops->op.replace(f.mountKey(), encodeMountLease(lease), got->etag, Retry::standard()), "fence-out"); + }); + + const MountRenewResult result = f.renew(MountRenewPolicy::UntilDefinitive); + + const DB::Exception failure = terminalException(result); + EXPECT_NE(failure.message().find("fenced by GC"), String::npos) << failure.message(); + EXPECT_EQ(f.conflictBranches(), std::vector{"fenced_by_gc"}); + EXPECT_GE(f.boot_ms, fenced_at); +} + +#if USE_AWS_S3 +/// After an unclear attempt the engine cannot tell whether that attempt will still land, so a refusal +/// is not an answer about the slot: the renewal keeps retrying at the spacing until a stop ends it. +TEST(CASHeartbeat, ARefusalAfterAnUnclearAttemptIsRetriedUntilStopped) +{ + UnboundedRenewalFixture f; + f.backend->actions = {RenewalScriptBackend::Action::ThrowBefore}; + f.backend->outage = [] { return true; }; + f.backend->outage_action = RenewalScriptBackend::Action::ThrowStoreRefusal; + std::vector sent_at; + f.backend->on_attempt = [&] { sent_at.push_back(f.boot_ms); }; + bool stopped = false; + f.ops->lease.setSleepFnForTest([&](uint64_t ms) + { + f.boot_ms += ms; + if (f.boot_ms >= f.renewal_start + 30'000) + stopped = true; + }); + std::vector reported; + + const MountRenewResult result = f.renew( + MountRenewPolicy::UntilDefinitive, /*live=*/[&] { return !stopped; }, /*cancelled=*/[&] { return stopped; }, + [&](const MountRenewRequestEvent & event) { reported.push_back(event); }); + + const DB::Exception failure = terminalException(result); + EXPECT_EQ(failure.code(), DB::ErrorCodes::NETWORK_ERROR) << failure.message(); + EXPECT_EQ(f.renewer->state(), MountLeaseRenewerState::RenewalTerminal); + EXPECT_TRUE(f.conflictBranches().empty()) << "the read kept showing our own unchanged body"; + + /// One PUT per spacing interval over the 30 s the store kept refusing. + ASSERT_GE(sent_at.size(), 2u); + for (size_t i = 1; i < sent_at.size(); ++i) + EXPECT_TRUE(inSpacing(sent_at[i] - sent_at[i - 1])) << i << ": " << sent_at[i] - sent_at[i - 1]; + EXPECT_GE(sent_at.size(), 30'000 / (kMountRenewRetrySpacingMs * 12 / 10)); + EXPECT_LE(sent_at.size(), 30'000 / (kMountRenewRetrySpacingMs * 8 / 10) + 1); + + /// Every refusal was reported as a failed request. + size_t refusals = 0; + for (const MountRenewRequestEvent & event : reported) + if (event.failed && event.failure_text.find("MalformedXML") != String::npos) + ++refusals; + EXPECT_EQ(refusals, sent_at.size() - 1) << "every PUT after the unclear first one was refused"; +} +#endif + +TEST(CASHeartbeat, RenewalSpacesRetries) +{ +#if USE_AWS_S3 + { + SCOPED_TRACE("fast connect failures"); + UnboundedRenewalFixture f; + RequestLog log; + f.backend->on_attempt = [&] { log.emplace_back('P', f.boot_ms); }; + f.backend->on_read = [&] { log.emplace_back('R', f.boot_ms); }; + f.backend->actions = {RenewalScriptBackend::Action::ThrowFirstAttemptFuse}; + f.backend->outage = [&] { return f.boot_ms < f.renewal_start + 30'500; }; + f.backend->outage_action = RenewalScriptBackend::Action::ThrowConnectHint; + + ASSERT_EQ(f.renew(MountRenewPolicy::UntilDefinitive).outcome, MountRenewOutcome::Committed); + + ASSERT_GE(log.size(), 4u); + /// The fuse: its settling read and its reissue follow at once. + EXPECT_EQ(log[0], std::make_pair('P', f.renewal_start)); + EXPECT_EQ(log[1], std::make_pair('R', f.renewal_start)); + EXPECT_EQ(log[2], std::make_pair('P', f.renewal_start)); + /// A connect failure is reissued without a read, one spacing interval after it started. + for (size_t i = 3; i < log.size(); ++i) + { + EXPECT_EQ(log[i].first, 'P') << i; + EXPECT_TRUE(inSpacing(log[i].second - log[i - 1].second)) << i << ": " << log[i].second - log[i - 1].second; + } + EXPECT_GE(log.back().second, f.renewal_start + 30'000) << "still renewing after 30 s"; + } +#endif + { + SCOPED_TRACE("unclear PUT with a failing read"); + UnboundedRenewalFixture f; + RequestLog log; + f.backend->on_attempt = [&] { log.emplace_back('P', f.boot_ms); }; + f.backend->on_read = [&] { log.emplace_back('R', f.boot_ms); }; + f.backend->read_actions = {RenewalScriptBackend::Action::ThrowBefore, RenewalScriptBackend::Action::ThrowBefore}; + f.backend->outage = [&] { return f.boot_ms < f.renewal_start + 30'500; }; + + ASSERT_EQ(f.renew(MountRenewPolicy::UntilDefinitive).outcome, MountRenewOutcome::Committed); + + ASSERT_GE(log.size(), 6u); + const uint64_t t0 = f.renewal_start; + EXPECT_EQ(log[0], std::make_pair('P', t0)); + EXPECT_EQ(log[1], std::make_pair('R', t0)) << "the read that settles an unclear PUT is sent at once"; + EXPECT_EQ(log[2].first, 'R'); + EXPECT_TRUE(inSpacing(log[2].second - log[1].second)) << log[2].second - log[1].second; + EXPECT_EQ(log[3].first, 'R'); + EXPECT_TRUE(inSpacing(log[3].second - log[2].second)) << log[3].second - log[2].second; + EXPECT_EQ(log[4], std::make_pair('P', log[3].second)) + << "more than one interval has passed since the PUT started, so it is retried at once"; + uint64_t previous_put = log[4].second; + for (size_t i = 5; i < log.size(); ++i) + { + if (log[i].first == 'R') + { + EXPECT_EQ(log[i].second, log[i - 1].second) << i; + continue; + } + EXPECT_TRUE(inSpacing(log[i].second - previous_put)) << i << ": " << log[i].second - previous_put; + previous_put = log[i].second; + } + EXPECT_GE(previous_put, t0 + 30'000) << "still renewing after 30 s"; + } +#if USE_AWS_S3 + { + SCOPED_TRACE("first-attempt fuse of the read"); + UnboundedRenewalFixture f; + RequestLog log; + f.backend->on_attempt = [&] { log.emplace_back('P', f.boot_ms); }; + f.backend->on_read = [&] { log.emplace_back('R', f.boot_ms); }; + f.backend->actions = {RenewalScriptBackend::Action::ThrowBefore}; + f.backend->read_actions = {RenewalScriptBackend::Action::ThrowFirstAttemptFuse}; + + ASSERT_EQ(f.renew(MountRenewPolicy::UntilDefinitive).outcome, MountRenewOutcome::Committed); + + ASSERT_EQ(log.size(), 4u); + EXPECT_EQ(log[1], std::make_pair('R', f.renewal_start)); + EXPECT_EQ(log[2], std::make_pair('R', f.renewal_start)) << "a fused read is reissued at once"; + EXPECT_EQ(log[3].first, 'P'); + EXPECT_TRUE(inSpacing(log[3].second - log[0].second)) << log[3].second - log[0].second; + } +#endif + { + SCOPED_TRACE("slow requests"); + UnboundedRenewalFixture f; + RequestLog log; + bool failing = false; + f.backend->on_attempt = [&] + { + log.emplace_back('P', f.boot_ms); + failing = f.boot_ms < f.renewal_start + 30'500; + if (failing) + f.boot_ms += 5'000; + }; + f.backend->on_read = [&] { log.emplace_back('R', f.boot_ms); }; + f.backend->outage = [&] { return failing; }; + + ASSERT_EQ(f.renew(MountRenewPolicy::UntilDefinitive).outcome, MountRenewOutcome::Committed); + + ASSERT_GE(log.size(), 3u); + std::optional previous_put; + for (size_t i = 0; i < log.size(); ++i) + { + if (log[i].first == 'R') + { + EXPECT_EQ(log[i].second, log[i - 1].second + 5'000) << i; + continue; + } + if (previous_put) + EXPECT_EQ(log[i].second, *previous_put + 5'000) << i << ": a 5 s request is retried with no wait"; + previous_put = log[i].second; + } + EXPECT_GE(*previous_put, f.renewal_start + 30'000) << "still renewing after 30 s"; + } +} + +TEST(CASHeartbeat, StopDuringARetryWaitEndsTheRenewal) +{ + { + SCOPED_TRACE("a stop during the wait"); + UnboundedRenewalFixture f; + bool stopped = false; + std::vector waits; + f.backend->outage = [] { return true; }; + /// The stop wakes the wait, so the clock does not move. + f.ops->lease.setSleepFnForTest([&](uint64_t ms) + { + waits.push_back(ms); + stopped = true; + }); + + const MountRenewResult result = f.renew( + MountRenewPolicy::UntilDefinitive, /*live=*/[&] { return !stopped; }, /*cancelled=*/[&] { return stopped; }); + + const DB::Exception failure = terminalException(result); + EXPECT_EQ(failure.code(), DB::ErrorCodes::NETWORK_ERROR) << failure.message(); + ASSERT_EQ(waits.size(), 1u); + EXPECT_TRUE(inSpacing(waits[0])) << waits[0]; + EXPECT_EQ(f.backend->attempts.size(), 1u) << "nothing is sent after the stop"; + EXPECT_EQ(f.renewer->state(), MountLeaseRenewerState::RenewalTerminal); + EXPECT_FALSE(f.renewer->canRelease()); + } + { + SCOPED_TRACE("a stop after the renewal began, before its first send"); + UnboundedRenewalFixture f; + + const MountRenewResult result = f.renew( + MountRenewPolicy::UntilDefinitive, /*live=*/[] { return false; }, /*cancelled=*/[] { return false; }); + + (void)terminalException(result); + EXPECT_TRUE(f.backend->attempts.empty()); + EXPECT_FALSE(f.renewer->canRelease()); + } +} + +TEST(CASHeartbeat, RenewalReportsEachRequestAsItHappens) +{ + UnboundedRenewalFixture f; + f.backend->actions = {RenewalScriptBackend::Action::ThrowBefore, RenewalScriptBackend::Action::ThrowBefore}; + std::vector reported; + + const MountRenewResult result = f.renew( + MountRenewPolicy::UntilDefinitive, {}, {}, [&](const MountRenewRequestEvent & event) { reported.push_back(event); }); + + ASSERT_EQ(result.outcome, MountRenewOutcome::Committed); + const std::vector> expected{{1, false}, {1, true}, {2, false}, {2, true}, {3, false}}; + ASSERT_EQ(reported.size(), expected.size()); + for (size_t i = 0; i < expected.size(); ++i) + { + EXPECT_EQ(reported[i].request_no, expected[i].first) << i; + EXPECT_EQ(reported[i].failed, expected[i].second) << i; + if (reported[i].failed) + EXPECT_NE(reported[i].failure_text.find("injected renewal response uncertainty"), String::npos) + << reported[i].failure_text; + else + EXPECT_TRUE(reported[i].failure_text.empty()) << i; + } +} + +TEST(CASHeartbeat, AThrowingRequestReportChangesNoOutcome) +{ + UnboundedRenewalFixture f; + f.backend->actions = {RenewalScriptBackend::Action::ThrowBefore, RenewalScriptBackend::Action::ThrowBefore}; + + const MountRenewResult result = f.renew( + MountRenewPolicy::UntilDefinitive, {}, {}, + [](const MountRenewRequestEvent &) { throw std::runtime_error("injected report failure"); }); + + ASSERT_EQ(result.outcome, MountRenewOutcome::Committed); + EXPECT_EQ(result.attempts_sent, 3u); + EXPECT_EQ(f.renewer->state(), MountLeaseRenewerState::Active); +} + +/// The startup, remount and direct renewals keep their lease bound. +TEST(CASHeartbeat, BoundedPathsStillStopAtTheLeaseDeadline) +{ + for (const bool remount : {false, true}) + { + SCOPED_TRACE(remount ? "renewForRemount" : "renew"); + UnboundedRenewalFixture f; + std::vector sent_at; + f.backend->on_attempt = [&] { sent_at.push_back(f.boot_ms); }; + /// Longer than any lease, so only the bound can end the renewal. + f.backend->outage = [&] { return f.boot_ms < f.renewal_start + 60'000; }; + MountRenewOperationEnvironment environment = renewalEnvironment(f.boot_ms); + environment.policy = MountRenewPolicy::LeaseBound; + + const MountRenewResult result = remount ? f.renewer->renewForRemount(environment) : f.renewer->renew(environment); + + const DB::Exception failure = terminalException(result); + EXPECT_NE(failure.message().find("external_lease_deadline"), String::npos) << failure.message(); + ASSERT_TRUE(result.deadline_source.has_value()); + EXPECT_EQ(*result.deadline_source, GaveUp::Source::Lease); + ASSERT_FALSE(sent_at.empty()); + const uint64_t lease_safe = f.anchor + UnboundedRenewalFixture::ttl_ms - UnboundedRenewalFixture::margin_ms; + for (uint64_t at : sent_at) + EXPECT_LT(at, lease_safe) << "no request starts past the lease-safe bound"; + } +} From 01ffe71e8b09a85ee261db70951382ce043527c0 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 02:32:47 +0200 Subject: [PATCH 10/31] Run the worker's CAS mount renewal without a lease deadline The background worker renews with `MountRenewPolicy::UntilDefinitive`: it retries one second apart on the lease plane until the store answers, and a stop, a park or a terminal lifecycle still ends it. Startup, remount and direct renewals stay `LeaseBound`. Three worker tests that ended a renewal by moving the clock to the deadline now end it with a fenced slot, which is still terminal. Co-Authored-By: Claude Opus 5.5 --- .../ContentAddressed/Pool/CasMountRuntime.cpp | 4 +- .../ContentAddressed/Pool/CasMountRuntime.h | 5 +- src/Disks/tests/gtest_cas_pool.cpp | 282 +++++++++++++++++- 3 files changed, 273 insertions(+), 18 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index 62f7c045c705..1f84591d54d6 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -431,7 +431,9 @@ MountRenewOperationEnvironment CasMountRuntime::renewalEnvironment(bool worker_c return config.renewal_live_for_test ? config.renewal_live_for_test() : renewalLive(worker_call); }, .cancelled = [this] { return renewalCancelled(); }, - .policy = MountRenewPolicy::LeaseBound, + /// Only the worker keeps renewing past the lease; startup, remount and direct renewals stay + /// bounded by it. + .policy = worker_call ? MountRenewPolicy::UntilDefinitive : MountRenewPolicy::LeaseBound, .on_request = {}, }; } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index c0175613c678..1c151e2601e9 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -447,8 +447,9 @@ class CasMountRuntime void renewalLoop(); void remountLoop(); ThreadFromGlobalPool makeWorker(std::function body); - /// The renewal's liveness: facts the mount fence cannot see -- a shutdown request, a parked or - /// park-requested driver, a pool that left `Live`. FALSE ends the renewal. + /// The renewal's liveness: a shutdown request and, for the worker, a parked or park-requested + /// driver, a pool that left `Live`, or a lost fence -- the worker's plane has no fence of its own. + /// FALSE ends the renewal. bool renewalLive(bool worker_call) const; /// Whether this node has already been asked to stop. Sampled ONCE, before the write, so a refusal /// caused by the stop cannot be mistaken for one that preceded it. diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp index f4d3cb258dc5..cda43519cd66 100644 --- a/src/Disks/tests/gtest_cas_pool.cpp +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -2020,6 +2020,12 @@ class RuntimeRenewBackend final : public DB::Cas::tests::CountingBackend /// rather than reissued has to move the injected clock here -- from inside the attempt, which is the /// only point between admission and the resolve read a test can reach. std::function before_throw; + /// While it answers TRUE, every conditional write throws a transport timeout, before `fault` is + /// consulted. Read on the renewing thread; set it before the workers start. + std::function outage; + /// Runs on every write `outage` fails, before it throws. + std::function on_outage_write; + std::atomic outage_writes{0}; /// The fault sits on the WRITE PRIMITIVE, and only on a CONDITIONAL one: a lease renewal is a /// replace, so a create on the same key must not consume the one-shot fault. @@ -2028,6 +2034,13 @@ class RuntimeRenewBackend final : public DB::Cas::tests::CountingBackend { if (!expected_value) return DB::Cas::tests::CountingBackend::write(key, bytes, expected_value, access); + if (outage && outage()) + { + outage_writes.fetch_add(1, std::memory_order_relaxed); + if (on_outage_write) + on_outage_write(); + throw Poco::TimeoutException("injected runtime renewal outage"); + } const Fault current = std::exchange(fault, Fault::None); if (current == Fault::BlockThenDelegate || current == Fault::BlockThenThrow) { @@ -2103,6 +2116,14 @@ class RuntimeUnderTest CasMountRuntime & operator*() { return runtime; } + /// The retry wait of the mount and lease planes, the two a renewal can run on. Call before the + /// workers start. + void setRetrySleepForTest(const std::function & sleep_fn) + { + mount.setSleepFnForTest(sleep_fn); + lease.setSleepFnForTest(sleep_fn); + } + private: DB::Cas::CasRequests mount; DB::Cas::CasRequests farewell; @@ -4075,12 +4096,8 @@ TEST(CASPoolRemount, TerminalDepositionDoesNotTouchRenewerAfterReplacement) runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); - backend->fault = RuntimeRenewBackend::Fault::ThrowBefore; - /// Expire the lease from inside the attempt. The fault alone no longer ends a renewal: the engine - /// settles the ambiguity by reading and then reissues, and the reissue commits. With the clock past - /// the deadline the renewal was admitted under, neither the settling read nor the reissue is - /// admitted, so the renewal ends terminal -- which is what this test deposits. - backend->before_throw = [&, deadline = anchor + 1000] { boot_ms = deadline; }; + /// A definitive answer ends the worker's renewal; a transient fault would only be retried. + fenceOutMount(*backend, layout.mountKey("test")); runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); terminal_deposited.waitUntilArrived(); EXPECT_TRUE(replaced.load(std::memory_order_acquire)); @@ -4165,11 +4182,9 @@ TEST(CASPoolRemount, ImmediatePostRemountRenewalFailureIsNotDropped) runtime_ptr->armMountFence(uuid, 2, fresh_anchor + 10'000); runtime_ptr->noteRemounted(); boot_ms = 2'000; - backend->fault = RuntimeRenewBackend::Fault::ThrowBefore; - /// Expire the fresh lease from inside the attempt, so the ambiguity can be neither - /// settled by a read nor reissued: otherwise the engine reissues and the renewal - /// commits, and there is no dropped failure to catch up on. - backend->before_throw = [&, deadline = fresh_anchor + 10'000] { boot_ms = deadline; }; + /// A definitive answer for the fresh incarnation's first worker renewal; a transient + /// fault would only be retried. + fenceOutMount(*backend, layout.mountKey("test")); first.arriveAndWait(); return true; } @@ -4597,10 +4612,8 @@ TEST(CASPool, DeterministicWorkerFailureFencesWithoutWaitingForCadence) runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); - backend->fault = RuntimeRenewBackend::Fault::ThrowBefore; - /// Expire the lease from inside the attempt, so the ambiguity can be neither settled by a read nor - /// reissued: without that the engine reissues and the renewal commits, and this worker never fences. - backend->before_throw = [&, deadline = anchor + 1000] { boot_ms = deadline; }; + /// A definitive answer ends the worker's renewal; a transient fault would only be retried. + fenceOutMount(*backend, layout.mountKey("test")); runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); remount_entered.waitUntilArrived(); EXPECT_FALSE(runtime.mayMutate()); @@ -4610,6 +4623,245 @@ TEST(CASPool, DeterministicWorkerFailureFencesWithoutWaitingForCadence) runtime.finishTeardown(false); } +/// A remount request ends a worker renewal that is retrying past its lease, inside the wait or the +/// request it is in: no request and no wait starts after it. +TEST(CASMountRuntime, ParkEndsAnUnboundedRenewal) +{ + enum class ParkDuring : uint8_t { Wait, Request }; + const auto run = [](ParkDuring park_during) + { + auto backend = std::make_shared(); + const Layout layout(park_during == ParkDuring::Wait ? "unbounded-park-in-wait" : "unbounded-park-in-request"); + const UInt128 uuid{1}; + uint64_t wall_ms = 1000; + std::atomic boot_ms{100}; + std::atomic past_the_lease{std::numeric_limits::max()}; + std::atomic held{false}; + std::atomic waits{0}; + std::atomic writes_at_remount{0}; + std::atomic waits_at_remount{0}; + DB::Cas::tests::ManualBarrier holding; + DB::Cas::tests::ManualBarrier remount_entered; + const auto hold_once_past_the_lease = [&] + { + if (boot_ms.load() >= past_the_lease.load() && !held.exchange(true)) + holding.arriveAndWait(); + }; + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, + MountClaimResult::Claimed); + CasEventSink sink; + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms.load(); }}, + "test", sink, runtimeRenewBudget(), [&] + { + writes_at_remount = backend->outage_writes.load(); + waits_at_remount = waits.load(); + remount_entered.arriveAndWait(); + return false; + }); + CasMountRuntime & runtime = *runtime_holder; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + runtime.armMountFence(uuid, 1, anchor + 1000); + /// Five lease lengths on: a renewal bounded by its lease ended long before. + past_the_lease = anchor + 5'000; + runtime_holder.setRetrySleepForTest([&](uint64_t ms) + { + ++waits; + boot_ms += ms; + if (park_during == ParkDuring::Wait) + hold_once_past_the_lease(); + }); + if (park_during == ParkDuring::Request) + backend->on_outage_write = hold_once_past_the_lease; + backend->outage = [] { return true; }; + runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + + holding.waitUntilArrived(); + const uint64_t writes_at_park = backend->outage_writes.load(); + const uint64_t waits_at_park = waits.load(); + runtime.scheduleRemount(); + EXPECT_EQ(runtime.renewalDriverStateForTest(), RenewalDriverState::ParkRequested); + holding.release(); + remount_entered.waitUntilArrived(); + EXPECT_EQ(runtime.renewalDriverStateForTest(), RenewalDriverState::Parked); + EXPECT_EQ(writes_at_remount.load(), writes_at_park) << "no request starts after the park"; + EXPECT_EQ(waits_at_remount.load(), waits_at_park) << "no wait starts after the park"; + remount_entered.release(); + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); + }; + run(ParkDuring::Wait); + run(ParkDuring::Request); +} + +/// FORGET while the worker's renewal retries past its lease: the intent then the trip, in the order +/// `Pool::forgetDisk` uses, end the renewal inside the wait it is in, both workers exit, and no remount +/// generation is raised. +TEST(CASMountRuntime, ForgetEndsAnUnboundedRenewal) +{ + auto backend = std::make_shared(); + const Layout layout("unbounded-forget"); + const UInt128 uuid{1}; + uint64_t wall_ms = 1000; + std::atomic boot_ms{100}; + std::atomic past_the_lease{std::numeric_limits::max()}; + std::atomic held{false}; + std::atomic waits{0}; + DB::Cas::tests::ManualBarrier holding; + WorkerExitLatch exits; + RuntimeWorkerFactory factory = [&](std::function worker_body) + { + return ThreadFromGlobalPool([&, body = std::move(worker_body)] + { + body(); + exits.recordExit(); + }); + }; + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, + MountClaimResult::Claimed); + CasEventSink sink; + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms.load(); }, .worker_factory = factory}, + "test", sink, runtimeRenewBudget(), [] { return false; }); + CasMountRuntime & runtime = *runtime_holder; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + runtime.armMountFence(uuid, 1, anchor + 1000); + past_the_lease = anchor + 5'000; + runtime_holder.setRetrySleepForTest([&](uint64_t ms) + { + ++waits; + boot_ms += ms; + if (boot_ms.load() >= past_the_lease.load() && !held.exchange(true)) + holding.arriveAndWait(); + }); + backend->outage = [] { return true; }; + runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + + holding.waitUntilArrived(); + const uint64_t writes_at_forget = backend->outage_writes.load(); + const uint64_t waits_at_forget = waits.load(); + const uint64_t generation_at_forget = runtime.remountRequestedGenerationForTest(); + runtime.publishVanishedIntent(); + runtime.tripMountLost(); + holding.release(); + + const bool both_exited = exits.waitForAtLeast(2); + EXPECT_TRUE(both_exited) << "the worker loops must exit on the published intent"; + EXPECT_EQ(backend->outage_writes.load(), writes_at_forget) << "no request starts after FORGET"; + EXPECT_EQ(waits.load(), waits_at_forget) << "no wait starts after FORGET"; + EXPECT_EQ(runtime.remountRequestedGenerationForTest(), generation_at_forget); + EXPECT_FALSE(runtime.mayMutate()); + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); +} + +TEST(CASMountRuntime, StopWakesTheRetryWaitOfAnUnboundedRenewal) +{ + auto backend = std::make_shared(); + const Layout layout("unbounded-stop-in-wait"); + const UInt128 uuid{1}; + uint64_t wall_ms = 1000; + const uint64_t boot_ms = 100; + std::promise wait_entered; + std::future wait_requested = wait_entered.get_future(); + std::atomic first_wait{true}; + std::atomic first_wait_slept_ms{-1}; + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, + MountClaimResult::Claimed); + CasEventSink sink; + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }}, + "test", sink, runtimeRenewBudget(), [] { return false; }); + CasMountRuntime & runtime = *runtime_holder; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + runtime.armMountFence(uuid, 1, anchor + 1000); + runtime_holder.setRetrySleepForTest([&](uint64_t ms) + { + const bool first = first_wait.exchange(false); + if (first) + wait_entered.set_value(ms); + const auto started = std::chrono::steady_clock::now(); + runtime.sleepInterruptibly(ms); + if (first) + first_wait_slept_ms = std::chrono::duration_cast( + std::chrono::steady_clock::now() - started).count(); + }); + backend->outage = [] { return true; }; + runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + + ASSERT_EQ(wait_requested.wait_for(std::chrono::seconds(20)), std::future_status::ready); + const uint64_t requested_ms = wait_requested.get(); + runtime.stopBackgroundWorkers(); + + EXPECT_GE(requested_ms, kMountRenewRetrySpacingMs * 8 / 10); + EXPECT_LE(requested_ms, kMountRenewRetrySpacingMs * 12 / 10); + ASSERT_GE(first_wait_slept_ms.load(), 0); + EXPECT_LT(static_cast(first_wait_slept_ms.load()), requested_ms) + << "the stop woke the wait instead of letting it run out"; + EXPECT_EQ(backend->outage_writes.load(), 1u) << "nothing is sent after the stop"; + runtime.finishTeardown(false); +} + +/// A success whose lease is already over is followed by the next renewal with no cadence wait. +TEST(CASMountRuntime, AStaleSuccessIsFollowedAtOnceByTheNextRenewal) +{ + auto backend = std::make_shared(); + const Layout layout("unbounded-stale-success"); + const UInt128 uuid{1}; + uint64_t wall_ms = 1000; + std::atomic boot_ms{100}; + std::atomic outage_until{0}; + std::vector commit_boot_ms; + std::vector may_mutate_at_commit; + DB::Cas::tests::ManualBarrier second_commit; + CasMountRuntime * runtime_ptr = nullptr; + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, + MountClaimResult::Claimed); + CasEventSink sink; + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms.load(); }}, + "test", sink, runtimeRenewBudget(), [] { return false; }); + CasMountRuntime & runtime = *runtime_holder; + runtime_ptr = &runtime; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + runtime.armMountFence(uuid, 1, anchor + 1000); + runtime_holder.setRetrySleepForTest([&](uint64_t ms) { boot_ms += ms; }); + /// The first renewal starts one period after the anchor and fails for three lease lengths. + boot_ms = anchor + 500; + outage_until = anchor + 3'500; + backend->outage = [&] { return boot_ms.load() < outage_until.load(); }; + backend->after_commit = [&] + { + commit_boot_ms.push_back(boot_ms.load()); + may_mutate_at_commit.push_back(runtime_ptr->mayMutate()); + if (commit_boot_ms.size() == 2) + second_commit.arriveAndWait(); + }; + runtime.startBackgroundWorkers(std::chrono::milliseconds(500)); + + second_commit.waitUntilArrived(); + ASSERT_EQ(commit_boot_ms.size(), 2u); + EXPECT_GE(commit_boot_ms[0], anchor + 3'500) << "the first renewal outlived its own lease"; + EXPECT_EQ(commit_boot_ms[1], commit_boot_ms[0]) + << "no time passed: a cadence wait on this frozen clock would never have ended"; + EXPECT_FALSE(may_mutate_at_commit[1]) << "the stale success left the lease expired"; + second_commit.release(); + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); +} + TEST(CASPool, RenewWatermarkOnceRefreshesFenceAndDepositsOneFailure) { auto backend = std::make_shared(); From 73ed84b979b14bb5ccb3eab43b16ce85c7d9369b Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 02:43:21 +0200 Subject: [PATCH 11/31] Derive the expired state of a CAS mount lease The lease is expired while the pool is Live, the fence is not lost and the confirmed deadline has passed. Add the accessor and the CASMountLeaseExpired counter that the restoring renewal will bump. Co-Authored-By: Claude Opus 5.5 --- src/Common/ProfileEvents.cpp | 1 + .../ContentAddressed/Pool/CasMountRuntime.cpp | 22 ++++++++++++++++ .../ContentAddressed/Pool/CasMountRuntime.h | 16 ++++++++++++ src/Disks/tests/gtest_cas_mount_runtime.cpp | 25 +++++++++++++++++++ 4 files changed, 64 insertions(+) diff --git a/src/Common/ProfileEvents.cpp b/src/Common/ProfileEvents.cpp index a9986b5aa485..5cb260b8006c 100644 --- a/src/Common/ProfileEvents.cpp +++ b/src/Common/ProfileEvents.cpp @@ -958,6 +958,7 @@ The server successfully detected this situation and will download merged part fr M(CASMountRenewalRecovered, "Number of logical CAS mount-lease renewals that committed after a physical retry or exact resolving GET.", ValueType::Number) \ M(CASMountRenewalDeadlineExceeded, "Number of logical CAS mount-lease renewals stopped by the external lease-safety deadline. Growth means the last confirmed lease no longer had enough safe time for another physical attempt.", ValueType::Number) \ M(CASMountLeaseLost, "Counts exactly once per operational CAS mount-lease Live-to-TransientNotLive loss/recovery generation. The initiating external loss or the first ordinary terminal renewal consumer owns the increment, including external lease-safety deadline exhaustion; parked/classification/shutdown paths do not duplicate it.", ValueType::Number) \ + M(CASMountLeaseExpired, "Number of times a renewal restored a CAS mount lease that had expired. While the lease is expired this server refuses writes and system.cas_mounts shows lifecycle_reason = 'lease_expired'; the watermark_renew event of the restoring renewal carries expired_ms.", ValueType::Number) \ M(CASRemountAttempts, "Number of invocations of the CAS whole-chain remount attempt.", ValueType::Number) \ M(CASRemountSucceeded, "Number of CAS whole-chain remount attempts that restored Live under a fresh writer epoch.", ValueType::Number) \ M(CASRemountFailed, "Number of CAS whole-chain remount attempts that returned without restoring Live, including caught step exceptions.", ValueType::Number) \ diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index 1f84591d54d6..b642e9c06365 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -165,6 +165,27 @@ bool CasMountRuntime::refAppendFenceOk() const return admit(fenceGeneration(), needed_ms) == Fence::Admit::Ok; } +std::optional CasMountRuntime::leaseExpiredAt(uint64_t now_boot_ms) const +{ + if (lifecycle() != PoolLifecycle::Live || mount_fence.lost.load(std::memory_order_acquire)) + return std::nullopt; + const uint64_t deadline = mount_fence.deadline_boot_ms.load(std::memory_order_acquire); + if (now_boot_ms < deadline) + return std::nullopt; + return std::min(deadline, lease_expired_at_boot_ms.load(std::memory_order_acquire)); +} + +std::optional CasMountRuntime::leaseExpiredSinceBootMs() const +{ + return leaseExpiredAt(bootMsNow()); +} + +String CasMountRuntime::lastRenewFailure() const +{ + std::lock_guard lock(renew_failure_mutex); + return last_renew_failure; +} + void CasMountRuntime::setMountDeadline(uint64_t deadline_boot_ms) { mount_fence.deadline_boot_ms.store(deadline_boot_ms, std::memory_order_release); @@ -175,6 +196,7 @@ void CasMountRuntime::armMountFence(UInt128 server_uuid, uint64_t writer_epoch, mount_fence.server_uuid = server_uuid; mount_fence.writer_epoch = writer_epoch; mount_fence.deadline_boot_ms.store(deadline_boot_ms, std::memory_order_release); + lease_expired_at_boot_ms.store(std::numeric_limits::max(), std::memory_order_release); /// A fresh lease incarnation is a fresh generation too: a durable-effect caller admitted under the /// PRIOR incarnation must re-check and abort rather than ride this re-arm through (rev.7 [C2]). fence_generation.fetch_add(1, std::memory_order_acq_rel); diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index 1c151e2601e9..c5b35adf75fb 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -16,6 +16,7 @@ #include #include #include +#include #include namespace DB::Cas @@ -311,6 +312,15 @@ class CasMountRuntime /// it cannot plausibly finish before the fence expires. bool refAppendFenceOk() const; + /// ---- lease expiry ---- + /// The instant this server's confirmed lease expired, on the fence clock, while it stays expired: + /// the lifecycle is `Live`, the fence is not lost and `bootMsNow` has reached the deadline. A + /// renewal that commits with a start more than a TTL ago does not end the expiry, so its start is + /// kept until a renewal restores the lease. Empty otherwise. + std::optional leaseExpiredSinceBootMs() const; + /// Text of the last failed request of the worker's renewal; empty when none failed. + String lastRenewFailure() const; + /// TRUE once the pool has reached — or is being driven toward — a state on which the self-remount /// worker must stop: a published terminal `Vanished` intent (`vanished_intent` — set early by /// FORGET, or by a natural `enterVanished`, and already subsuming every settled `Vanished*` state since @@ -439,6 +449,7 @@ class CasMountRuntime bool propagate_failure, bool worker_call); MountRenewOperationEnvironment renewalEnvironment(bool worker_call); + std::optional leaseExpiredAt(uint64_t now_boot_ms) const; void consumeRenewResult( const MountRenewResult & result, RenewalDriverState active_state, @@ -524,6 +535,11 @@ class CasMountRuntime /// `fenceGeneration`/`checkFenceOrThrow`. std::atomic fence_generation{0}; std::function arm_mount_fence_interposition_hook_for_test; + /// The first expired deadline of an expiry that a renewal committed past did not end. `UINT64_MAX` + /// when there is none. Written by the renewal consumer and by `armMountFence`. + std::atomic lease_expired_at_boot_ms{std::numeric_limits::max()}; + mutable std::mutex renew_failure_mutex; + String last_renew_failure; /// The pool lifecycle condition (rev.7 §1). Starts `Live`. Non-terminal transitions /// (`noteLeaseLost`/`noteRemounted`) are lock-free compare-exchanges guarded by their exact diff --git a/src/Disks/tests/gtest_cas_mount_runtime.cpp b/src/Disks/tests/gtest_cas_mount_runtime.cpp index 8ff8a4e4af64..3c516034460e 100644 --- a/src/Disks/tests/gtest_cas_mount_runtime.cpp +++ b/src/Disks/tests/gtest_cas_mount_runtime.cpp @@ -176,3 +176,28 @@ TEST(CASMountRuntime, RefAppendFenceOkIsAdmitAtTwoEnvelopesWithANonzeroCap) EXPECT_FALSE(f->refAppendFenceOk()); EXPECT_STREQ(admitName(f->admit(f->fenceGeneration(), 400)), "NoBudget"); } + +/// An expiry is this server's own confirmed deadline passing while nothing else is wrong. A lost fence or +/// a lifecycle that left `Live` is a different state and reports as itself. +TEST(CASMountRuntime, LeaseExpiredOnlyWhileLiveAndNotLost) +{ + RuntimeFixture f(/*lease_safety_margin_ms=*/0); + f.boot_ms = 1'000; + EXPECT_FALSE(f->leaseExpiredSinceBootMs().has_value()) << "an unarmed fence has no deadline to pass"; + + f->armMountFence(kUuid, 1, /*deadline_boot_ms=*/1'100); + f.boot_ms = 1'099; + EXPECT_FALSE(f->leaseExpiredSinceBootMs().has_value()); + f.boot_ms = 1'100; + ASSERT_TRUE(f->leaseExpiredSinceBootMs().has_value()) << "the deadline instant is already past, as `admit` reads it"; + EXPECT_EQ(*f->leaseExpiredSinceBootMs(), 1'100u); + EXPECT_TRUE(f->lastRenewFailure().empty()) << "no renewal request has failed"; + + f->setLifecycleForTest(PoolLifecycle::IdentityLost); + EXPECT_FALSE(f->leaseExpiredSinceBootMs().has_value()) << "only a `Live` pool reports an expiry"; + f->setLifecycleForTest(PoolLifecycle::Live); + ASSERT_TRUE(f->leaseExpiredSinceBootMs().has_value()); + + f->tripMountLost(); + EXPECT_FALSE(f->leaseExpiredSinceBootMs().has_value()) << "a lost fence is a lease loss, not an expiry"; +} From 08e824282218b9089df02a97ea852b9a3a296900 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 02:47:45 +0200 Subject: [PATCH 12/31] Report a CAS mount lease expiry and its restore The worker's renewal counts its requests as they are sent and keeps the last failure text. A renewal that restores an expired lease bumps CASMountLeaseExpired, logs a warning with the duration and the last failure, and puts expired_ms into its watermark_renew event. A renewal that commits with a start already more than a TTL ago keeps the expiry and when it began. Co-Authored-By: Claude Opus 5.5 --- .../ContentAddressed/Pool/CasMountRuntime.cpp | 68 ++- .../ContentAddressed/Pool/CasMountRuntime.h | 6 + .../ContentAddressed/Pool/CasServerRoot.cpp | 28 +- src/Disks/tests/gtest_cas_event_log.cpp | 4 +- src/Disks/tests/gtest_cas_pool.cpp | 399 ++++++++++++++++++ 5 files changed, 484 insertions(+), 21 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index b642e9c06365..4b7c737bc284 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -25,6 +25,7 @@ namespace ProfileEvents extern const Event CASIdentityLost; extern const Event CASDataRootVanished; extern const Event CASMountLeaseLost; + extern const Event CASMountLeaseExpired; extern const Event CASMountRenewalAttempts; extern const Event CASMountRenewalRetries; extern const Event CASMountRenewalResolved; @@ -35,7 +36,7 @@ namespace ProfileEvents namespace DB::Cas { -void reportMountRenewCompletion(const MountRenewResult & result) noexcept; +void reportMountRenewCompletion(const MountRenewResult & result, std::optional expired_ms) noexcept; void configureMountRenewObservability( const String * server_root_id, const CasEventSink * event_sink, bool deferred) noexcept; @@ -186,6 +187,42 @@ String CasMountRuntime::lastRenewFailure() const return last_renew_failure; } +void CasMountRuntime::noteRenewRequest(const MountRenewRequestEvent & event) noexcept +{ + if (!event.failed) + { + ProfileEvents::incrementNoTrace(ProfileEvents::CASMountRenewalAttempts); + if (event.request_no > 1) + ProfileEvents::incrementNoTrace(ProfileEvents::CASMountRenewalRetries); + return; + } + try + { + std::lock_guard lock(renew_failure_mutex); + last_renew_failure = event.failure_text; + } + catch (...) // NOLINT(bugprone-empty-catch) + { + /// A lost diagnostic must not end the renewal. + } +} + +std::optional CasMountRuntime::publishRenewedDeadline(uint64_t deadline_boot_ms) +{ + const uint64_t now = bootMsNow(); + const std::optional expired_at = leaseExpiredAt(now); + setMountDeadline(deadline_boot_ms); + if (expired_at && deadline_boot_ms <= now) + { + lease_expired_at_boot_ms.store(*expired_at, std::memory_order_release); + return std::nullopt; + } + lease_expired_at_boot_ms.store(std::numeric_limits::max(), std::memory_order_release); + if (!expired_at) + return std::nullopt; + return now - *expired_at; +} + void CasMountRuntime::setMountDeadline(uint64_t deadline_boot_ms) { mount_fence.deadline_boot_ms.store(deadline_boot_ms, std::memory_order_release); @@ -456,7 +493,11 @@ MountRenewOperationEnvironment CasMountRuntime::renewalEnvironment(bool worker_c /// Only the worker keeps renewing past the lease; startup, remount and direct renewals stay /// bounded by it. .policy = worker_call ? MountRenewPolicy::UntilDefinitive : MountRenewPolicy::LeaseBound, - .on_request = {}, + /// The worker counts its requests as they are sent, so an outage shows while it lasts. + .on_request = worker_call + ? std::function( + [this](const MountRenewRequestEvent & event) { noteRenewRequest(event); }) + : nullptr, }; } @@ -492,9 +533,9 @@ void CasMountRuntime::consumeRenewResult( { /// Driver ownership has already been restored by `DriverLease::finish`; this is the single logical /// consumption boundary and it runs without `driver_mutex` or renewer access. - /// The physical counters come off the result rather than off a per-attempt callback, so they count - /// the same on every ending: a renewal that gave up still sent what it sent. - if (result.attempts_sent > 0) + /// The worker's renewal counts its requests as they are sent (`noteRenewRequest`). Every other + /// renewal counts them here, from its result. + if (active_state != RenewalDriverState::WorkerCall && result.attempts_sent > 0) { ProfileEvents::incrementNoTrace(ProfileEvents::CASMountRenewalAttempts, result.attempts_sent); ProfileEvents::incrementNoTrace(ProfileEvents::CASMountRenewalRetries, result.attempts_sent - 1); @@ -511,16 +552,24 @@ void CasMountRuntime::consumeRenewResult( if (result.outcome == MountRenewOutcome::Committed) { const uint64_t ttl_ms = static_cast(config.mount_lease_ttl_ms.count()); - setMountDeadline( + const std::optional expired_ms = publishRenewedDeadline( result.attempt_start_boot_ms > std::numeric_limits::max() - ttl_ms ? std::numeric_limits::max() : result.attempt_start_boot_ms + ttl_ms); - reportMountRenewCompletion(result); + if (expired_ms) + { + ProfileEvents::incrementNoTrace(ProfileEvents::CASMountLeaseExpired); + LOG_WARNING(getLogger("CasPool"), + "CAS mount lease of '{}' was expired for {} ms; a renewal restored it and writes resume. " + "Last failed renewal request: {}", + server_root_id, *expired_ms, lastRenewFailure()); + } + reportMountRenewCompletion(result, expired_ms); return; } if (result.outcome == MountRenewOutcome::NotAttempted) { - reportMountRenewCompletion(result); + reportMountRenewCompletion(result, std::nullopt); return; } @@ -539,9 +588,8 @@ void CasMountRuntime::consumeRenewResult( { } - reportMountRenewCompletion(result); + reportMountRenewCompletion(result, std::nullopt); - (void)active_state; (void)returned_state; if (propagate_failure) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index c5b35adf75fb..99126c275cc0 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -320,6 +320,9 @@ class CasMountRuntime std::optional leaseExpiredSinceBootMs() const; /// Text of the last failed request of the worker's renewal; empty when none failed. String lastRenewFailure() const; + /// Counts each `PUT` of the worker's renewal as it is sent and keeps the text of every failed + /// `PUT` or resolve read. Runs on the renewing thread. + void noteRenewRequest(const MountRenewRequestEvent & event) noexcept; /// TRUE once the pool has reached — or is being driven toward — a state on which the self-remount /// worker must stop: a published terminal `Vanished` intent (`vanished_intent` — set early by @@ -450,6 +453,9 @@ class CasMountRuntime bool worker_call); MountRenewOperationEnvironment renewalEnvironment(bool worker_call); std::optional leaseExpiredAt(uint64_t now_boot_ms) const; + /// Publishes a committed renewal's deadline. Returns how long the lease had been expired when this + /// deadline restores it; empty when it was not expired or is still expired. + std::optional publishRenewedDeadline(uint64_t deadline_boot_ms); void consumeRenewResult( const MountRenewResult & result, RenewalDriverState active_state, diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp index 902c569a2731..9b08fc516f29 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp @@ -46,7 +46,7 @@ namespace ErrorCodes namespace DB::Cas { -void reportMountRenewCompletion(const MountRenewResult & result) noexcept; +void reportMountRenewCompletion(const MountRenewResult & result, std::optional expired_ms) noexcept; void configureMountRenewObservability( const String * server_root_id, const CasEventSink * event_sink, bool deferred) noexcept; void deliverDeferredMountRenewObservability(uint64_t remount_attempt_no) noexcept; @@ -132,6 +132,8 @@ struct MountRenewObservabilityContext MountRenewTerminalClassification terminal_classification = MountRenewTerminalClassification::Unclassified; uint32_t attempts_sent = 0; bool resolved_by_read = false; + /// How long the lease had been expired when this renewal restored it; empty unless it did. + std::optional expired_ms = std::nullopt; }; static_assert(std::is_trivially_copyable_v); @@ -294,9 +296,12 @@ void emitMountRenewEvent( CasEvent event; event.type = CasEventType::WatermarkRenew; event.outcome = String{outcome}; - event.reason = outcome == "recovered" - ? "CAS mount renewal recovered before its confirmed lease-safety deadline" - : "CAS mount renewal ended without retained authority and fenced the mount"; + if (outcome != "recovered") + event.reason = "CAS mount renewal ended without retained authority and fenced the mount"; + else if (context.expired_ms) + event.reason = "CAS mount renewal restored a lease that had expired"; + else + event.reason = "CAS mount renewal committed after a retry or a resolving read"; event.detail = { {"server_root_id", *context.server_root_id}, {"writer_epoch", std::to_string(context.writer_epoch)}, @@ -309,6 +314,8 @@ void emitMountRenewEvent( }; if (remount_attempt_no != 0) event.detail["remount_attempt_no"] = std::to_string(remount_attempt_no); + if (context.expired_ms) + event.detail["expired_ms"] = std::to_string(*context.expired_ms); (*context.event_sink)(std::move(event)); } catch (...) @@ -363,12 +370,14 @@ void deliverMountRenewObservability( } const bool recovered = context.outcome == MountRenewOutcome::Committed - && (context.attempts_sent > 1 || context.resolved_by_read); + && (context.attempts_sent > 1 || context.resolved_by_read || context.expired_ms.has_value()); if (recovered) { - const std::string_view classification = context.resolved_by_read - ? "committed_by_read" - : "committed_after_retry"; + std::string_view classification = "committed_after_expiry"; + if (context.resolved_by_read) + classification = "committed_by_read"; + else if (context.attempts_sent > 1) + classification = "committed_after_retry"; emitMountRenewEvent( context, write_attempt_id, @@ -464,7 +473,7 @@ void configureMountRenewObservability( }; } -void reportMountRenewCompletion(const MountRenewResult & result) noexcept +void reportMountRenewCompletion(const MountRenewResult & result, std::optional expired_ms) noexcept { if (mount_renew_observability.suppressed_depth != 0) { @@ -478,6 +487,7 @@ void reportMountRenewCompletion(const MountRenewResult & result) noexcept context->outcome = result.outcome; context->attempts_sent = std::max(context->attempts_sent, result.attempts_sent); context->resolved_by_read = result.resolved_by_read; + context->expired_ms = expired_ms; if (context->deferred) return; diff --git a/src/Disks/tests/gtest_cas_event_log.cpp b/src/Disks/tests/gtest_cas_event_log.cpp index 79cde4da3f8f..49fd515323d3 100644 --- a/src/Disks/tests/gtest_cas_event_log.cpp +++ b/src/Disks/tests/gtest_cas_event_log.cpp @@ -29,7 +29,7 @@ namespace DB::Cas { void configureMountRenewObservability( const String * server_root_id, const CasEventSink * event_sink, bool deferred) noexcept; -void reportMountRenewCompletion(const MountRenewResult & result) noexcept; +void reportMountRenewCompletion(const MountRenewResult & result, std::optional expired_ms) noexcept; } namespace @@ -345,7 +345,7 @@ TEST(CASEvent, DeepReentrancyPreservesDeterministicPhysicalAttemptTruth) { configureMountRenewObservability(&server_root_ids[index], &sinks[index], /*deferred=*/false); MountRenewResult result = renewers[index]->renew(MountRenewOperationEnvironment{}); - reportMountRenewCompletion(result); + reportMountRenewCompletion(result, std::nullopt); return result; }; diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp index cda43519cd66..9b8d8171d048 100644 --- a/src/Disks/tests/gtest_cas_pool.cpp +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -53,6 +53,9 @@ extern const Event CASMountReleaseSkippedForeignOccupant; extern const Event CASRemountAttempts; extern const Event CASRemountSucceeded; extern const Event CASRemountFailed; +extern const Event CASMountLeaseExpired; +extern const Event CASMountRenewalAttempts; +extern const Event CASMountRenewalRetries; } using namespace DB::Cas; @@ -5127,3 +5130,399 @@ TEST(CASPool, ConcurrentNamespaceCreationsNeverRaceEachOtherOnTheCatalog) << "the next hold starts from what the resolve read saw"; EXPECT_EQ(backend->writeCount(key) - writes_mid, 3u) << "one refused, two landed"; } + +namespace +{ + +/// Fails the guarded mount `PUT`s that `on_put` says to fail, with a timeout raised before the store +/// applied anything, and the reads of the mount slot that `on_read` says to fail. Both run on the +/// renewing thread with the 1-based number of the request of their kind. +class ExpiryScriptBackend final : public DB::Cas::tests::CountingBackend +{ +public: + std::function on_put; + std::function on_read; + + std::optional read(const String & key, DB::Cas::TransportAccess & access) override + { + if (on_read && key.ends_with("/mount") && on_read(++reads)) + throw Poco::TimeoutException("injected resolve read timeout"); + return DB::Cas::tests::CountingBackend::read(key, access); + } + + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, DB::Cas::TransportAccess & access) override + { + if (on_put && expected_value && key.ends_with("/mount") && on_put(++puts)) + throw Poco::TimeoutException("injected renewal timeout before the store applied it"); + return DB::Cas::tests::CountingBackend::write(key, bytes, expected_value, access); + } + +private: + uint32_t puts = 0; + uint32_t reads = 0; +}; + +constexpr uint64_t kExpiryTtlMs = 30'000; +constexpr uint64_t kExpiryPeriodMs = 10'000; +constexpr uint64_t kExpiryClaimBootMs = 100'000; +constexpr uint64_t kExpiryDeadlineBootMs = kExpiryClaimBootMs + kExpiryTtlMs; +constexpr uint64_t kExpiryFirstStartBootMs = kExpiryClaimBootMs + kExpiryPeriodMs; +/// Longer than the largest retry-spacing draw, so a failed request is retried at once. +constexpr uint64_t kExpiryFailedPutMs = 1'300; +/// Enough failures to carry the first renewal past its own start + TTL. +constexpr uint32_t kExpiryFailedPuts = 24; +constexpr uint64_t kExpiryRestoreBootMs = kExpiryFirstStartBootMs + kExpiryFailedPuts * kExpiryFailedPutMs; +/// A request sent while the lease is expired and the first renewal still retries. +constexpr uint32_t kExpiryObservedPut = 20; +static_assert(kExpiryFirstStartBootMs + (kExpiryObservedPut - 1) * kExpiryFailedPutMs > kExpiryDeadlineBootMs); +static_assert(kExpiryRestoreBootMs > kExpiryFirstStartBootMs + kExpiryTtlMs); + +const char * expiryAdmitName(Fence::Admit verdict) +{ + switch (verdict) + { + case Fence::Admit::Ok: return "Ok"; + case Fence::Admit::LostOrRearmed: return "LostOrRearmed"; + case Fence::Admit::NoBudget: return "NoBudget"; + } + return "unknown"; +} + +uint64_t eventCount(ProfileEvents::Event event) +{ + return ProfileEvents::global_counters[event].load(); +} + +struct ExpiryObservation +{ + uint64_t generation_before = 0; + uint64_t attempts_before = 0; + uint64_t retries_before = 0; + uint64_t lease_expired_before = 0; + uint64_t lease_lost_before = 0; + + /// Inside `PUT` number `kExpiryObservedPut`. + Fence::Admit admit_during = Fence::Admit::Ok; + bool may_mutate_during = true; + PoolLifecycle lifecycle_during = PoolLifecycle::TransientNotLive; + std::optional expired_since_during; + String last_failure_during; + uint64_t attempts_during = 0; + uint64_t retries_during = 0; + + /// At the loop pass after the renewal that committed with a start more than a TTL ago. + Fence::Admit admit_after_stale = Fence::Admit::Ok; + std::optional expired_since_after_stale; + uint64_t lease_expired_after_stale = 0; + uint64_t boot_ms_after_stale = 0; + uint64_t boot_ms_at_next_admission = 0; + + /// At the loop pass after the restoring renewal. + Fence::Admit admit_after_restore = Fence::Admit::LostOrRearmed; + std::optional expired_since_after_restore; + PoolLifecycle lifecycle_after_restore = PoolLifecycle::TransientNotLive; + uint64_t generation_after_restore = 0; + uint64_t lease_expired_after_restore = 0; + uint64_t lease_lost_after_restore = 0; + uint64_t attempts_after_restore = 0; + uint64_t retries_after_restore = 0; + + uint64_t writer_epoch_on_store = 0; + uint32_t remount_calls = 0; + std::vector events; + String log; +}; + +void runExpiryScenario(const String & layout_prefix, ExpiryObservation & seen) +{ + /// Everything the hooks capture is declared before the backend and the runtime that store them. + const Layout layout(layout_prefix); + const UInt128 uuid{1}; + uint64_t wall_ms = 1000; + std::atomic boot_ms{kExpiryClaimBootMs}; + std::atomic remount_calls{0}; + uint32_t loop_passes = 0; + DB::Cas::tests::ManualBarrier third_pass; + CasMountRuntime * runtime_ptr = nullptr; + auto events = std::make_shared(); + CasEventSink sink = [events](CasEvent event) { events->push(std::move(event)); }; + ScopedRemountLogCapture log_capture; + auto backend = std::make_shared(); + + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, kExpiryTtlMs).kind, + MountClaimResult::Claimed); + + const auto observe_after_stale = [&] + { + CasMountRuntime & runtime = *runtime_ptr; + seen.admit_after_stale = runtime.admit(seen.generation_before, 0); + seen.expired_since_after_stale = runtime.leaseExpiredSinceBootMs(); + seen.lease_expired_after_stale = eventCount(ProfileEvents::CASMountLeaseExpired) - seen.lease_expired_before; + seen.boot_ms_after_stale = boot_ms.load(); + }; + const auto observe_after_restore = [&] + { + CasMountRuntime & runtime = *runtime_ptr; + seen.admit_after_restore = runtime.admit(seen.generation_before, 0); + seen.expired_since_after_restore = runtime.leaseExpiredSinceBootMs(); + seen.lifecycle_after_restore = runtime.lifecycle(); + seen.generation_after_restore = runtime.fenceGeneration(); + seen.lease_expired_after_restore = eventCount(ProfileEvents::CASMountLeaseExpired) - seen.lease_expired_before; + seen.lease_lost_after_restore = eventCount(ProfileEvents::CASMountLeaseLost) - seen.lease_lost_before; + seen.attempts_after_restore = eventCount(ProfileEvents::CASMountRenewalAttempts) - seen.attempts_before; + seen.retries_after_restore = eventCount(ProfileEvents::CASMountRenewalRetries) - seen.retries_before; + }; + + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{ + .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), + .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms.load(); }, + .renewal_before_driver_lock_hook_for_test = [&] + { + ++loop_passes; + if (loop_passes == 1) + { + boot_ms.store(kExpiryFirstStartBootMs); + } + else if (loop_passes == 2) + { + observe_after_stale(); + } + else if (loop_passes == 3) + { + observe_after_restore(); + third_pass.arriveAndWait(); + } + }, + .renewal_admitted_hook_for_test = [&] + { + if (loop_passes == 2) + seen.boot_ms_at_next_admission = boot_ms.load(); + }}, + "test", sink, runtimeRenewBudget(), [&] + { + ++remount_calls; + return false; + }); + CasMountRuntime & runtime = *runtime_holder; + runtime_ptr = &runtime; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + ASSERT_EQ(anchor, kExpiryClaimBootMs); + runtime.armMountFence(uuid, 1, anchor + kExpiryTtlMs); + + seen.generation_before = runtime.fenceGeneration(); + seen.attempts_before = eventCount(ProfileEvents::CASMountRenewalAttempts); + seen.retries_before = eventCount(ProfileEvents::CASMountRenewalRetries); + seen.lease_expired_before = eventCount(ProfileEvents::CASMountLeaseExpired); + seen.lease_lost_before = eventCount(ProfileEvents::CASMountLeaseLost); + + backend->on_put = [&](uint32_t put_no) + { + if (put_no == kExpiryObservedPut) + { + CasMountRuntime & renewing = *runtime_ptr; + seen.admit_during = renewing.admit(seen.generation_before, 0); + seen.may_mutate_during = renewing.mayMutate(); + seen.lifecycle_during = renewing.lifecycle(); + seen.expired_since_during = renewing.leaseExpiredSinceBootMs(); + seen.last_failure_during = renewing.lastRenewFailure(); + seen.attempts_during = eventCount(ProfileEvents::CASMountRenewalAttempts) - seen.attempts_before; + seen.retries_during = eventCount(ProfileEvents::CASMountRenewalRetries) - seen.retries_before; + } + if (put_no > kExpiryFailedPuts) + return false; + boot_ms.fetch_add(kExpiryFailedPutMs); + return true; + }; + + runtime.startBackgroundWorkers(std::chrono::milliseconds(kExpiryPeriodMs)); + third_pass.waitUntilArrived(); + + seen.writer_epoch_on_store = decodeMountLease(readObj(*backend, layout.mountKey("test"))->bytes).writer_epoch; + seen.remount_calls = remount_calls.load(); + seen.events = events->snapshot(); + seen.log = log_capture.captured(); + + third_pass.release(); + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); +} + +/// One renewal whose first `PUT` is unclear and whose resolve reads then fail long enough for the lease +/// to expire; the 19th read finds the slot unchanged and the reissued `PUT` lands. +constexpr uint32_t kReadOutageFailedReads = 18; +constexpr uint32_t kReadOutageObservedRead = 17; +static_assert(kExpiryFirstStartBootMs + (kReadOutageObservedRead - 1) * kExpiryFailedPutMs > kExpiryDeadlineBootMs); + +struct ReadOutageObservation +{ + uint64_t attempts_before = 0; + uint64_t retries_before = 0; + /// Inside read number `kReadOutageObservedRead`. + std::optional expired_since_during; + String last_failure_during; + uint64_t attempts_during = 0; + uint64_t retries_during = 0; + /// At the loop pass after the renewal. + uint64_t attempts_after = 0; + uint64_t retries_after = 0; +}; + +void runReadOutageScenario(const String & layout_prefix, ReadOutageObservation & seen) +{ + const Layout layout(layout_prefix); + const UInt128 uuid{1}; + uint64_t wall_ms = 1000; + std::atomic boot_ms{kExpiryClaimBootMs}; + uint32_t loop_passes = 0; + DB::Cas::tests::ManualBarrier second_pass; + CasMountRuntime * runtime_ptr = nullptr; + CasEventSink sink; + auto backend = std::make_shared(); + + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, kExpiryTtlMs).kind, + MountClaimResult::Claimed); + + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{ + .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), + .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms.load(); }, + .renewal_before_driver_lock_hook_for_test = [&] + { + ++loop_passes; + if (loop_passes == 1) + { + boot_ms.store(kExpiryFirstStartBootMs); + } + else if (loop_passes == 2) + { + seen.attempts_after = eventCount(ProfileEvents::CASMountRenewalAttempts) - seen.attempts_before; + seen.retries_after = eventCount(ProfileEvents::CASMountRenewalRetries) - seen.retries_before; + second_pass.arriveAndWait(); + } + }}, + "test", sink, runtimeRenewBudget(), [] { return false; }); + CasMountRuntime & runtime = *runtime_holder; + runtime_ptr = &runtime; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + runtime.armMountFence(uuid, 1, anchor + kExpiryTtlMs); + + seen.attempts_before = eventCount(ProfileEvents::CASMountRenewalAttempts); + seen.retries_before = eventCount(ProfileEvents::CASMountRenewalRetries); + backend->on_put = [](uint32_t put_no) { return put_no == 1; }; + backend->on_read = [&](uint32_t read_no) + { + if (read_no == kReadOutageObservedRead) + { + seen.expired_since_during = runtime_ptr->leaseExpiredSinceBootMs(); + seen.last_failure_during = runtime_ptr->lastRenewFailure(); + seen.attempts_during = eventCount(ProfileEvents::CASMountRenewalAttempts) - seen.attempts_before; + seen.retries_during = eventCount(ProfileEvents::CASMountRenewalRetries) - seen.retries_before; + } + if (read_no > kReadOutageFailedReads) + return false; + boot_ms.fetch_add(kExpiryFailedPutMs); + return true; + }; + + runtime.startBackgroundWorkers(std::chrono::milliseconds(kExpiryPeriodMs)); + second_pass.waitUntilArrived(); + second_pass.release(); + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); +} + +std::vector renewEventsOf(const std::vector & events) +{ + std::vector renewals; + for (const CasEvent & event : events) + if (event.type == CasEventType::WatermarkRenew) + renewals.push_back(event); + return renewals; +} + +} + +/// An expiry refuses writes without touching the fence, a renewal that commits with a start more than a +/// TTL ago restores nothing and is followed at once, and the next one restores writes under the same +/// epoch and generation. Runs the real worker loop. +TEST(CASMountRuntime, ExpiryRefusesWritesAndResumesUnderTheSameEpoch) +{ + ExpiryObservation seen; + ASSERT_NO_FATAL_FAILURE(runExpiryScenario("runtime-expiry-resume", seen)); + + EXPECT_STREQ(expiryAdmitName(seen.admit_during), "NoBudget") << "refused, and not because the fence is lost"; + EXPECT_FALSE(seen.may_mutate_during); + EXPECT_EQ(seen.lifecycle_during, PoolLifecycle::Live); + + EXPECT_STREQ(expiryAdmitName(seen.admit_after_stale), "NoBudget") + << "a renewal whose start + TTL is already past restores nothing"; + /// The clock moves only when the test moves it, so a renewal admitted at all after the stale success + /// was admitted without a cadence wait; a wait would have left loop pass 3 unreached. + EXPECT_EQ(seen.boot_ms_at_next_admission, seen.boot_ms_after_stale); + + EXPECT_STREQ(expiryAdmitName(seen.admit_after_restore), "Ok"); + EXPECT_EQ(seen.lifecycle_after_restore, PoolLifecycle::Live); + EXPECT_EQ(seen.generation_after_restore, seen.generation_before) << "an expiry is not a re-arm"; + EXPECT_EQ(seen.lease_lost_after_restore, 0u); + EXPECT_EQ(seen.writer_epoch_on_store, 1u) << "the same epoch holds the slot"; + EXPECT_EQ(seen.remount_calls, 0u); +} + +/// What an operator sees: the expiry and its last failure while it lasts, the attempt counters moving +/// during the outage, and one counter, event field and warning when a renewal restores the lease. +TEST(CASMountRuntime, ExpiryIsReported) +{ + ExpiryObservation seen; + ASSERT_NO_FATAL_FAILURE(runExpiryScenario("runtime-expiry-report", seen)); + + ASSERT_TRUE(seen.expired_since_during.has_value()); + EXPECT_EQ(*seen.expired_since_during, kExpiryDeadlineBootMs); + EXPECT_NE(seen.last_failure_during.find("injected renewal timeout"), String::npos) << seen.last_failure_during; + /// The request being sent is counted before it reaches the store. + EXPECT_EQ(seen.attempts_during, kExpiryObservedPut) << "attempts advance as requests are sent"; + EXPECT_EQ(seen.retries_during, kExpiryObservedPut - 1); + + ASSERT_TRUE(seen.expired_since_after_stale.has_value()); + EXPECT_EQ(*seen.expired_since_after_stale, kExpiryDeadlineBootMs) + << "a renewal committed past its own start + TTL keeps the expiry and when it began"; + EXPECT_EQ(seen.lease_expired_after_stale, 0u) << "that renewal is not a restore"; + + EXPECT_FALSE(seen.expired_since_after_restore.has_value()); + EXPECT_EQ(seen.lease_expired_after_restore, 1u); + /// 25 requests by the first renewal and one by the second, each counted once. + EXPECT_EQ(seen.attempts_after_restore, kExpiryFailedPuts + 2); + EXPECT_EQ(seen.retries_after_restore, kExpiryFailedPuts); + + const std::vector renewals = renewEventsOf(seen.events); + ASSERT_EQ(renewals.size(), 2u) << "the retried renewal and the restoring one"; + EXPECT_FALSE(renewals[0].detail.contains("expired_ms")) << "the first renewal left the lease expired"; + EXPECT_EQ(renewals[0].detail.at("classification"), "committed_after_retry"); + EXPECT_EQ(renewals[1].outcome, "recovered"); + EXPECT_EQ(renewals[1].detail.at("classification"), "committed_after_expiry"); + ASSERT_TRUE(renewals[1].detail.contains("expired_ms")); + EXPECT_EQ(renewals[1].detail.at("expired_ms"), std::to_string(kExpiryRestoreBootMs - kExpiryDeadlineBootMs)); + + EXPECT_NE(seen.log.find(fmt::format("expired for {} ms", kExpiryRestoreBootMs - kExpiryDeadlineBootMs)), String::npos) + << seen.log; + EXPECT_NE(seen.log.find("injected renewal timeout"), String::npos) << "the warning names the last failure: " << seen.log; + + /// An unclear `PUT` whose resolve reads then fail: the reads are what the operator needs to see, + /// and they are not requests of the renewal. + ReadOutageObservation reads; + ASSERT_NO_FATAL_FAILURE(runReadOutageScenario("runtime-expiry-read-outage", reads)); + ASSERT_TRUE(reads.expired_since_during.has_value()) << "the read outage outlasted the lease, so the row shows this text"; + EXPECT_NE(reads.last_failure_during.find("injected resolve read timeout"), String::npos) + << "lifecycle_detail carries the last failed read: " << reads.last_failure_during; + EXPECT_EQ(reads.attempts_during, 1u) << "failed reads do not advance the attempt counter"; + EXPECT_EQ(reads.retries_during, 0u); + EXPECT_EQ(reads.attempts_after, 2u) << "the unclear PUT and its reissue"; + EXPECT_EQ(reads.retries_after, 1u); +} From 45381f7af5e07ba2dd02d9c8a699bbb7df9077a9 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 02:50:27 +0200 Subject: [PATCH 13/31] Show an expired CAS mount lease in system.cas_mounts While the lease is expired and the pool is Live, the row reads lifecycle = 'not_live', lifecycle_reason = 'lease_expired', lifecycle_since = the expired deadline as wall time and lifecycle_detail = the last failed renewal request. The PoolLifecycle enum and the state column are unchanged. Co-Authored-By: Claude Opus 5.5 --- .../ContentAddressedMetadataStorage.cpp | 4 +-- .../ContentAddressedMetadataStorage.h | 13 +++++--- .../ContentAddressed/Pool/CasPool.cpp | 18 ++++++++++ .../ContentAddressed/Pool/CasPool.h | 3 ++ .../tests/gtest_cas_lifecycle_snapshot.cpp | 33 +++++++++++++++++++ .../StorageSystemContentAddressedMounts.cpp | 6 ++-- 6 files changed, 67 insertions(+), 10 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.cpp index b20d299df60c..24b7c95f0bc5 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.cpp @@ -465,8 +465,8 @@ CasLifecycleSnapshot ContentAddressedMetadataStorage::lifecycleSnapshot() const } const Cas::Pool::LifecycleSnapshot ps = pool->lifecycleSnapshot(); - snap.lifecycle = casLifecycleToString(ps.lifecycle); - snap.reason = casLifecycleReasonWord(ps.lifecycle); + snap.lifecycle = ps.lease_expired ? "not_live" : casLifecycleToString(ps.lifecycle); + snap.reason = ps.lease_expired ? "lease_expired" : casLifecycleReasonWord(ps.lifecycle); snap.detail = ps.detail; snap.since = ps.since; return snap; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.h index 5f0aadb2d7d4..be5d5299e504 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.h @@ -94,14 +94,17 @@ enum class CasOpAdmission : uint8_t /// operator instead of silently missing from the table. /// - `lifecycle` — one of `live` / `not_live` / `identity_lost` / `vanished` (a live pool), or /// `constructing` / `shutdown` (no pool published). -/// - `reason` — the ENUM-CLEAN sub-state word: `replaced` / `forgotten` for a -/// `vanished` pool, empty otherwise. Kept a small closed vocabulary so a downstream +/// - `reason` — the ENUM-CLEAN sub-state word: `replaced` / `forgotten` for a `vanished` pool, +/// `lease_expired` for a `not_live` row whose lifecycle enum is still `Live` but +/// whose lease expired, empty otherwise. A small closed vocabulary, so a downstream /// `lifecycle || '(' || reason || ')'` yields e.g. exactly `vanished(forgotten)` — /// the [D5] free text lives in `detail`, never here. /// - `detail` — the full [D5] reason text naming the actual failure (the replaced -/// diagnosis, the timestamped `FORGET` message, or the identity-loss message); spec §1 -/// requires it appear verbatim in the snapshot. Empty while `live` and for a null pool. -/// - `since` — wall-clock second the current non-`live` state was entered; 0 while `live`/no pool. +/// diagnosis, the timestamped `FORGET` message, or the identity-loss message), verbatim. +/// For `lease_expired`, the text of the last failed renewal request. Empty while +/// `live` and for a null pool. +/// - `since` — wall-clock second the current non-`live` state was entered (for `lease_expired`, +/// the confirmed deadline that passed); 0 while `live`/no pool. /// - `pool_id` — last-known pool UUID (empty before the first `startup`); the disk stays /// introspectable under its identity even once the pool is gone. /// - `server_root_id` — this server's node-local root id owning the mount slot. diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp index 98009fe3a20f..afcaf4cefbfd 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp @@ -77,6 +77,16 @@ namespace std::atomic remount_attempt_sequence{0}; +/// The wall-clock second of a past `CLOCK_BOOTTIME` instant: wall now minus how long ago it was on the +/// boot clock. Both clocks are this server's own, sampled at one moment. +time_t wallSecondOfPastBootInstant(uint64_t past_boot_ms, uint64_t now_boot_ms) +{ + const int64_t wall_now_ms = std::chrono::duration_cast( + std::chrono::system_clock::now().time_since_epoch()).count(); + const uint64_t ago_ms = now_boot_ms > past_boot_ms ? now_boot_ms - past_boot_ms : 0; + return static_cast((wall_now_ms - static_cast(ago_ms)) / 1000); +} + /// Validate every writable factory's lease/request/cadence relationship before it can publish /// owner, epoch, mount, or probe authority. Decommission forces background renewal before calling /// this helper, so it is held to the same cadence window as an ordinary production mount. @@ -380,6 +390,14 @@ Pool::LifecycleSnapshot Pool::lifecycleSnapshot() const snap.lifecycle = mount_runtime.lifecycle(); snap.detail = lifecycleReasonDetail(snap.lifecycle); snap.since = mount_runtime.lifecycleSinceWallS(); + if (snap.lifecycle != PoolLifecycle::Live) + return snap; + const std::optional expired_at = mount_runtime.leaseExpiredSinceBootMs(); + if (!expired_at) + return snap; + snap.lease_expired = true; + snap.detail = mount_runtime.lastRenewFailure(); + snap.since = wallSecondOfPastBootInstant(*expired_at, mount_runtime.bootMsNow()); return snap; } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h index 7f1151341437..81ff3423ba31 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h @@ -543,6 +543,9 @@ class Pool : public std::enable_shared_from_this PoolLifecycle lifecycle = PoolLifecycle::Live; String detail; time_t since = 0; + /// The lifecycle is `Live` but this server's lease expired and no renewal has restored it yet. + /// `detail` is then the last failed renewal request and `since` the expired deadline. + bool lease_expired = false; }; LifecycleSnapshot lifecycleSnapshot() const; diff --git a/src/Disks/tests/gtest_cas_lifecycle_snapshot.cpp b/src/Disks/tests/gtest_cas_lifecycle_snapshot.cpp index ea504c894641..e9b9499eca8b 100644 --- a/src/Disks/tests/gtest_cas_lifecycle_snapshot.cpp +++ b/src/Disks/tests/gtest_cas_lifecycle_snapshot.cpp @@ -7,6 +7,7 @@ #include #include +#include #include #include #include @@ -235,3 +236,35 @@ TEST(CASLifecycleSnapshot, NaturalIdentityLostMatchesThrowDetail) << "snapshot detail and the typed error must not drift\n detail: " << snap.detail << "\n thrown: " << thrown; } + +/// An expired lease, while the lifecycle enum stays `Live`: `not_live` / `lease_expired`, `since` is the +/// deadline as wall time, and a deadline back in the future returns the row to `live`. +TEST(CASLifecycleSnapshot, ExpiredLeaseIsNotLiveUntilRestored) +{ + auto storage = openSnapshotStorage(); + commitOnePart(*storage); + auto pool = storage->store(); + const auto wall_now_s = [] + { + return static_cast(std::chrono::duration_cast( + std::chrono::system_clock::now().time_since_epoch()).count()); + }; + + const int64_t before_s = wall_now_s(); + pool->setMountDeadline(pool->bootMsNow() - 5'000); + const CasLifecycleSnapshot expired = storage->lifecycleSnapshot(); + const int64_t after_s = wall_now_s(); + + EXPECT_EQ(expired.lifecycle, "not_live"); + EXPECT_EQ(expired.reason, "lease_expired"); + EXPECT_GE(static_cast(expired.since), before_s - 6) << "the deadline passed about 5 s ago"; + EXPECT_LE(static_cast(expired.since), after_s - 4); + EXPECT_TRUE(expired.detail.empty()) << "no renewal request failed, so there is no failure text: " << expired.detail; + EXPECT_EQ(pool->lifecycle(), PoolLifecycle::Live) << "the lifecycle enum is not touched"; + + pool->setMountDeadline(pool->bootMsNow() + 60'000); + const CasLifecycleSnapshot restored = storage->lifecycleSnapshot(); + EXPECT_EQ(restored.lifecycle, "live"); + EXPECT_TRUE(restored.reason.empty()) << restored.reason; + EXPECT_EQ(restored.since, 0); +} diff --git a/src/Storages/System/StorageSystemContentAddressedMounts.cpp b/src/Storages/System/StorageSystemContentAddressedMounts.cpp index ba8b8670cc10..fb4311833d68 100644 --- a/src/Storages/System/StorageSystemContentAddressedMounts.cpp +++ b/src/Storages/System/StorageSystemContentAddressedMounts.cpp @@ -54,9 +54,9 @@ StorageSystemContentAddressedMounts::StorageSystemContentAddressedMounts(const S {"last_success_age_seconds", std::make_shared(std::make_shared()), "Seconds since this disk's GC last led a round (0 if it never led). NULL on rows describing other servers' mounts."}, {"wedged_namespace_count", std::make_shared(std::make_shared()), "Ref-append lanes currently wedged on this disk. NULL on rows describing other servers' mounts."}, {"lifecycle", std::make_shared(), "This server's content-addressed pool lifecycle for the disk (non-gated snapshot, always populated so a not-live disk stays visible): live, not_live, identity_lost, vanished, constructing (never started) or shutdown (torn down)."}, - {"lifecycle_reason", std::make_shared(), "The enum-clean sub-state word for a vanished disk: replaced or forgotten. Empty for every other lifecycle (so lifecycle || '(' || lifecycle_reason || ')' reads e.g. vanished(forgotten))."}, - {"lifecycle_detail", std::make_shared(), "The full typed reason text naming the actual cause when not live: the vanish diagnosis (data root replaced by a foreign pool / decommissioned by SYSTEM CAS FORGET at