From 2835741752d5d51e7f5c0282cc27dcc78bd8ed02 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 05:05:14 +0200 Subject: [PATCH 01/37] Route the CAS reclaim's lease-loss step and fence arm through the runtime Pool::tryRemountOnce calls CasMountRuntime::beginReclaim at step 0 and armIfAdmissible at the end. armMountFence and the new arm share armFence. No behaviour change. Co-Authored-By: Claude Opus 5.5 --- .../ContentAddressed/Pool/CasMountRuntime.cpp | 21 ++++++++++++++++++- .../ContentAddressed/Pool/CasMountRuntime.h | 8 +++++++ .../ContentAddressed/Pool/CasPool.cpp | 15 ++++++------- 3 files changed, 34 insertions(+), 10 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index c29ab3ab657b..78ed6cff46c8 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -259,19 +259,38 @@ void CasMountRuntime::armMountFence(UInt128 server_uuid, uint64_t writer_epoch, { mount_fence.server_uuid = server_uuid; mount_fence.writer_epoch = writer_epoch; + armFence(deadline_boot_ms, /*report_live=*/false); +} + +void CasMountRuntime::armFence(uint64_t deadline_boot_ms, bool report_live) +{ mount_fence.deadline_boot_ms.store(deadline_boot_ms, std::memory_order_release); lease_expired_at_boot_ms.store(std::numeric_limits::max(), std::memory_order_release); /// A fresh lease incarnation is a fresh generation too: a durable-effect caller admitted under the - /// PRIOR incarnation must re-check and abort rather than ride this re-arm through (rev.7 [C2]). + /// PRIOR incarnation must re-check and abort rather than ride this re-arm through. fence_generation.fetch_add(1, std::memory_order_acq_rel); if (arm_mount_fence_interposition_hook_for_test) arm_mount_fence_interposition_hook_for_test(); + if (report_live) + noteRemounted(); /// Open the gate LAST. A caller that observes `lost == false` with acquire semantics must also see /// the fresh generation; publishing the latch first exposes one admission window in which the dead /// generation looks live again. mount_fence.lost.store(false, std::memory_order_release); } +void CasMountRuntime::beginReclaim() +{ + (void)noteLeaseLost(); +} + +bool CasMountRuntime::armIfAdmissible(uint64_t deadline_boot_ms) +{ + armFence(deadline_boot_ms, /*report_live=*/false); + noteRemounted(); + return true; +} + uint64_t CasMountRuntime::minActive() { std::lock_guard lk(builds_mutex); diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index bbe256703be2..72ed714481da 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -185,6 +185,11 @@ class CasMountRuntime void setMountDeadline(uint64_t deadline_boot_ms); /// Arm a new lease incarnation and clear any loss latched for the prior incarnation. void armMountFence(UInt128 server_uuid, uint64_t writer_epoch, uint64_t deadline_boot_ms); + /// Step 0 of a reclaim, before its identity probe: move a `Live` pool to `TransientNotLive`. + void beginReclaim(); + /// End of a reclaim that claimed: publish `deadline_boot_ms`, arm the fence and report `Live`. + /// Returns whether it armed. + bool armIfAdmissible(uint64_t deadline_boot_ms); /// Test-only interposition at the publication boundary between the re-armed generation and the /// live fence. A caller admitted from this hook must be refused: the old generation is already /// dead, while the new generation is not live until `lost` is cleared. @@ -478,6 +483,9 @@ class CasMountRuntime /// caused by the stop cannot be mistaken for one that preceded it. bool renewalCancelled() const; void tripFenceWithoutOperationalLoss(); + /// Publish `deadline_boot_ms` as a fresh lease incarnation and open the fence. With `report_live`, + /// `Live` is published before the fence opens, so a reader that sees the fence armed reads `Live`. + void armFence(uint64_t deadline_boot_ms, bool report_live); std::unique_lock lockTerminalPublication(); /// ---- injected environment (no `Pool` back-reference); initialized first, in this order ---- diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp index e527e9803d82..e6ebc04a5552 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp @@ -1346,7 +1346,7 @@ bool Pool::tryRemountOnce() /// `Recover` (a present `_pool_meta` whose identity matches, in a non-`IdentityLost` state) falls /// through to the existing recovery below; every other verdict resolves here and returns false. step = "lease_loss_transition"; - mount_runtime.noteLeaseLost(); + mount_runtime.beginReclaim(); /// A fully-terminal `Vanished` pool never probes/claims/writes again. step = "terminal_gate"; if (mount_runtime.isVanished()) @@ -1570,14 +1570,11 @@ bool Pool::tryRemountOnce() /// No-throw commit section: publish the fence and lifecycle only after epoch, renewer, recovery /// cancellation, and ref-runtime quiescence are complete. step = "arm_fence"; - mount_runtime.armMountFence( - our_uuid, - writer_epoch, - remount_anchor_boot_ms > std::numeric_limits::max() - ttl_ms - ? std::numeric_limits::max() - : remount_anchor_boot_ms + ttl_ms); - step = "publish_live"; - mount_runtime.noteRemounted(); + const uint64_t deadline_boot_ms = remount_anchor_boot_ms > std::numeric_limits::max() - ttl_ms + ? std::numeric_limits::max() + : remount_anchor_boot_ms + ttl_ms; + if (mount_runtime.armIfAdmissible(deadline_boot_ms)) + step = "publish_live"; succeeded = true; return true; } From 76979a4080acd7143426902c647acd566a4a1538 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 05:15:34 +0200 Subject: [PATCH 02/37] Latch the fence at the start of a CAS reclaim and arm only when admissible beginReclaim records the remount generation the attempt serves and trips the fence. armIfAdmissible acknowledges that generation and arms only when no newer request is pending, no stop is requested and the lifecycle is not terminal; the check and the arm are one step under driver_mutex, and Live is published before the fence opens. A request raised during a reclaim no longer leaves the pool not Live on an armed fence, and a reclaim that finishes after a FORGET intent or a stop arms nothing. Pool::tryRemountOnce still returns true for a reclaim that claimed and did not arm. Co-Authored-By: Claude Opus 5.5 --- .../ContentAddressed/Pool/CasMountRuntime.cpp | 25 +- .../ContentAddressed/Pool/CasMountRuntime.h | 17 +- .../ContentAddressed/Pool/CasPool.cpp | 33 +- src/Disks/tests/gtest_cas_forget.cpp | 36 ++- src/Disks/tests/gtest_cas_pool.cpp | 287 ++++++++++++++++++ 5 files changed, 353 insertions(+), 45 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index 78ed6cff46c8..a74f436230fd 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -281,14 +281,31 @@ void CasMountRuntime::armFence(uint64_t deadline_boot_ms, bool report_live) void CasMountRuntime::beginReclaim() { - (void)noteLeaseLost(); + std::lock_guard lock(driver_mutex); + reclaim_generation = remount_requested_generation; + /// A request can come without a trip; latching here keeps the pool from running not `Live` on an + /// armed fence. Atomics only. + tripMountLost(); } bool CasMountRuntime::armIfAdmissible(uint64_t deadline_boot_ms) { - armFence(deadline_boot_ms, /*report_live=*/false); - noteRemounted(); - return true; + std::lock_guard lock(driver_mutex); + remount_handled_generation = std::max(remount_handled_generation, reclaim_generation); + const bool arm = canArm(deadline_boot_ms); + if (arm) + armFence(deadline_boot_ms, /*report_live=*/true); + else + setMountDeadline(deadline_boot_ms); + driver_cv.notify_all(); + return arm; +} + +bool CasMountRuntime::canArm(uint64_t /*deadline_boot_ms*/) const +{ + return !workers_stop_requested + && !remountTerminal() + && remount_requested_generation <= remount_handled_generation; } uint64_t CasMountRuntime::minActive() diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index 72ed714481da..e8fe587b14bf 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -185,14 +185,17 @@ class CasMountRuntime void setMountDeadline(uint64_t deadline_boot_ms); /// Arm a new lease incarnation and clear any loss latched for the prior incarnation. void armMountFence(UInt128 server_uuid, uint64_t writer_epoch, uint64_t deadline_boot_ms); - /// Step 0 of a reclaim, before its identity probe: move a `Live` pool to `TransientNotLive`. + /// Step 0 of a reclaim. One step under `driver_mutex`: record the requested generation this attempt + /// serves and latch the fence. void beginReclaim(); - /// End of a reclaim that claimed: publish `deadline_boot_ms`, arm the fence and report `Live`. - /// Returns whether it armed. + /// End of a reclaim that claimed. One step under `driver_mutex`: acknowledge the generation + /// `beginReclaim` recorded, publish the deadline, and arm the fence and report `Live` if the arming + /// rule holds. Returns whether it armed. bool armIfAdmissible(uint64_t deadline_boot_ms); /// Test-only interposition at the publication boundary between the re-armed generation and the /// live fence. A caller admitted from this hook must be refused: the old generation is already - /// dead, while the new generation is not live until `lost` is cleared. + /// dead, while the new generation is not live until `lost` is cleared. Through `armIfAdmissible` + /// it runs with `driver_mutex` held. void setArmMountFenceInterpositionHookForTest(std::function hook) { arm_mount_fence_interposition_hook_for_test = std::move(hook); @@ -486,6 +489,10 @@ class CasMountRuntime /// Publish `deadline_boot_ms` as a fresh lease incarnation and open the fence. With `report_live`, /// `Live` is published before the fence opens, so a reader that sees the fence armed reads `Live`. void armFence(uint64_t deadline_boot_ms, bool report_live); + /// The arming rule: no remount request pending, no stop requested, a lifecycle that is not + /// terminal. Requires `driver_mutex`, the mutex that a stop, a request and a terminal publication + /// take, so the check and the arm are one step. + bool canArm(uint64_t deadline_boot_ms) const; std::unique_lock lockTerminalPublication(); /// ---- injected environment (no `Pool` back-reference); initialized first, in this order ---- @@ -541,6 +548,8 @@ class CasMountRuntime std::chrono::milliseconds renewal_period{0}; uint64_t remount_requested_generation = 0; uint64_t remount_handled_generation = 0; + /// The requested generation `beginReclaim` recorded; `armIfAdmissible` acknowledges it. + uint64_t reclaim_generation = 0; ThreadFromGlobalPool renewal_worker; ThreadFromGlobalPool remount_worker; /// Counted entries into `scheduleRemount`; retained as a test-only observability seam. diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp index e6ebc04a5552..5cd49334efb2 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp @@ -1189,12 +1189,10 @@ void Pool::forgetDisk(const std::function & stop_and_join_gc, const Stri /// (5a) Stop and join both persistent mount-runtime workers outside `remount_mutex`. mount_runtime.stopBackgroundWorkers(); - /// A remount attempt already IN FLIGHT when step 1 published the intent completes its current step - /// before the loop bails (the "one step + one backend timeout" bound of §5), and a successful reclaim in - /// that window re-arms the local fence (`lost = false`). Now that the remount worker is JOINED and can - /// never run again, re-latch the fence so the terminal `mayMutate() == false` holds regardless of any - /// such raced reclaim. Idempotent; the durable mount lease the reclaim wrote is retired by the - /// `finishTeardown` below (it operates on whatever renewer is current — the reclaimed one). + /// A reclaim in flight when step 1 published the intent finishes its current step, and the arming + /// rule refuses to arm after the intent. This second trip is a backstop: nothing can arm the fence + /// once the workers are joined. Idempotent; the durable mount lease such a reclaim wrote is retired + /// by the `finishTeardown` below (it operates on whatever renewer is current — the reclaimed one). mount_runtime.tripMountLost(); /// (5b) Drain the ref lanes (bounded by one attempt's budget + safety margin) to learn whether a clean @@ -1339,12 +1337,11 @@ bool Pool::tryRemountOnce() const uint64_t poll_interval_ms = std::max( 1, static_cast(config.mount_renew_period.count()) / 2); - /// ==== Step 0 (rev.7 §2): pool lifecycle identity gate — BEFORE any claim/allocate/mount write ==== - /// A remount attempt means the lease is presumed lost, so first ensure we are at least transient - /// (production reaches here already transient via `tripMountLost`; a direct/forced call may still be - /// `Live`). Then authoritatively probe the pool sentinels and dispatch per the §2 verdict table. Only - /// `Recover` (a present `_pool_meta` whose identity matches, in a non-`IdentityLost` state) falls - /// through to the existing recovery below; every other verdict resolves here and returns false. + /// ==== Step 0: pool lifecycle identity gate — BEFORE any claim/allocate/mount write ==== + /// A reclaim starts with the fence latched and records the remount generation it serves. Then it + /// authoritatively probes the pool sentinels. Only `Recover` (a present `_pool_meta` whose identity + /// matches, in a non-`IdentityLost` state) falls through to the recovery below; every other verdict + /// resolves here and returns false. step = "lease_loss_transition"; mount_runtime.beginReclaim(); /// A fully-terminal `Vanished` pool never probes/claims/writes again. @@ -1399,10 +1396,8 @@ bool Pool::tryRemountOnce() /// BEFORE the intent was published, which that gate therefore cannot catch — could otherwise /// settle `Vanished(replaced)` mid-FORGET, stranding FORGET's own /// `enterVanished(VanishedForgotten)` (first terminal STATE transition wins) and mislabeling - /// the operator-visible reason. The bail lives HERE, at the terminal settle, so the - /// Recover/`armMountFence` reclaim path is untouched — its mid-FORGET fence re-arm is the - /// SEPARATE hazard `forgetDisk`'s post-join re-trip (trip#2) guards. Post-excision this is the - /// ONLY surviving mid-FORGET natural-terminal race (the old erasure-proof promotion is gone). + /// the operator-visible reason. Post-excision this is the ONLY surviving mid-FORGET + /// natural-terminal race (the old erasure-proof promotion is gone). if (mount_runtime.vanishedIntentPublished()) return false; mount_runtime.enterVanished(PoolLifecycle::VanishedReplaced, gate.reason); @@ -1567,8 +1562,10 @@ bool Pool::tryRemountOnce() remount_anchor_boot_ms = mount_runtime.renewRenewerForRemountOnce(); } - /// No-throw commit section: publish the fence and lifecycle only after epoch, renewer, recovery - /// cancellation, and ref-runtime quiescence are complete. + /// No-throw commit section: arm only after epoch, renewer, recovery cancellation, and ref-runtime + /// quiescence are complete, and only under the arming rule. A reclaim that claimed and did not + /// arm still returns true: its request is acknowledged, and a false would make the loop back off + /// and claim yet another epoch. step = "arm_fence"; const uint64_t deadline_boot_ms = remount_anchor_boot_ms > std::numeric_limits::max() - ttl_ms ? std::numeric_limits::max() diff --git a/src/Disks/tests/gtest_cas_forget.cpp b/src/Disks/tests/gtest_cas_forget.cpp index d1b49b4c41a7..db7ba970d24b 100644 --- a/src/Disks/tests/gtest_cas_forget.cpp +++ b/src/Disks/tests/gtest_cas_forget.cpp @@ -396,34 +396,32 @@ TEST(CASForget, ForgetRacingActiveRemountThreadCompletesBounded) EXPECT_FALSE(store->mayMutate()) << "the fence must stay latched even if a raced reclaim re-armed it"; } -/// (b2) FENCE RE-LATCH REGRESSION GUARD (the fix's raison d'être): a self-remount that reaches -/// `armMountFence` re-arms the local fence (`lost=false`) after FORGET has already tripped it. FORGET's -/// SECOND `tripMountLost` — placed AFTER the remount worker is joined — must override it. -/// -/// (b1)'s fault keeps every attempt at `StayTransient`, so it can NOT catch removal of that second trip. To -/// make EXACTLY ONE reclaim reach `armMountFence` inside FORGET's window, deterministically and without a -/// sleep, we drive a REAL `tryRemountOnce` from FORGET's own GC-stop step (invoked at spec §5 step 3/4, -/// strictly AFTER the fence trip): the mount is fenced-out so the reclaim succeeds fast and re-arms the -/// fence, and `tryRemountOnce`'s step-0 gate checks `isVanished()` — still false in this window — so it does -/// NOT bail. The re-arm therefore lands after trip#1 and before trip#2, exactly the interval trip#2 guards. -/// Verified to go RED when trip#2 is removed (see task-10-report.md — test_task10b_reddemo.log). -TEST(CASForget, ForgetReLatchesFenceAfterAReclaimReachesArmMountFence) +/// (b2) A reclaim that finishes after FORGET published its intent arms nothing. The reclaim runs from +/// FORGET's own GC-stop step, after the intent and the first trip; the mount is fenced out, so it claims a +/// fresh incarnation at once. FORGET's second trip stays as a backstop. +TEST(CASForget, AReclaimDuringForgetArmsNothing) { auto backend = std::make_shared(); auto store = DB::Cas::tests::openPoolForTest(backend); - /// Make the current mount claimable so a self-remount SUCCEEDS fast and reaches `armMountFence`. + /// Make the current mount claimable so the reclaim claims at once. fenceOutMount(*backend, store->layout().mountKey(kSrid)); bool reclaimed = false; - store->forgetDisk([&] { reclaimed = store->tryRemountOnce(); }, kForgetReason); + bool may_mutate_after_reclaim = true; + store->forgetDisk( + [&] + { + reclaimed = store->tryRemountOnce(); + may_mutate_after_reclaim = store->mayMutate(); + }, + kForgetReason); - /// Guard against a vacuous pass: if the injected reclaim did not actually succeed (reach - /// `armMountFence`), there is no re-arm for trip#2 to override and the test proves nothing. - ASSERT_TRUE(reclaimed) << "the injected reclaim must reach armMountFence, else this guard is vacuous"; + /// Guard against a vacuous pass: a reclaim that never claimed reaches no arm at all. + ASSERT_TRUE(reclaimed) << "the injected reclaim must claim, else this test proves nothing"; + EXPECT_FALSE(may_mutate_after_reclaim) << "a reclaim after the FORGET intent must not arm the fence"; EXPECT_EQ(store->lifecycle(), PoolLifecycle::VanishedForgotten); - EXPECT_FALSE(store->mayMutate()) - << "FORGET's post-join fence re-latch (trip#2) must override the fence the reclaim re-armed"; + EXPECT_FALSE(store->mayMutate()); } /// (b3) PROMOTION-GUARD REGRESSION (spec §9 rev.8 item 7): with the erasure-proof excised, the natural diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp index 76b594e7bfe9..0bf56f36b193 100644 --- a/src/Disks/tests/gtest_cas_pool.cpp +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -4906,6 +4906,293 @@ TEST(CASMountRuntime, AStaleSuccessIsFollowedAtOnceByTheNextRenewal) runtime.finishTeardown(false); } +/// An interference report raised while a reclaim runs: that reclaim arms nothing, the next one starts +/// with the fence latched, and no write is admitted between the two. +TEST(CASMountRuntime, NothingArmsWhileARemountRequestIsPending) +{ + auto backend = std::make_shared(); + const Layout layout("runtime-nothing-arms-while-pending"); + uint64_t wall_ms = 1000; + const uint64_t boot_ms = 100; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, + MountClaimResult::Claimed); + std::atomic calls{0}; + bool first_armed = true; + bool may_mutate_after_first = true; + bool may_mutate_at_second_entry = true; + bool may_mutate_after_second_latch = true; + PoolLifecycle lifecycle_after_second_latch = PoolLifecycle::Live; + bool second_armed = false; + bool may_mutate_after_second = false; + DB::Cas::tests::ManualBarrier second_done; + CasMountRuntime * runtime_ptr = nullptr; + CasEventSink sink; + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }}, + "test", sink, runtimeRenewBudget(), [&] + { + CasMountRuntime & reclaiming = *runtime_ptr; + if (++calls == 1) + { + reclaiming.beginReclaim(); + /// An interference report while this reclaim runs. + reclaiming.tripMountLost(); + reclaiming.scheduleRemount(); + first_armed = reclaiming.armIfAdmissible(boot_ms + 1000); + may_mutate_after_first = reclaiming.mayMutate(); + return true; + } + may_mutate_at_second_entry = reclaiming.mayMutate(); + reclaiming.beginReclaim(); + may_mutate_after_second_latch = reclaiming.mayMutate(); + lifecycle_after_second_latch = reclaiming.lifecycle(); + second_armed = reclaiming.armIfAdmissible(boot_ms + 1000); + may_mutate_after_second = reclaiming.mayMutate(); + second_done.arriveAndWait(); + return true; + }); + /// Declared after the runtime so it runs first: a failed expectation must not leave the reclaim + /// parked on the barrier while the runtime's destructor joins the thread. + SCOPE_EXIT({ second_done.release(); }); + CasMountRuntime & runtime = *runtime_holder; + runtime_ptr = &runtime; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.startBackgroundWorkers(std::chrono::hours(1)); + runtime.tripMountLost(); + runtime.scheduleRemount(); + + second_done.waitUntilArrived(); + EXPECT_FALSE(first_armed) << "a reclaim must not arm while a newer remount request is pending"; + EXPECT_FALSE(may_mutate_after_first); + EXPECT_FALSE(may_mutate_at_second_entry) << "no write may be admitted between the two reclaims"; + EXPECT_FALSE(may_mutate_after_second_latch) << "a reclaim starts with the fence latched"; + EXPECT_EQ(lifecycle_after_second_latch, PoolLifecycle::TransientNotLive); + EXPECT_TRUE(second_armed) << "the reclaim that served the last request arms"; + EXPECT_TRUE(may_mutate_after_second); + EXPECT_EQ(calls.load(), 2u); + second_done.release(); + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); +} + +/// One interference report produces one remount generation, one epoch change and one lease-loss +/// count, also when the reclaim fails twice before it succeeds. Two failures cost one and two seconds +/// of the loop's real backoff. +TEST(CASMountRuntime, AReclaimAcknowledgesOnlyTheGenerationItServed) +{ + auto backend = std::make_shared(); + const Layout layout("runtime-reclaim-acknowledges"); + uint64_t wall_ms = 1000; + const uint64_t boot_ms = 100; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, + MountClaimResult::Claimed); + std::atomic calls{0}; + std::atomic fresh_epochs{0}; + DB::Cas::tests::ManualBarrier reclaimed; + CasMountRuntime * runtime_ptr = nullptr; + CasEventSink sink; + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }}, + "test", sink, runtimeRenewBudget(), [&] + { + CasMountRuntime & reclaiming = *runtime_ptr; + reclaiming.beginReclaim(); + if (++calls < 3) + return false; + fenceOutMount(*backend, layout.mountKey("test")); + const MountClaimResult fresh + = claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 2, wall_ms, 1000); + EXPECT_EQ(fresh.kind, MountClaimResult::Claimed); + if (fresh.kind != MountClaimResult::Claimed) + return false; + ++fresh_epochs; + reclaiming.installRenewer(uuid, 2, [&] { return wall_ms; }); + const uint64_t fresh_anchor = reclaiming.startRenewer(); + reclaiming.setLiveWriterEpoch(2); + EXPECT_TRUE(reclaiming.armIfAdmissible(fresh_anchor + 1000)); + reclaimed.arriveAndWait(); + return true; + }); + SCOPE_EXIT({ reclaimed.release(); }); + CasMountRuntime & runtime = *runtime_holder; + runtime_ptr = &runtime; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + runtime.armMountFence(uuid, 1, anchor + 1000); + const uint64_t lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); + runtime.startBackgroundWorkers(std::chrono::hours(1)); + /// One interference report. + runtime.tripMountLost(); + runtime.scheduleRemount(); + + reclaimed.waitUntilArrived(); + EXPECT_EQ(calls.load(), 3u); + EXPECT_EQ(fresh_epochs.load(), 1u); + EXPECT_EQ(runtime.remountRequestedGenerationForTest(), 1u) << "a failed reclaim raises no generation"; + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(), lost_before + 1) + << "the latch of each attempt counts no further loss"; + EXPECT_TRUE(runtime.mayMutate()); + reclaimed.release(); + runtime.stopBackgroundWorkers(); + EXPECT_EQ(calls.load(), 3u) << "the served generation is not reclaimed again"; + runtime.finishTeardown(false); +} + +/// The first two steps of `Pool::forgetDisk` land while a reclaim runs: the reclaim that finishes +/// after them arms nothing, so the fence is latched before FORGET's second trip, and the thread exits. +TEST(CASMountRuntime, AReclaimFinishedAfterTheForgetIntentArmsNothing) +{ + auto backend = std::make_shared(); + const Layout layout("runtime-reclaim-after-intent"); + uint64_t wall_ms = 1000; + const uint64_t boot_ms = 100; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, + MountClaimResult::Claimed); + WorkerExitLatch exits; + uint64_t threads = 0; + RuntimeWorkerFactory factory = [&](std::function worker_body) + { + ++threads; + return ThreadFromGlobalPool([&, body = std::move(worker_body)] + { + body(); + exits.recordExit(); + }); + }; + DB::Cas::tests::ManualBarrier latched; + bool armed = true; + bool may_mutate_after_reclaim = true; + CasMountRuntime * runtime_ptr = nullptr; + CasEventSink sink; + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }, .worker_factory = factory}, + "test", sink, runtimeRenewBudget(), [&] + { + CasMountRuntime & reclaiming = *runtime_ptr; + reclaiming.beginReclaim(); + latched.arriveAndWait(); + armed = reclaiming.armIfAdmissible(boot_ms + 1000); + may_mutate_after_reclaim = reclaiming.mayMutate(); + return true; + }); + SCOPE_EXIT({ latched.release(); }); + CasMountRuntime & runtime = *runtime_holder; + runtime_ptr = &runtime; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.startBackgroundWorkers(std::chrono::hours(1)); + runtime.tripMountLost(); + runtime.scheduleRemount(); + + latched.waitUntilArrived(); + runtime.publishVanishedIntent(); + runtime.tripMountLost(); + latched.release(); + + const bool exited = exits.waitForAtLeast(threads); + EXPECT_TRUE(exited) << "the published intent must end the loop"; + EXPECT_FALSE(armed) << "a reclaim that finished after the intent must not arm the fence"; + EXPECT_FALSE(may_mutate_after_reclaim); + EXPECT_FALSE(runtime.mayMutate()) << "the fence is latched before FORGET's second trip"; + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); +} + +/// A stop requested during a reclaim: the join returns once the attempt returns, the reclaim arms +/// nothing, the stop counts no lease loss, and no farewell is written for the slot this runtime did +/// not claim back. +TEST(CASMountRuntime, StopDuringAReclaimJoinsTheThread) +{ + auto backend = std::make_shared(); + const Layout layout("runtime-stop-during-reclaim"); + uint64_t wall_ms = 1000; + const uint64_t boot_ms = 100; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, + MountClaimResult::Claimed); + WorkerExitLatch exits; + uint64_t threads = 0; + RuntimeWorkerFactory factory = [&](std::function worker_body) + { + ++threads; + return ThreadFromGlobalPool([&, body = std::move(worker_body)] + { + body(); + exits.recordExit(); + }); + }; + DB::Cas::tests::ManualBarrier reclaim_entered; + std::atomic stop_requested_by_test{false}; + bool stop_seen_by_reclaim = false; + bool armed = true; + std::atomic reclaim_returned{false}; + uint64_t lost_at_latch = 0; + CasMountRuntime * runtime_ptr = nullptr; + CasEventSink sink; + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }, .worker_factory = factory}, + "test", sink, runtimeRenewBudget(), [&] + { + CasMountRuntime & reclaiming = *runtime_ptr; + reclaiming.beginReclaim(); + lost_at_latch = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); + reclaim_entered.arriveAndWait(); + /// Returns on the stop; the timeout only bounds a failing run. + reclaiming.sleepInterruptibly(20'000); + stop_seen_by_reclaim = stop_requested_by_test.load(); + armed = reclaiming.armIfAdmissible(boot_ms + 1000); + reclaim_returned = true; + return true; + }); + SCOPE_EXIT({ reclaim_entered.release(); }); + CasMountRuntime & runtime = *runtime_holder; + runtime_ptr = &runtime; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + runtime.armMountFence(uuid, 1, anchor + 1000); + const String key = layout.mountKey("test"); + /// A definitive answer ends the first renewal and requests the reclaim; nobody claims the slot back. + fenceOutMount(*backend, key); + const uint64_t lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); + runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + + reclaim_entered.waitUntilArrived(); + stop_requested_by_test = true; + auto stop = std::async(std::launch::async, [&] { runtime.stopBackgroundWorkers(); }); + reclaim_entered.release(); + stop.get(); + + EXPECT_TRUE(reclaim_returned.load()) << "the join waits for the reclaim attempt to return"; + EXPECT_EQ(exits.count(), threads); + EXPECT_TRUE(stop_seen_by_reclaim) << "the reclaim must finish after the stop was requested"; + EXPECT_FALSE(armed) << "a reclaim that finishes after a stop must not arm the fence"; + EXPECT_FALSE(runtime.mayMutate()); + EXPECT_EQ(lost_at_latch, lost_before + 1) << "the fenced-out renewal counts the one loss"; + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(), lost_at_latch) + << "the stop counts no lease loss"; + const auto slot_before_teardown = readObj(*backend, key); + ASSERT_TRUE(slot_before_teardown.has_value()); + runtime.finishTeardown(true); + const auto slot_after_teardown = readObj(*backend, key); + ASSERT_TRUE(slot_after_teardown.has_value()); + EXPECT_EQ(slot_after_teardown->bytes, slot_before_teardown->bytes) + << "no farewell for a slot this runtime did not claim back"; +} + TEST(CASPool, RenewWatermarkOnceRefreshesFenceAndDepositsOneFailure) { auto backend = std::make_shared(); From 07ecdee50d4b05d84c4adf83bca8e30b0ebf983a Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 05:31:34 +0200 Subject: [PATCH 03/37] Run the CAS mount lease and its reclaim on one thread The lease loop serves a pending remount request first, then renews at the cadence. A renewal ends early on a pending request, and its result is consumed under driver_mutex before the loop reads the generations again, so a commit can no longer overwrite the deadline a later reclaim armed. The remount worker, the nine RenewalDriverState values, DriverLease, admitRenewerCall and ThreadName::CAS_REMOUNT are removed; the startup, remount and direct renewals stay as plain calls. Tests of the parking protocol are deleted; tests of guarantees that remain are adapted to one thread. Co-Authored-By: Claude Opus 5.5 --- src/Common/setThreadName.h | 1 - .../ContentAddressed/Pool/CasMountRuntime.cpp | 608 +++++--------- .../ContentAddressed/Pool/CasMountRuntime.h | 189 ++--- .../ContentAddressed/Pool/CasPool.cpp | 36 +- .../ContentAddressed/Pool/CasPool.h | 23 +- .../ContentAddressed/Pool/CasServerRoot.cpp | 6 +- .../ContentAddressed/Pool/CasServerRoot.h | 6 +- src/Disks/tests/gtest_cas_forget.cpp | 10 +- src/Disks/tests/gtest_cas_pool.cpp | 762 +++++++++--------- 9 files changed, 686 insertions(+), 955 deletions(-) diff --git a/src/Common/setThreadName.h b/src/Common/setThreadName.h index 8a0222dc9639..f92276d9d5b2 100644 --- a/src/Common/setThreadName.h +++ b/src/Common/setThreadName.h @@ -38,7 +38,6 @@ namespace DB M(CAS_GC_SCHEDULER, "CasGcSched") \ M(CAS_LEASE_RENEWER, "CasLeaseRenewer") \ M(CAS_REF_SNAPSHOT_PUBLISH, "CasRefSnapPub") \ - M(CAS_REMOUNT, "CasRemount") \ M(CGROUP_MEMORY_OBSERVER, "CgrpMemUsgObsr") \ M(CLICKHOUSE_WATCH, "ClickHouseWatch") \ M(CLUSTER_DISCOVERY, "ClusterDiscover") \ diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index a74f436230fd..47e6e990ff42 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -219,7 +219,7 @@ void CasMountRuntime::noteRenewRequest(const MountRenewRequestEvent & event) noe } } -std::optional CasMountRuntime::publishRenewedDeadline(uint64_t deadline_boot_ms) +std::optional CasMountRuntime::publishRenewedDeadline(uint64_t deadline_boot_ms) { const uint64_t now = bootMsNow(); const std::optional expired_at = leaseExpiredAt(now); @@ -241,13 +241,8 @@ std::optional CasMountRuntime::publishRenewedDeadline(uint64_t deadlin } if (!expired_at) return std::nullopt; - const uint64_t expired_ms = now - *expired_at; ProfileEvents::incrementNoTrace(ProfileEvents::CASMountLeaseExpired); - LOG_WARNING(getLogger("CasPool"), - "CAS mount lease of '{}' was expired for {} ms; a renewal restored it and writes resume. " - "Last failed renewal request: {}", - server_root_id, expired_ms, ended_failure); - return expired_ms; + return RestoredLease{.expired_ms = now - *expired_at, .last_failure = std::move(ended_failure)}; } void CasMountRuntime::setMountDeadline(uint64_t deadline_boot_ms) @@ -322,11 +317,7 @@ uint64_t CasMountRuntime::peekNextBuildSeq() void CasMountRuntime::renewWatermarkOnce() { - auto call = admitRenewerCall(RenewalDriverState::Dormant, RenewalDriverState::DirectCall); - (void)renewRenewerOnce( - std::move(call), - RenewalDriverState::DirectCall, - /*propagate_failure=*/true); + (void)renewRenewerOnce(RenewCaller::Direct); } uint64_t CasMountRuntime::allocateBuildSeq() @@ -390,107 +381,12 @@ void CasMountRuntime::setLiveWriterEpoch(uint64_t v) live_writer_epoch.store(v, std::memory_order_release); } -CasMountRuntime::DriverLease::DriverLease(CasMountRuntime & runtime_, RenewalDriverState active_) - : runtime(runtime_) - , active(active_) -{ -} - -bool CasMountRuntime::renewalWorkerMayRenew() const -{ - return renewal_driver_state == RenewalDriverState::WorkerIdle - && mount_renewer - && mount_renewer->state() == MountLeaseRenewerState::Active; -} - -CasMountRuntime::DriverLease::~DriverLease() -{ - if (finished) - return; - std::lock_guard lock(runtime.driver_mutex); - if (runtime.workers_stop_requested) - runtime.renewal_driver_state = RenewalDriverState::Stopping; - else if (active == RenewalDriverState::WorkerCall - && runtime.renewal_driver_state == RenewalDriverState::ParkRequested) - runtime.renewal_driver_state = RenewalDriverState::Parked; - else if (active == RenewalDriverState::WorkerCall) - runtime.renewal_driver_state = RenewalDriverState::WorkerIdle; - else if (active == RenewalDriverState::RemountCall) - runtime.renewal_driver_state = RenewalDriverState::Parked; - else - runtime.renewal_driver_state = RenewalDriverState::Dormant; - runtime.driver_cv.notify_all(); -} - -RenewalDriverState CasMountRuntime::DriverLease::finish( - RenewalDriverState ordinary_destination, - const MountRenewResult * result) -{ - std::lock_guard lock(runtime.driver_mutex); - if (runtime.workers_stop_requested) - runtime.renewal_driver_state = RenewalDriverState::Stopping; - else if (active == RenewalDriverState::WorkerCall - && runtime.renewal_driver_state == RenewalDriverState::ParkRequested) - runtime.renewal_driver_state = RenewalDriverState::Parked; - else - runtime.renewal_driver_state = ordinary_destination; - if (result && result->outcome == MountRenewOutcome::Terminal) - { - if (runtime.renewal_driver_state == RenewalDriverState::WorkerIdle - || active == RenewalDriverState::DirectCall) - { - runtime.tripMountLost(); - runtime.schedule_remount_calls_for_test.fetch_add(1, std::memory_order_relaxed); - if (runtime.config.background_watermark - && !runtime.workers_stop_requested - && !runtime.remountTerminal()) - { - ++runtime.remount_requested_generation; - if (active == RenewalDriverState::WorkerCall) - runtime.renewal_driver_state = RenewalDriverState::Parked; - } - } - else if (runtime.renewal_driver_state == RenewalDriverState::Stopping - || active == RenewalDriverState::StartupCall) - { - runtime.tripFenceWithoutOperationalLoss(); - } - } - finished = true; - runtime.driver_cv.notify_all(); - return runtime.renewal_driver_state; -} - -CasMountRuntime::AdmittedRenewerCall CasMountRuntime::admitRenewerCall( - RenewalDriverState required, - RenewalDriverState active) -{ - std::lock_guard lock(driver_mutex); - if (active == RenewalDriverState::DirectCall && config.background_watermark) - throw Exception( - ErrorCodes::LOGICAL_ERROR, - "CAS mount runtime: direct renewal is disabled when background ownership is configured"); - if (renewal_driver_state != required) - throw Exception( - ErrorCodes::LOGICAL_ERROR, - "CAS mount runtime: renewal driver is not admitted from the required state"); - if (!mount_renewer) - throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: renewal without a renewer"); - if (mount_renewer->state() != MountLeaseRenewerState::Active) - throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: renewal requires an Active renewer"); - MountLeaseRenewer * renewer = mount_renewer.get(); - auto lease = std::make_unique(*this, active); - renewal_driver_state = active; - driver_cv.notify_all(); - return AdmittedRenewerCall{std::move(lease), renewer}; -} - void CasMountRuntime::installRenewer( UInt128 our_uuid, uint64_t writer_epoch, const std::function & now_ms) { - auto replacement = std::make_unique( + std::unique_ptr replaced = std::make_unique( mount_requests, farewell_requests, lease_requests, layout, server_root_id, our_uuid, writer_epoch, config.mount_lease_ttl_ms, now_ms, [this] { return minActive(); }, @@ -498,81 +394,56 @@ void CasMountRuntime::installRenewer( std::chrono::milliseconds(cas_request_budget.lease_safety_margin_ms), [this] { return bootMsNow(); }); + /// The previous renewer is destroyed after the unlock. std::lock_guard lock(driver_mutex); - if (renewal_driver_state != RenewalDriverState::Dormant - && renewal_driver_state != RenewalDriverState::Parked) - throw Exception( - ErrorCodes::LOGICAL_ERROR, - "CAS mount runtime: renewer replacement requires Dormant or Parked renewal ownership"); - mount_renewer = std::move(replacement); + std::swap(mount_renewer, replaced); } uint64_t CasMountRuntime::startRenewer() { - RenewalDriverState active; - RenewalDriverState destination; - MountLeaseRenewer * renewer; + MountLeaseRenewer * renewer = nullptr; { std::lock_guard lock(driver_mutex); if (!mount_renewer) throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: startRenewer without a renewer"); if (mount_renewer->state() != MountLeaseRenewerState::New) throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: startRenewer requires a New renewer"); - if (renewal_driver_state == RenewalDriverState::Dormant) - { - active = RenewalDriverState::StartupCall; - destination = RenewalDriverState::Dormant; - } - else if (renewal_driver_state == RenewalDriverState::Parked) - { - active = RenewalDriverState::RemountCall; - destination = RenewalDriverState::Parked; - } - else - { - throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: startRenewer is not admitted in the current state"); - } renewer = mount_renewer.get(); - renewal_driver_state = active; - driver_cv.notify_all(); } - - DriverLease lease(*this, active); - const uint64_t anchor = renewer->start([this] { return !renewalCancelled(); }); - (void)lease.finish(destination); - return anchor; + return renewer->start([this] { return !renewalCancelled(); }); } -MountRenewOperationEnvironment CasMountRuntime::renewalEnvironment(bool worker_call) +MountRenewOperationEnvironment CasMountRuntime::renewalEnvironment(RenewCaller caller) { + const bool loop = caller == RenewCaller::Loop; return MountRenewOperationEnvironment{ .boot_ms = [this] { return bootMsNow(); }, - .live = [this, worker_call] + .live = [this, caller] { - return renewalLive(worker_call) && (!config.renewal_live_for_test || config.renewal_live_for_test()); + return renewalLive(caller) && (!config.renewal_live_for_test || config.renewal_live_for_test()); }, .cancelled = [this] { return renewalCancelled(); }, - /// Only the worker keeps renewing past the lease; startup, remount and direct renewals stay + /// Only the loop keeps renewing past the lease; startup, remount and direct renewals stay /// bounded by it. - .policy = worker_call ? MountRenewPolicy::UntilDefinitive : MountRenewPolicy::LeaseBound, - /// The worker counts its requests as they are sent, so an outage shows while it lasts. - .on_request = worker_call + .policy = loop ? MountRenewPolicy::UntilDefinitive : MountRenewPolicy::LeaseBound, + /// The loop counts its requests as they are sent, so an outage shows while it lasts. + .on_request = loop ? std::function( [this](const MountRenewRequestEvent & event) { noteRenewRequest(event); }) : nullptr, }; } -bool CasMountRuntime::renewalLive(bool worker_call) const +bool CasMountRuntime::renewalLive(RenewCaller caller) const { std::lock_guard lock(driver_mutex); if (workers_stop_requested) return false; - return !(worker_call - && (renewal_driver_state == RenewalDriverState::ParkRequested - || renewal_driver_state == RenewalDriverState::Parked - || lifecycle() != PoolLifecycle::Live - || mount_fence.lost.load(std::memory_order_acquire))); + if (caller != RenewCaller::Loop) + return true; + return remount_requested_generation <= remount_handled_generation + && lifecycle() == PoolLifecycle::Live + && !mount_fence.lost.load(std::memory_order_acquire); } bool CasMountRuntime::renewalCancelled() const @@ -587,17 +458,11 @@ void CasMountRuntime::sleepInterruptibly(uint64_t ms) driver_cv.wait_for(lock, std::chrono::milliseconds(ms), [this] { return workers_stop_requested; }); } -void CasMountRuntime::consumeRenewResult( - const MountRenewResult & result, - RenewalDriverState active_state, - RenewalDriverState returned_state, - bool propagate_failure) +void CasMountRuntime::consumeRenewResult(const MountRenewResult & result, RenewCaller caller) { - /// Driver ownership has already been restored by `DriverLease::finish`; this is the single logical - /// consumption boundary and it runs without `driver_mutex` or renewer access. - /// The worker's renewal counts its requests as they are sent (`noteRenewRequest`). Every other - /// renewal counts them here, from its result. - if (active_state != RenewalDriverState::WorkerCall && result.attempts_sent > 0) + /// The loop counts its requests as they are sent (`noteRenewRequest`). Every other renewal counts + /// them here, from its result. + if (caller != RenewCaller::Loop && result.attempts_sent > 0) { ProfileEvents::incrementNoTrace(ProfileEvents::CASMountRenewalAttempts, result.attempts_sent); ProfileEvents::incrementNoTrace(ProfileEvents::CASMountRenewalRetries, result.attempts_sent - 1); @@ -611,13 +476,63 @@ void CasMountRuntime::consumeRenewResult( && result.deadline_source == GaveUp::Source::Lease) ProfileEvents::incrementNoTrace(ProfileEvents::CASMountRenewalDeadlineExceeded); + /// One step under `driver_mutex`. The loop reads the remount generations only after it, so no + /// reclaim can arm between a commit and the publication of its deadline. + std::optional restored; + { + std::lock_guard lock(driver_mutex); + if (result.outcome == MountRenewOutcome::Committed) + { + const uint64_t ttl_ms = static_cast(config.mount_lease_ttl_ms.count()); + restored = publishRenewedDeadline( + result.attempt_start_boot_ms > std::numeric_limits::max() - ttl_ms + ? std::numeric_limits::max() + : result.attempt_start_boot_ms + ttl_ms); + } + else if (result.outcome == MountRenewOutcome::Terminal) + { + switch (caller) + { + case RenewCaller::Loop: + if (workers_stop_requested) + { + tripFenceWithoutOperationalLoss(); + break; + } + tripMountLost(); + schedule_remount_calls_for_test.fetch_add(1, std::memory_order_relaxed); + /// A pending request already covers this loss. + if (!remountTerminal() && remount_requested_generation == remount_handled_generation) + ++remount_requested_generation; + break; + case RenewCaller::Direct: + tripMountLost(); + schedule_remount_calls_for_test.fetch_add(1, std::memory_order_relaxed); + break; + case RenewCaller::Startup: + tripFenceWithoutOperationalLoss(); + break; + case RenewCaller::Remount: + /// The reclaim latched the fence at its start and reports this failure itself. + if (workers_stop_requested) + tripFenceWithoutOperationalLoss(); + break; + } + } + driver_cv.notify_all(); + } + + if (restored) + LOG_WARNING(getLogger("CasPool"), + "CAS mount lease of '{}' was expired for {} ms; a renewal restored it and writes resume. " + "Last failed renewal request: {}", + server_root_id, restored->expired_ms, restored->last_failure); + if (result.outcome == MountRenewOutcome::Committed) { - const uint64_t ttl_ms = static_cast(config.mount_lease_ttl_ms.count()); - const std::optional expired_ms = publishRenewedDeadline( - result.attempt_start_boot_ms > std::numeric_limits::max() - ttl_ms - ? std::numeric_limits::max() - : result.attempt_start_boot_ms + ttl_ms); + std::optional expired_ms; + if (restored) + expired_ms = restored->expired_ms; reportMountRenewCompletion(result, expired_ms); return; } @@ -644,64 +559,52 @@ void CasMountRuntime::consumeRenewResult( reportMountRenewCompletion(result, std::nullopt); - (void)returned_state; - - if (propagate_failure) + if (caller != RenewCaller::Loop) std::rethrow_exception(result.failure); } -uint64_t CasMountRuntime::renewRenewerOnce( - AdmittedRenewerCall call, - RenewalDriverState active, - bool propagate_failure) +MountRenewResult CasMountRuntime::renewRenewerOnce(RenewCaller caller) { - const bool worker_call = active == RenewalDriverState::WorkerCall; - /// Configuration is pointer/POD-only. A parked redo retains its completed observation for the + MountLeaseRenewer * renewer = nullptr; + { + std::lock_guard lock(driver_mutex); + if (caller == RenewCaller::Direct && config.background_watermark) + throw Exception( + ErrorCodes::LOGICAL_ERROR, + "CAS mount runtime: direct renewal is disabled when background ownership is configured"); + if (!mount_renewer) + throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: renewal without a renewer"); + if (mount_renewer->state() != MountLeaseRenewerState::Active) + throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: renewal requires an Active renewer"); + renewer = mount_renewer.get(); + } + /// Configuration is pointer/POD-only. A remount re-anchor retains its completed observation for the /// whole-chain finalizer to deliver after `remount_mutex` is released. - configureMountRenewObservability( - &server_root_id, &event_sink, active == RenewalDriverState::RemountCall); - /// The remount redo re-anchors the lease BEFORE `armMountFence`, with the fence still latched lost, - /// so it renews on the renewer's open plane: admitted under the mount fence it could only ever give - /// up, and every remount would fail at this step. `RemountCall` is reached from - /// `renewRenewerForRemountOnce` alone. - const MountRenewResult result = active == RenewalDriverState::RemountCall - ? call.renewer->renewForRemount(renewalEnvironment(worker_call)) - : call.renewer->renew(renewalEnvironment(worker_call)); - const RenewalDriverState destination = active == RenewalDriverState::WorkerCall - ? RenewalDriverState::WorkerIdle - : (active == RenewalDriverState::RemountCall ? RenewalDriverState::Parked : RenewalDriverState::Dormant); - const RenewalDriverState returned_state = call.lease->finish(destination, &result); - if (result.outcome == MountRenewOutcome::Terminal && config.renewal_terminal_deposited_hook_for_test) - config.renewal_terminal_deposited_hook_for_test(); - consumeRenewResult(result, active, returned_state, propagate_failure); - return result.attempt_start_boot_ms; + configureMountRenewObservability(&server_root_id, &event_sink, caller == RenewCaller::Remount); + /// The remount re-anchor runs before `armIfAdmissible`, with the fence still latched, so it renews on + /// the renewer's open plane: admitted under the mount fence it could only ever give up. + const MountRenewResult result = caller == RenewCaller::Remount + ? renewer->renewForRemount(renewalEnvironment(caller)) + : renewer->renew(renewalEnvironment(caller)); + consumeRenewResult(result, caller); + return result; } uint64_t CasMountRuntime::renewRenewerForStartupOnce() { - auto call = admitRenewerCall(RenewalDriverState::Dormant, RenewalDriverState::StartupCall); - return renewRenewerOnce( - std::move(call), - RenewalDriverState::StartupCall, - /*propagate_failure=*/true); + return renewRenewerOnce(RenewCaller::Startup).attempt_start_boot_ms; } uint64_t CasMountRuntime::renewRenewerForRemountOnce() { - auto call = admitRenewerCall(RenewalDriverState::Parked, RenewalDriverState::RemountCall); - return renewRenewerOnce( - std::move(call), - RenewalDriverState::RemountCall, - /*propagate_failure=*/true); + return renewRenewerOnce(RenewCaller::Remount).attempt_start_boot_ms; } void CasMountRuntime::renewerReset() { + std::unique_ptr released; std::lock_guard lock(driver_mutex); - if (renewal_driver_state != RenewalDriverState::Dormant - && renewal_driver_state != RenewalDriverState::Parked) - throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: renewer reset while renewal is active"); - mount_renewer.reset(); + std::swap(mount_renewer, released); } ThreadFromGlobalPool CasMountRuntime::makeWorker(std::function body) @@ -715,58 +618,35 @@ void CasMountRuntime::startBackgroundWorkers(std::chrono::milliseconds period) { { std::lock_guard lock(driver_mutex); - if (renewal_driver_state != RenewalDriverState::Dormant - || workers_starting || workers_started - || renewal_worker.joinable() || remount_worker.joinable()) - throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: background workers cannot start in the current state"); + if (workers_started || renewal_worker.joinable()) + throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: the lease thread cannot start in the current state"); if (!mount_renewer || mount_renewer->state() != MountLeaseRenewerState::Active) - throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: background workers require an Active renewer"); - workers_starting = true; + throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: the lease thread requires an Active renewer"); + workers_started = true; workers_stop_requested = false; - worker_loops_released = false; renewal_period = period; - renewal_driver_state = RenewalDriverState::WorkerIdle; } - ThreadFromGlobalPool first; - ThreadFromGlobalPool second; + ThreadFromGlobalPool worker; try { - first = makeWorker([this] { renewalLoop(); }); - second = makeWorker([this] { remountLoop(); }); + worker = makeWorker([this] { renewalLoop(); }); } catch (...) { + /// No thread exists, so nothing is joined. { std::lock_guard lock(driver_mutex); workers_stop_requested = true; - worker_loops_released = true; - renewal_driver_state = RenewalDriverState::Stopping; - driver_cv.notify_all(); - } - if (first.joinable()) - first.join(); - if (second.joinable()) - second.join(); - { - std::lock_guard lock(driver_mutex); - workers_starting = false; workers_started = false; - renewal_driver_state = RenewalDriverState::Dormant; + driver_cv.notify_all(); } tripFenceWithoutOperationalLoss(); throw; } - { - std::lock_guard lock(driver_mutex); - renewal_worker = std::move(first); - remount_worker = std::move(second); - workers_starting = false; - workers_started = true; - worker_loops_released = true; - driver_cv.notify_all(); - } + std::lock_guard lock(driver_mutex); + renewal_worker = std::move(worker); } void CasMountRuntime::renewalLoop() @@ -775,79 +655,92 @@ void CasMountRuntime::renewalLoop() /// The tracker still counts this thread's allocations but never throws on it: a renewal failed by a /// memory limit costs the mount, and an exception outside the request ends this thread. LockMemoryExceptionInThread memory_exception_lock(VariableContext::Global); + + /// The wake condition of every wait in this loop; the reclaim backoff ignores a request, which is + /// the one it is retrying. + const auto woken = [this](bool by_request) { - std::unique_lock lock(driver_mutex); - driver_cv.wait(lock, [this] { return worker_loops_released; }); - if (workers_stop_requested || remountTerminal()) - return; - } + const bool wake = workers_stop_requested + || remountTerminal() + || (by_request && remount_requested_generation > remount_handled_generation); + if (!wake && config.lease_wait_predicate_false_hook_for_test) + config.lease_wait_predicate_false_hook_for_test(); + return wake; + }; + uint64_t backoff_ms = 1000; while (true) { if (config.renewal_before_driver_lock_hook_for_test) config.renewal_before_driver_lock_hook_for_test(); - AdmittedRenewerCall call; + bool reclaim = false; + uint64_t snapshot = 0; { std::unique_lock lock(driver_mutex); if (workers_stop_requested || remountTerminal()) return; - if (renewal_driver_state == RenewalDriverState::ParkRequested) + if (remount_requested_generation > remount_handled_generation) { - renewal_driver_state = RenewalDriverState::Parked; - driver_cv.notify_all(); + reclaim = true; + snapshot = remount_requested_generation; } - if (!renewalWorkerMayRenew()) + else if (!mount_renewer || mount_renewer->state() != MountLeaseRenewerState::Active) { - driver_cv.wait(lock, [this] - { - const bool terminal = remountTerminal(); - if (!workers_stop_requested - && !terminal - && !renewalWorkerMayRenew() - && config.renewal_parked_predicate_false_hook_for_test) - config.renewal_parked_predicate_false_hook_for_test(); - return workers_stop_requested || terminal || renewalWorkerMayRenew(); - }); - if (workers_stop_requested || remountTerminal()) - return; + driver_cv.wait(lock, [&] { return woken(/*by_request=*/true); }); continue; } + else + { + const uint64_t last_anchor = mount_renewer->lastCommittedAttemptStartBootMs(); + const uint64_t period_ms = static_cast(std::max(0, renewal_period.count())); + const uint64_t due = last_anchor > std::numeric_limits::max() - period_ms + ? std::numeric_limits::max() + : last_anchor + period_ms; + const uint64_t now = bootMsNow(); + if (now < due) + { + driver_cv.wait_for(lock, std::chrono::milliseconds(due - now), [&] { return woken(/*by_request=*/true); }); + continue; + } + } + } - const uint64_t last_anchor = mount_renewer->lastCommittedAttemptStartBootMs(); - const uint64_t period_ms = static_cast(std::max(0, renewal_period.count())); - const uint64_t due = last_anchor > std::numeric_limits::max() - period_ms - ? std::numeric_limits::max() - : last_anchor + period_ms; - const uint64_t now = bootMsNow(); - if (now < due) + if (reclaim) + { + bool reclaimed = false; + try + { + reclaimed = remount_attempt(); + } + catch (...) { - /// Every runtime notification can change the cadence decision: park/resume may happen - /// entirely while this worker is idle, and a remount may publish an already-overdue - /// renewer anchor. Re-sample state and BOOTTIME after any wake instead of retaining the - /// old relative wait until its wall-clock timeout. - driver_cv.wait_for(lock, std::chrono::milliseconds(due - now)); + tryLogCurrentException(getLogger("CasPool"), "CAS self-remount attempt failed"); + } + + std::unique_lock lock(driver_mutex); + if (reclaimed) + { + /// `armIfAdmissible` already acknowledged the generation the attempt served. This keeps a + /// callback that returns true without it from reclaiming the same generation again. + remount_handled_generation = std::max(remount_handled_generation, snapshot); + backoff_ms = 1000; continue; } - MountLeaseRenewer * renewer = mount_renewer.get(); - auto lease = std::make_unique(*this, RenewalDriverState::WorkerCall); - renewal_driver_state = RenewalDriverState::WorkerCall; - driver_cv.notify_all(); - call = AdmittedRenewerCall{std::move(lease), renewer}; + driver_cv.wait_for(lock, std::chrono::milliseconds(backoff_ms), [&] { return woken(/*by_request=*/false); }); + backoff_ms = std::min(backoff_ms * 2, 30000); + continue; } if (config.renewal_admitted_hook_for_test) config.renewal_admitted_hook_for_test(); try { - (void)renewRenewerOnce( - std::move(call), - RenewalDriverState::WorkerCall, - /*propagate_failure=*/false); + (void)renewRenewerOnce(RenewCaller::Loop); } catch (...) { - /// The worker path does not propagate renewal failures, so anything arriving here is this + /// The loop's renewal does not propagate renewal failures, so anything arriving here is this /// loop's own state machine reporting that it was driven out of contract. A background loop /// must not take the process down, but it must not keep driving a state machine that just /// proved wrong either: every later renewal would be unaudited. @@ -856,12 +749,6 @@ void CasMountRuntime::renewalLoop() /// thread's existence -- `mayMutate` requires `bootMsNow() < mount_fence.deadline_boot_ms`, so /// writes stop being admitted within one TTL whether or not anyone is renewing. Tripping the /// fence first brings that boundary forward instead of waiting for the TTL to lapse. - /// - /// Residual, deliberately not handled here: `scheduleRemount` also has an external caller, so - /// a later remount can still re-arm the fence and buy another bounded TTL. That makes the pool - /// flap rather than settle. Pairing this exit with a terminal publication would settle it, but - /// which terminal state means "this runtime's own driver broke" is a user-visible choice that - /// does not belong in a rescue path. tripMountLost(); tryLogCurrentException(getLogger("CasPool"), "CAS mount-lease renewal loop"); chassert(false); @@ -870,122 +757,28 @@ void CasMountRuntime::renewalLoop() } } -void CasMountRuntime::remountLoop() -{ - setThreadName(ThreadName::CAS_REMOUNT); - { - std::unique_lock lock(driver_mutex); - driver_cv.wait(lock, [this] { return worker_loops_released; }); - if (workers_stop_requested || remountTerminal()) - return; - } - - uint64_t backoff_ms = 1000; - while (true) - { - uint64_t snapshot; - { - std::unique_lock lock(driver_mutex); - driver_cv.wait(lock, [this] - { - return workers_stop_requested - || remountTerminal() - || remount_requested_generation > remount_handled_generation; - }); - if (workers_stop_requested || remountTerminal()) - return; - snapshot = remount_requested_generation; - - if (renewal_driver_state == RenewalDriverState::WorkerCall) - renewal_driver_state = RenewalDriverState::ParkRequested; - else if (renewal_driver_state == RenewalDriverState::WorkerIdle) - renewal_driver_state = RenewalDriverState::Parked; - driver_cv.notify_all(); - driver_cv.wait(lock, [this] - { - return workers_stop_requested || remountTerminal() - || renewal_driver_state == RenewalDriverState::Parked; - }); - if (workers_stop_requested || remountTerminal()) - return; - if (config.remount_parked_hook_for_test) - config.remount_parked_hook_for_test(); - } - - bool recovered = false; - try - { - recovered = remount_attempt(); - } - catch (...) - { - tryLogCurrentException(getLogger("CasPool"), "CAS self-remount attempt failed"); - } - - if (!recovered) - { - std::unique_lock lock(driver_mutex); - if (remountTerminal()) - return; - driver_cv.wait_for(lock, std::chrono::milliseconds(backoff_ms), [this] - { - return workers_stop_requested || remountTerminal(); - }); - if (workers_stop_requested || remountTerminal()) - return; - backoff_ms = std::min(backoff_ms * 2, 30000); - continue; - } - - backoff_ms = 1000; - { - std::lock_guard lock(driver_mutex); - if (workers_stop_requested || remountTerminal()) - return; - remount_handled_generation = std::max(remount_handled_generation, snapshot); - if (remount_requested_generation > remount_handled_generation) - continue; - if (lifecycle() == PoolLifecycle::Live - && mount_renewer - && mount_renewer->state() == MountLeaseRenewerState::Active) - { - renewal_driver_state = RenewalDriverState::WorkerIdle; - driver_cv.notify_all(); - } - } - } -} - void CasMountRuntime::stopBackgroundWorkers() { - ThreadFromGlobalPool * renewal_to_join = nullptr; - ThreadFromGlobalPool * remount_to_join = nullptr; + ThreadFromGlobalPool * to_join = nullptr; { std::lock_guard lock(driver_mutex); - if (!workers_started && !workers_starting - && !renewal_worker.joinable() && !remount_worker.joinable()) + if (!workers_started && !renewal_worker.joinable()) return; workers_stop_requested = true; - worker_loops_released = true; - renewal_driver_state = RenewalDriverState::Stopping; driver_cv.notify_all(); - renewal_to_join = &renewal_worker; - remount_to_join = &remount_worker; + to_join = &renewal_worker; } - if (renewal_to_join->joinable()) - renewal_to_join->join(); - if (remount_to_join->joinable()) - remount_to_join->join(); + if (to_join->joinable()) + to_join->join(); { std::lock_guard lock(driver_mutex); - workers_starting = false; workers_started = false; - renewal_driver_state = RenewalDriverState::Dormant; driver_cv.notify_all(); } } + bool CasMountRuntime::isVanished() const { const PoolLifecycle s = lifecycle(); @@ -1074,8 +867,8 @@ void CasMountRuntime::enterIdentityLost() /// `noteLeaseLost` (which only ever moves `Live -> TransientNotLive`, never away from it). It does NOT /// set `vanished_intent` (that latch is reserved for the `Vanished*` idempotency/FORGET protocol); /// rev.8 makes `IdentityLost` a fail-loud TERMINAL state through `remountTerminal`, which folds it - /// into the worker-exit boundary alongside `vanished_intent`, so the remount/GC workers - /// self-exit rather than demote. + /// into the worker-exit boundary alongside `vanished_intent`, so the lease and GC threads exit + /// rather than demote. /// `since` for the `identity_lost` snapshot row — the wall-clock instant the observer proved the /// sentinels gone. Stamped (release) BEFORE the CAS that publishes `IdentityLost`, so a reader that /// acquire-observes `IdentityLost` is guaranteed to observe this timestamp too (the winning CAS's @@ -1182,11 +975,11 @@ void CasMountRuntime::enterVanished(PoolLifecycle which, const String & reason) void CasMountRuntime::publishVanishedIntent() { - /// spec §5 step 1: publish the terminal-intent latch WITHOUT settling the state. The terminal consumer - /// and the remount loop both consult `vanished_intent` at their step boundaries, so - /// this stops new remount scheduling and makes an in-flight remount loop bail at its next step — - /// bounding FORGET's subsequent joins to one step + one backend timeout. The state store + WARN follow - /// in `enterVanished` (step 6). Idempotent. + /// spec §5 step 1: publish the terminal-intent latch WITHOUT settling the state. `scheduleRemount` + /// and the lease loop consult `vanished_intent` at their step boundaries, so this stops new remount + /// scheduling and makes the lease loop exit at its next step — bounding FORGET's join of the lease + /// thread to one step + one backend timeout. The state store + WARN follow in `enterVanished` + /// (step 6). Idempotent. { auto lock = lockTerminalPublication(); vanished_intent.store(true, std::memory_order_release); @@ -1201,10 +994,6 @@ void CasMountRuntime::scheduleRemount() if (workers_stop_requested || remountTerminal()) return; ++remount_requested_generation; - if (renewal_driver_state == RenewalDriverState::WorkerCall) - renewal_driver_state = RenewalDriverState::ParkRequested; - else if (renewal_driver_state == RenewalDriverState::WorkerIdle) - renewal_driver_state = RenewalDriverState::Parked; driver_cv.notify_all(); } @@ -1219,30 +1008,13 @@ void CasMountRuntime::beginShutdownForTest() { std::lock_guard lock(driver_mutex); workers_stop_requested = true; - renewal_driver_state = RenewalDriverState::Stopping; driver_cv.notify_all(); } -RenewalDriverState CasMountRuntime::renewalDriverStateForTest() const -{ - std::lock_guard lock(driver_mutex); - return renewal_driver_state; -} - -void CasMountRuntime::waitForRenewalDriverStateForTest(RenewalDriverState expected) const -{ - std::unique_lock lock(driver_mutex); - if (!driver_cv.wait_for(lock, std::chrono::seconds(20), [this, expected] - { - return renewal_driver_state == expected; - })) - throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: timed out waiting for renewal driver state"); -} - bool CasMountRuntime::workersRunningForTest() const { std::lock_guard lock(driver_mutex); - return workers_started && renewal_worker.joinable() && remount_worker.joinable(); + return workers_started && renewal_worker.joinable(); } uint64_t CasMountRuntime::remountRequestedGenerationForTest() const diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index e8fe587b14bf..6d51e96af28f 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -33,7 +33,7 @@ using PartWriteTxnPtr = std::shared_ptr; /// - `TransientNotLive` — the lease was lost; access is uncertain and a self-remount retries. The §2 /// `Present`+identity-match recovery rule fires only from here (or `Live`). /// - `IdentityLost` — the pool sentinels are authoritatively absent (both KeyAbsent): -/// fail-loud and TERMINAL (rev.8). The remount/GC workers self-exit; +/// fail-loud and TERMINAL (rev.8). The lease and GC threads exit; /// matching-sentinel reappearance does NOT auto-revive it ([D3]); recovery is a /// restart or `SYSTEM CAS FORGET`. /// - `Vanished*` — fully terminal truth: the data root was replaced by a foreign pool, or the @@ -47,19 +47,6 @@ enum class PoolLifecycle : uint8_t VanishedForgotten, }; -enum class RenewalDriverState : uint8_t -{ - Dormant, - StartupCall, - DirectCall, - RemountCall, - WorkerIdle, - WorkerCall, - ParkRequested, - Parked, - Stopping, -}; - using RuntimeWorkerFactory = std::function)>; /// Configuration owned by `CasMountRuntime`. `PoolConfig::mountConfig` projects the flat pool settings @@ -75,25 +62,17 @@ struct MountConfig std::function boot_ms_fn = {}; std::function wait_sleep_fn = {}; RuntimeWorkerFactory worker_factory = {}; - /// Deterministic test interposition after the remount worker has confirmed renewal is parked, - /// immediately before it releases `driver_mutex` and begins the real remount callback. - /// Runs with `driver_mutex` held: it must issue no backend request and never wait on the pool's - /// hot-key lane, whose holders sleep under that mutex, or the test deadlocks itself. - std::function remount_parked_hook_for_test = {}; - /// Deterministic test interposition at the top of the renewal loop, before it acquires - /// `driver_mutex` to inspect cadence or parking state. + /// Deterministic test interposition at the top of each pass of the lease loop, before it takes + /// `driver_mutex`. std::function renewal_before_driver_lock_hook_for_test = {}; - /// Deterministic test interposition after a due worker has atomically reserved renewal ownership - /// and captured its renewer, but before renewer/backend I/O starts. + /// Deterministic test interposition after the loop decided to renew and released `driver_mutex`, + /// before the renewal's I/O starts. std::function renewal_admitted_hook_for_test = {}; - /// Deterministic test interposition after terminal ownership has been deposited and the renewer is - /// no longer reachable by the completed call. - std::function renewal_terminal_deposited_hook_for_test = {}; - /// Deterministic test interposition after the parked renewal predicate has sampled terminal false, - /// immediately before the condition-variable wait atomically releases `driver_mutex`. + /// Deterministic test interposition after a wait predicate of the lease loop sampled false, + /// immediately before the wait atomically releases `driver_mutex`. /// Runs with `driver_mutex` held: it must issue no backend request and never wait on the pool's /// hot-key lane, whose holders sleep under that mutex, or the test deadlocks itself. - std::function renewal_parked_predicate_false_hook_for_test = {}; + std::function lease_wait_predicate_false_hook_for_test = {}; /// Deterministic test interposition immediately before a terminal publisher attempts to acquire /// `driver_mutex`. std::function terminal_publication_waiting_for_driver_lock_hook_for_test = {}; @@ -137,7 +116,7 @@ struct MountFence /// Owns the live writer-incarnation mechanics shared by the pool's mount and recovery orchestration: /// the `MountLeaseRenewer`, local `MountFence`, build watermark and in-flight build registry, -/// `live_writer_epoch`, unclean-boundary marker, and both persistent workers. `Pool` retains the higher-level +/// `live_writer_epoch`, unclean-boundary marker, and the lease thread. `Pool` retains the higher-level /// claim/recovery sequence and its `remount_mutex`; in particular, the runtime does not acquire or own /// the ref-ledger locks. The runtime receives its backend, layout, configuration, event sink, request /// budget, and a callback that performs one pool-level remount attempt, so it has no `Pool` back-reference. @@ -148,8 +127,9 @@ class CasMountRuntime CasMountRuntime( BackendPtr backend_ptr_, /// The planes the `MountLeaseRenewer` runs on: a bounded renewal under the mount fence, the - /// claim and the farewell on an open one, and the worker's renewal on `lease_requests_`, which - /// has no lease budget and whose sleep a stop wakes. Owned by `Pool` and outliving this runtime. + /// claim and the farewell on an open one, and the lease loop's renewal on `lease_requests_`, + /// which has no lease budget and whose sleep a stop wakes. Owned by `Pool` and outliving this + /// runtime. CasRequests & mount_requests_, CasRequests & farewell_requests_, CasRequests & lease_requests_, @@ -159,7 +139,7 @@ class CasMountRuntime const CasEventSink & event_sink_, CasRequestBudget cas_request_budget_, /// One pool-level recovery attempt. The callback captures the owning `Pool` and is invoked only - /// after construction, from the recovery thread. + /// after construction, from the lease thread. std::function remount_attempt_); /// ---- per-server watermark and identity ---- @@ -183,7 +163,8 @@ class CasMountRuntime void tripMountLost(); /// Publish the BOOTTIME deadline from a successful lease renewal. void setMountDeadline(uint64_t deadline_boot_ms); - /// Arm a new lease incarnation and clear any loss latched for the prior incarnation. + /// Arm a new lease incarnation and clear any loss latched for the prior incarnation. Unconditional + /// and reports no `Live`; a reclaim arms through `armIfAdmissible`. void armMountFence(UInt128 server_uuid, uint64_t writer_epoch, uint64_t deadline_boot_ms); /// Step 0 of a reclaim. One step under `driver_mutex`: record the requested generation this attempt /// serves and latch the fence. @@ -233,15 +214,14 @@ class CasMountRuntime /// Whether the terminal-intent latch (`vanished_intent`) is published — set by a natural /// `enterVanished`, OR EARLY (spec §5 step 1) by FORGET's `publishVanishedIntent`, and NEVER by the /// non-absorbing `IdentityLost` ([C1]). This is the EARLIEST terminal signal: it can already be true - /// while the state is still pre-terminal (mid-FORGET). Consulted alongside `isVanished()` by every - /// background worker that must self-exit the moment the pool is (being driven) terminal — the renewer - /// callback (`scheduleRemount`), the remount loop, and the GC scheduler. + /// while the state is still pre-terminal (mid-FORGET). Consulted alongside `isVanished()` wherever + /// background work must stop the moment the pool is (being driven) terminal: `scheduleRemount`, the + /// lease loop and the GC scheduler. bool vanishedIntentPublished() const { return vanished_intent.load(std::memory_order_acquire); } /// Non-terminal lease-loss transition: `Live -> TransientNotLive`. Idempotent and lock-free; a /// compare-exchange FROM `Live` only, so it never downgrades a terminal state. `tripMountLost` - /// calls this (the lease-loss primitive), and the remount loop's identity gate calls it as its - /// first step so a direct/forced remount attempt has a valid non-terminal predecessor state. + /// calls this (the lease-loss primitive). /// Returns true only to the compare-exchange winner, which owns the one-per-loss metric. bool noteLeaseLost(); /// Non-terminal recovery transition: `TransientNotLive -> Live`. Called after a self-remount @@ -252,7 +232,7 @@ class CasMountRuntime /// One-way terminal transition to `IdentityLost`, from `TransientNotLive` only (a compare-exchange /// FROM `TransientNotLive`, so it is idempotent and cannot fire from `Live`/`Vanished`). On the /// transition it emits ONE WARN and one `CASIdentityLost` ProfileEvent. rev.8: `IdentityLost` is a - /// fail-loud TERMINAL state — `remountTerminal` reports it, so the remount worker self-exits + /// fail-loud TERMINAL state — `remountTerminal` reports it, so the lease thread exits /// (and the GC scheduler self-exits, through `Pool`) at its next boundary; there is no demoted observer. /// It deliberately does NOT publish the `vanished_intent` latch (which is reserved for the `Vanished*` /// idempotency/FORGET protocol); `remountTerminal` widens the worker-exit boundary to include it. @@ -265,17 +245,17 @@ class CasMountRuntime void setLifecycleForTest(PoolLifecycle lc); /// Publish the terminal-intent latch (`vanished_intent`) WITHOUT settling the lifecycle state. This is - /// spec §5 step 1 of `SYSTEM CAS FORGET`: publishing the latch FIRST makes the runtime terminal - /// consumer stop latching remount generations and the remount loop bail at its next step boundary, so - /// FORGET's subsequent worker joins are bounded to one step + one backend timeout. The state store + WARN happen + /// spec §5 step 1 of `SYSTEM CAS FORGET`: publishing the latch FIRST makes the runtime stop latching + /// remount generations and the lease loop exit at its next step boundary, so FORGET's join of the + /// lease thread is bounded by one step and one backend timeout. The state store + WARN happen /// later, in `enterVanished` at step 6. Idempotent. Publication is serialized by `driver_mutex` and - /// followed by a condition-variable notification, so a worker cannot miss the terminal edge between - /// its predicate sample and wait. A natural + /// followed by a condition-variable notification, so the lease loop cannot miss the terminal edge + /// between its predicate sample and wait. A natural /// terminal transition does NOT call this — its `enterVanished` publishes the latch itself. void publishVanishedIntent(); /// One-way transition to a fully-terminal `Vanished` value (spec §3). Publishes the terminal-intent - /// latch (so the runtime stops scheduling remount work and the remount loop exits at its next step + /// latch (so the runtime stops scheduling remount work and the lease loop exits at its next step /// boundary) if it is not already published, records `reason`, stores the state, then emits ONE WARN + /// one `CASDataRootVanished` ProfileEvent. Idempotent: the first terminal STATE transition wins (a /// dedicated latch keyed separately from `vanished_intent`, because FORGET publishes that intent latch @@ -336,12 +316,12 @@ class CasMountRuntime /// `PUT` or resolve read. Runs on the renewing thread. void noteRenewRequest(const MountRenewRequestEvent & event) noexcept; - /// TRUE once the pool has reached — or is being driven toward — a state on which the self-remount - /// worker must stop: a published terminal `Vanished` intent (`vanished_intent` — set early by + /// TRUE once the pool has reached — or is being driven toward — a state on which the lease thread + /// must stop: a published terminal `Vanished` intent (`vanished_intent` — set early by /// FORGET, or by a natural `enterVanished`, and already subsuming every settled `Vanished*` state since /// it is published before the state store) OR `IdentityLost` (a fail-loud TERMINAL state — no - /// demoted observer; recovery is restart or FORGET). Consulted by `scheduleRemount` before arming and by - /// the remount loop at every step boundary. (The GC scheduler applies the same three-way test through + /// demoted observer; recovery is restart or FORGET). Consulted by `scheduleRemount`, by the arming rule + /// and by the lease loop at every step boundary. (The GC scheduler applies the same three-way test through /// `Pool`.) bool remountTerminal() const { @@ -350,9 +330,9 @@ class CasMountRuntime } /// The inter-attempt sleep the mount plane runs on. A plain sleep would hold a stopping renewal - /// for the whole capped backoff; this one wakes on the same stop signal the workers watch. - /// A park or a remount request does not wake it: the wait runs out, at most one spacing draw for - /// the background renewal. It shortens a stop, not a fence loss: the fence cannot see a stop request, + /// for the whole capped backoff; this one wakes on the stop signal the lease thread watches. + /// A remount request does not wake it: the wait runs out, at most one spacing draw for the loop's + /// renewal. It shortens a stop, not a fence loss: the fence cannot see a stop request, /// so a woken operation still reissues unless its own liveness predicate refuses. void sleepInterruptibly(uint64_t ms); @@ -389,7 +369,7 @@ class CasMountRuntime /// Publish the live-incarnation `live_writer_epoch` with release ordering. void setLiveWriterEpoch(uint64_t v); - /// ---- mount-lease renewer and persistent workers ---- + /// ---- mount-lease renewer and the lease thread ---- void installRenewer(UInt128 our_uuid, uint64_t writer_epoch, const std::function & now_ms); uint64_t startRenewer(); uint64_t renewRenewerForStartupOnce(); @@ -397,7 +377,7 @@ class CasMountRuntime void renewerReset(); void startBackgroundWorkers(std::chrono::milliseconds period); void stopBackgroundWorkers(); - /// Latch a recovery generation. Persistent remount ownership means this never constructs a thread. + /// Latch a recovery generation for the lease thread. It never constructs a thread. void scheduleRemount(); bool scheduleRemountForTest(); void beginShutdownForTest(); @@ -408,12 +388,10 @@ class CasMountRuntime return schedule_remount_calls_for_test.load(std::memory_order_relaxed); } - RenewalDriverState renewalDriverStateForTest() const; - void waitForRenewalDriverStateForTest(RenewalDriverState expected) const; bool workersRunningForTest() const; uint64_t remountRequestedGenerationForTest() const; - /// Join both persistent workers before an `Active` renewer may write its clean farewell. + /// Join the lease thread before an `Active` renewer may write its clean farewell. void finishTeardown(bool drained); /// Sleep through the injected test hook when present; otherwise use the production thread sleep. @@ -423,8 +401,8 @@ class CasMountRuntime /// through a scenario (e.g. driving a second incarnation's renewal from inside the observed /// incarnation's own poll) cannot express that through `PoolConfig::wait_sleep_fn` alone, since /// that value is fixed at open time. Unsynchronized against `waitSleep`'s `const` read of the same - /// field: safe only called from the test's own thread before any worker is running (no persistent - /// renewal/remount worker reads `config.wait_sleep_fn` concurrently with this write). + /// field: safe only called from the test's own thread before the lease thread starts (nothing else + /// reads `config.wait_sleep_fn` concurrently with this write). void setWaitSleepForTest(std::function fn) { config.wait_sleep_fn = std::move(fn); } /// Forward renewer events to the injected sink. The sink is held by reference so it observes the @@ -432,56 +410,39 @@ class CasMountRuntime void emitEvent(CasEvent && e) const { if (event_sink) event_sink(std::move(e)); } private: - class DriverLease + /// Who drives a renewal. Only the loop's renewal is unbounded, counts its requests as they are + /// sent, and raises a remount request when it ends terminal. + enum class RenewCaller : uint8_t { - public: - DriverLease(CasMountRuntime & runtime_, RenewalDriverState active_); - ~DriverLease(); - RenewalDriverState finish(RenewalDriverState ordinary_destination, const MountRenewResult * result = nullptr); - - private: - CasMountRuntime & runtime; - RenewalDriverState active; - bool finished = false; + Loop, + Startup, + Remount, + Direct, }; - struct AdmittedRenewerCall + /// A committed renewal that ended an expiry: how long the lease was expired, and the text of the + /// last failed request. + struct RestoredLease { - std::unique_ptr lease; - MountLeaseRenewer * renewer = nullptr; + uint64_t expired_ms = 0; + String last_failure; }; - /// The renewal worker may drive a renewal only while it exclusively owns the driver and the renewer - /// is Active. Requires `driver_mutex`. `admitRenewerCall` enforces the same three conditions for every - /// other driver; the worker loop must park rather than throw when they do not hold, so it needs the - /// predicate separately. Both the park test and the wake predicate use this one definition, so they - /// cannot drift apart. - bool renewalWorkerMayRenew() const; - - AdmittedRenewerCall admitRenewerCall(RenewalDriverState required, RenewalDriverState active); - uint64_t renewRenewerOnce( - AdmittedRenewerCall call, - RenewalDriverState active, - bool propagate_failure); - MountRenewOperationEnvironment renewalEnvironment(bool worker_call); + MountRenewResult renewRenewerOnce(RenewCaller caller); + MountRenewOperationEnvironment renewalEnvironment(RenewCaller caller); std::optional leaseExpiredAt(uint64_t now_boot_ms) const; - /// Publishes a committed renewal's deadline. A deadline in the future ends the current run of - /// trouble: it clears the failure text and, when the lease was expired, counts and logs the - /// restore. Returns how long the lease had been expired when this deadline restores it; empty - /// when it was not expired or is still expired. - std::optional publishRenewedDeadline(uint64_t deadline_boot_ms); - void consumeRenewResult( - const MountRenewResult & result, - RenewalDriverState active_state, - RenewalDriverState returned_state, - bool propagate_failure); + /// Publishes a committed renewal's deadline. Requires `driver_mutex`. A deadline in the future ends + /// the current run of trouble: it clears the failure text and, when the lease was expired, counts + /// the restore and returns it for the caller to log after the unlock. Empty when the lease was not + /// expired or is still expired. + std::optional publishRenewedDeadline(uint64_t deadline_boot_ms); + void consumeRenewResult(const MountRenewResult & result, RenewCaller caller); void renewalLoop(); - void remountLoop(); ThreadFromGlobalPool makeWorker(std::function body); - /// The renewal's liveness: a shutdown request and, for the worker, a parked or park-requested - /// driver, a pool that left `Live`, or a lost fence -- the worker's plane has no fence of its own. - /// FALSE ends the renewal. - bool renewalLive(bool worker_call) const; + /// The renewal's liveness: no stop requested and, for the loop, no pending remount request, a pool + /// that is `Live` and a fence that is not lost -- the loop's plane has no fence of its own. FALSE + /// ends the renewal. + bool renewalLive(RenewCaller caller) const; /// Whether this node has already been asked to stop. Sampled ONCE, before the write, so a refusal /// caused by the stop cannot be mistaken for one that preceded it. bool renewalCancelled() const; @@ -514,7 +475,7 @@ class CasMountRuntime /// active_build_seqs holds the seqs of in-flight builds, so `minActive` yields the GC floor. The floor /// is published by the merged `mount_renewer` /// beat (there is no standalone watermark object anymore). ATOMIC because a self-remount re-stamps it - /// (kept equal to `live_writer_epoch`) from the runtime-owned remount worker while `epoch`/`writerEpoch` + /// (kept equal to `live_writer_epoch`) from the lease thread while `epoch`/`writerEpoch` /// may observe it; the ref-lane hot readers were moved to `liveWriterEpoch`, so this now backs only /// the identity accessors. std::atomic process_epoch{0}; @@ -527,23 +488,20 @@ class CasMountRuntime std::map> inflight_builds; /// Synchronous mount-lease protocol state. Constructed and started on a writable open after the - /// owner/epoch/mount startup protocol; the runtime-owned renewal worker is its sole background - /// driver and publishes successful anchors or terminal loss into the local fence. After both - /// workers join, teardown releases an `Active` renewer so a same-server reopen can reclaim - /// immediately. Null on a read-only open. + /// owner/epoch/mount startup protocol; while the lease thread runs it is the renewer's only driver + /// and publishes successful anchors or terminal loss into the local fence. After the lease thread + /// joins, teardown releases an `Active` renewer so a same-server reopen can reclaim immediately. + /// Null on a read-only open. std::unique_ptr mount_renewer; std::atomic live_writer_epoch{0}; - /// One mutex/condition pair owns driver admission, worker lifecycle, cadence, and the remount - /// generation latch, and terminal predicates paired with `driver_cv`. It is never held across - /// renewer/backend calls, remount callbacks, logging, or joins. + /// One mutex/condition pair guards the lease thread's lifecycle, the cadence, the remount + /// generations, and the terminal predicates paired with `driver_cv`. It is never held across a + /// renewer or backend call, the reclaim, logging or a join. mutable std::mutex driver_mutex; mutable std::condition_variable driver_cv; - RenewalDriverState renewal_driver_state = RenewalDriverState::Dormant; - bool workers_starting = false; bool workers_started = false; - bool worker_loops_released = false; bool workers_stop_requested = false; std::chrono::milliseconds renewal_period{0}; uint64_t remount_requested_generation = 0; @@ -551,7 +509,6 @@ class CasMountRuntime /// The requested generation `beginReclaim` recorded; `armIfAdmissible` acknowledges it. uint64_t reclaim_generation = 0; ThreadFromGlobalPool renewal_worker; - ThreadFromGlobalPool remount_worker; /// Counted entries into `scheduleRemount`; retained as a test-only observability seam. std::atomic schedule_remount_calls_for_test{0}; @@ -572,15 +529,15 @@ class CasMountRuntime /// The pool lifecycle condition (rev.7 §1). Starts `Live`. Non-terminal transitions /// (`noteLeaseLost`/`noteRemounted`) are lock-free compare-exchanges guarded by their exact - /// predecessor state; terminal predicate publication is serialized with worker waits by + /// predecessor state; terminal predicate publication is serialized with the lease loop's waits by /// `driver_mutex`, while the caller's `Pool::remount_mutex` serializes the higher-level remount flow. std::atomic pool_lifecycle{PoolLifecycle::Live}; /// Terminal-intent latch (spec §3), published before the state store — by `enterVanished` for a /// natural transition, or EARLY (step 1) by `publishVanishedIntent` for FORGET. Only the fully-terminal /// `Vanished*` transition sets it — `IdentityLost` deliberately does NOT (rev.8 folds IdentityLost into /// the worker-exit boundary via `remountTerminal` instead). Consulted (with `IdentityLost`) by - /// `remountTerminal`, so a terminal pool's runtime consumer never schedules a remount and the remount - /// loop bails at its next step boundary — no claim/allocate/write after the pool is (being driven) terminal. + /// `remountTerminal`, so a terminal pool's runtime consumer never schedules a remount and the lease + /// loop exits at its next step boundary — no claim/allocate/write after the pool is (being driven) terminal. std::atomic vanished_intent{false}; /// Idempotency guard for the terminal STATE transition (`enterVanished`'s body). Distinct from /// `vanished_intent`: FORGET publishes that intent latch at step 1, so it can no longer serve as the diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp index 5cd49334efb2..396e03a28f2e 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp @@ -912,11 +912,11 @@ void Pool::mountWritable(PoolPtr & store, UInt128 our_uuid, MountClaimPolicy pol claim_anchor_boot_ms > std::numeric_limits::max() - ttl_ms_u ? std::numeric_limits::max() : claim_anchor_boot_ms + ttl_ms_u); - /// Gate the two persistent runtime workers with `background_watermark`: they run only in production + /// Gate the lease thread with `background_watermark`: it runs only in production /// (`background_watermark` = context != nullptr && !read_only), never in unit tests — which /// drive `renewWatermarkOnce` explicitly and rely on the armed sub-TTL deadline, never on a loop. /// The synchronous renewer is still started above (it must adopt the mount and arm the fence on - /// every writable open); only the worker pair is conditional. The merged + /// every writable open); only the lease thread is conditional. The merged /// heartbeat renews at `mount_renew_period` — one beat now renews the lease and the floor. if (store->config.background_watermark) store->mount_runtime.startBackgroundWorkers(store->config.mount_renew_period); @@ -1021,7 +1021,7 @@ Pool::~Pool() } }; - /// 1. Stop and join both persistent mount-runtime workers before draining or releasing the renewer. + /// 1. Stop and join the lease thread before draining or releasing the renewer. guarded([this] { if (config.teardown_phase1_throw_for_test) @@ -1145,12 +1145,12 @@ void Pool::setDetachedDrainDeadlineBudgetForTest(const CasRequestBudget & budget void Pool::forgetDisk(const std::function & stop_and_join_gc, const String & reason) { - /// Hazard C6: FORGET joins both mount-runtime workers (and, via `stop_and_join_gc`, the GC threads), so it - /// MUST run on the admin/query thread — never a pool thread, whose join of itself would deadlock. The - /// guard is a programming-error assertion (a self-join hangs; it never corrupts), so a chassert is the - /// right severity, not a release fail-close. + /// FORGET joins the lease thread (and, via `stop_and_join_gc`, the GC threads), so it MUST run on the + /// admin/query thread — never a pool thread, whose join of itself would deadlock. The guard is a + /// programming-error assertion (a self-join hangs; it never corrupts), so a chassert is the right + /// severity, not a release fail-close. const ThreadName tn = getThreadName(); - chassert(tn != ThreadName::CAS_REMOUNT && tn != ThreadName::CAS_GC_SCHEDULER + chassert(tn != ThreadName::CAS_LEASE_RENEWER && tn != ThreadName::CAS_GC_SCHEDULER && tn != ThreadName::CAS_GC_HEARTBEAT && "SYSTEM CAS FORGET must not run on a CAS pool thread (self-join deadlock)"); @@ -1186,7 +1186,7 @@ void Pool::forgetDisk(const std::function & stop_and_join_gc, const Stri if (stop_and_join_gc) stop_and_join_gc(); - /// (5a) Stop and join both persistent mount-runtime workers outside `remount_mutex`. + /// (5a) Stop and join the lease thread outside `remount_mutex`. mount_runtime.stopBackgroundWorkers(); /// A reclaim in flight when step 1 published the intent finishes its current step, and the arming @@ -1209,7 +1209,7 @@ void Pool::forgetDisk(const std::function & stop_and_join_gc, const Stri /// The pool object OUTLIVES this FORGET (it stays registered, `Vanished(forgotten)`, until DROP/restart), /// so `~Pool` will re-run the same teardown. Drop the renewer now so that later teardown finds none and /// skips it: `MountLeaseRenewer::release` is admitted only from `Active`, so a renewer already released - /// here must not be released again. `renewerReset` is safe now: both renewer-driving workers are joined. + /// here must not be released again. `renewerReset` is safe now: the lease thread is joined. mount_runtime.renewerReset(); /// (6) Publish the terminal state + WARN, under remount serialization — matching the natural-transition @@ -1260,7 +1260,7 @@ bool Pool::tryRemountOnce() String error; SCOPE_EXIT( { - /// A parked redo completed while the whole-chain serializer was held. Drain its POD snapshot + /// A remount re-anchor completed while the whole-chain serializer was held. Drain its POD snapshot /// first, after lock destruction, so renewal recovery/failure precedes and correlates with the /// containing remount result without any callback or allocation under `remount_mutex`. deliverDeferredMountRenewObservability(attempt_no); @@ -1391,7 +1391,7 @@ bool Pool::tryRemountOnce() break; /// fall through to the existing fresh-incarnation recovery. case LifecycleGateVerdict::Replaced: /// NOT while a FORGET is in progress (spec §9 rev.8 item 7). `forgetDisk` publishes the - /// terminal-intent latch at step 1 (`publishVanishedIntent`), then joins the remount worker; + /// terminal-intent latch at step 1 (`publishVanishedIntent`), then joins the lease thread; /// a `tryRemountOnce` already IN FLIGHT — one that passed the step-0 `isVanished()` gate /// BEFORE the intent was published, which that gate therefore cannot catch — could otherwise /// settle `Vanished(replaced)` mid-FORGET, stranding FORGET's own @@ -1405,7 +1405,7 @@ bool Pool::tryRemountOnce() case LifecycleGateVerdict::IdentityLost: /// Both sentinels authoritatively absent. Enter `IdentityLost` once (from `TransientNotLive`); /// a repeat probe while already `IdentityLost` is a no-op. rev.8: `IdentityLost` is a - /// fail-loud TERMINAL state — the remount worker self-exits at its next boundary (see + /// fail-loud TERMINAL state — the lease thread exits at its next boundary (see /// `CasMountRuntime::remountTerminal`), so there is no demoted observer. if (mount_runtime.lifecycle() != PoolLifecycle::IdentityLost) mount_runtime.enterIdentityLost(); @@ -1503,8 +1503,8 @@ bool Pool::tryRemountOnce() /// build's own tests the moment someone reuses an epoch across a remount. chassert(writer_epoch > mount_runtime.liveWriterEpoch()); - /// The persistent renewal worker is parked before this callback is entered, so renewer - /// replacement cannot race any synchronous lease operation. + /// This runs on the lease thread, or with no lease thread running, so no renewal is in flight + /// while the renewer is replaced. step = "renewer_install"; mount_runtime.installRenewer(our_uuid, writer_epoch, now_ms); step = "renewer_start"; @@ -1537,7 +1537,7 @@ bool Pool::tryRemountOnce() config.remount_quiesce_hook_for_test(); /// Quiescence may consume most of the new lease. The same renewal-window gate used at startup - /// admits one synchronous parked redo while at least one physical attempt still fits the old + /// admits one synchronous re-anchor while at least one physical attempt still fits the old /// authority window and before the fence is armed. const uint64_t safety_ms = config.cas_request_budget.lease_safety_margin_ms; const uint64_t safe_deadline = remount_anchor_boot_ms > std::numeric_limits::max() - ttl_ms @@ -1589,7 +1589,7 @@ bool Pool::tryRemountOnce() } } -/// The persistent self-remount and merged-heartbeat renewal workers live in `mount_runtime` +/// The lease thread, which renews and runs the self-remount, lives in `mount_runtime` /// (`Pool/CasMountRuntime.h`); these are thin delegates. `mount_runtime`'s `remount_attempt` callback is /// bound to `Pool::tryRemountOnce` (the claim/recovery orchestration that stays on Pool). bool Pool::scheduleRemountForTest() @@ -1788,7 +1788,7 @@ void Pool::reportImpossibleInterference(const String & key, const String & reaso }); /// Incidental-only detection has the same fail-closed reaction as a foreign/superseded lease - /// renewal. The fence and persistent self-remount ownership live on `mount_runtime`. + /// renewal. The fence and the lease thread, which runs the self-remount, live on `mount_runtime`. mount_runtime.tripMountLost(); mount_runtime.scheduleRemount(); diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h index c1dd8de03781..45048640540a 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h @@ -182,7 +182,7 @@ struct PoolConfig /// overlap. `1` issues no read-ahead and no fan-out at all and is the sequential round, request /// for request. uint64_t gc_io_concurrency = 16; - /// Tests drive `renewWatermarkOnce` explicitly; gates both persistent runtime workers. + /// Tests drive `renewWatermarkOnce` explicitly; gates the lease thread. bool background_watermark = false; /// Installed on the pool before a writable mount can start its runtime-owned workers. CasEventSink event_sink = {}; @@ -552,15 +552,15 @@ class Pool : public std::enable_shared_from_this /// `SYSTEM CAS FORGET` — the operator force-Vanish (spec §5). Drives THIS pool to /// `Vanished(forgotten)` with the fence-first protocol, node-locally, regardless of the current /// lifecycle (it works precisely on a NOT-live disk — a stuck transient/`IdentityLost` pool). In order: - /// (1) publish the terminal-intent latch FIRST (so both runtime worker loops bail at their + /// (1) publish the terminal-intent latch FIRST (so the lease loop exits at its /// next step boundary, bounding the joins below); (2) trip the local fence (the deliberate /// decommission act, allowed on a live disk); (3+4) stop the GC scheduler via `stop_and_join_gc` — /// injected because the scheduler is owned above the Pool, a no-op in contexts that run none — and stop - /// + join both persistent workers; (5) drain the ref lanes (bounded) and retire the renewer WITHOUT an + /// + join the lease thread; (5) drain the ref lanes (bounded) and retire the renewer WITHOUT an /// unearned clean farewell (the lease expires by observation unless the lanes provably drained); then /// (6) publish `Vanished(forgotten)` carrying `reason` (the [D5] message with the operator's decommission /// timestamp). Idempotent: an already-`Vanished` pool returns immediately (first terminal transition - /// wins). MUST run on the admin/query thread, never a pool (remount/GC) thread — the joins would + /// wins). MUST run on the admin/query thread, never a pool (lease or GC) thread — the joins would /// otherwise self-deadlock (hazard C6). void forgetDisk(const std::function & stop_and_join_gc, const String & reason); @@ -827,7 +827,8 @@ class Pool : public std::enable_shared_from_this bool tryRemountOnce(); /// Test seam: latch the private self-remount path directly. In production the runtime terminal - /// consumer calls `scheduleRemount`, while external loss paths may latch the same persistent worker. + /// consumer calls `scheduleRemount`, while external loss paths raise the same generation for the lease + /// thread. /// Returns true iff an unhandled recovery generation exists after the call. bool scheduleRemountForTest(); /// Test seam: how many times `scheduleRemount` has been ENTERED, counted @@ -1175,8 +1176,8 @@ class Pool : public std::enable_shared_from_this return std::forward(mutation)(); } - /// The mount plane's inter-attempt sleep: woken only by a stop of the workers, so a stopping renewal - /// is not held for a whole capped backoff. A park or a remount request does not wake it; the wait + /// The mount plane's inter-attempt sleep: woken only by a stop of the lease thread, so a stopping + /// renewal is not held for a whole capped backoff. A remount request does not wake it; the wait /// runs out. Named rather than inlined because the test seam has to be able to put it back. std::function mountPlaneSleepFn() { @@ -1212,8 +1213,8 @@ class Pool : public std::enable_shared_from_this mutable CasRequests mount_requests; mutable CasRequests farewell_requests; mutable CasRequests gc_requests; - /// The worker renewal's plane: no lease budget, because the renewal keeps trying after the lease - /// expired; a stop, a park or a terminal lifecycle ends it through its liveness. + /// The lease loop's renewal plane: no lease budget, because the renewal keeps trying after the lease + /// expired; a stop, a remount request or a terminal lifecycle ends it through its liveness. mutable CasRequests lease_requests; std::shared_ptr detached_work = std::make_shared(); @@ -1259,7 +1260,7 @@ class Pool : public std::enable_shared_from_this /// from Pool. Owns the `MountLeaseRenewer`, the local `MountFence`, the per-server /// build watermark (`process_epoch` + the `builds_mutex`-guarded seq/registry) and its in-flight-build /// map, the live-incarnation `live_writer_epoch`, the unclean-epoch high-water-mark, and the - /// persistent renewal and remount workers (with one driver mutex/condition pair). Injected with backend/layout + /// lease thread (with one driver mutex/condition pair). Injected with backend/layout /// the `MountConfig` slice + `server_root_id` + the event-sink reference + the pool `cas_request_budget` /// + a `remount_attempt` callback (== `Pool::tryRemountOnce`, which STAYS on Pool: the claim/recovery /// ORCHESTRATION drives these owned primitives). @@ -1276,7 +1277,7 @@ class Pool : public std::enable_shared_from_this CasMountRuntime mount_runtime; /// Serializes `tryRemountOnce` (whose claim/recovery ORCHESTRATION stays on Pool). STAYS here with - /// its guarded critical section: persistent worker ownership, fence atomics, and the build registry + /// its guarded critical section: lease-thread ownership, fence atomics, and the build registry /// moved to `mount_runtime`, but the top-level remount serialization guards /// the Pool-side orchestration, so it stays on Pool. std::mutex remount_mutex; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp index a0ef5b43b888..cdf82fc68a47 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp @@ -147,8 +147,8 @@ struct MountRenewObservabilityConfiguration }; /// Event sinks may synchronously renew another Pool on the same thread. A fixed stack keeps every -/// registered outer per-call snapshot stable without allocation, including while a parked redo holds -/// `remount_mutex`. Overflow suppresses rich event/log delivery for the nested call rather than +/// registered outer per-call snapshot stable without allocation, including while a remount re-anchor +/// holds `remount_mutex`. Overflow suppresses rich event/log delivery for the nested call rather than /// aliasing an outer call or changing protocol behavior; physical attempt truth is independently /// retained by the stack-local observer in `MountLeaseRenewer::renew`. struct MountRenewObservabilityStack @@ -1522,7 +1522,7 @@ uint64_t MountLeaseRenewer::start(Liveness liveness) /// This decoded authoritative observation is the exact point at which this incarnation learns /// that a foreign successor owns the slot. Terminal teardown intentionally performs no release /// I/O, so account the skipped farewell here, once, before the renewer enters its terminal state. - /// The renewal may be parked under `remount_mutex`; keep the increment trace-free. + /// The renewal may run under `remount_mutex`; keep the increment trace-free. ProfileEvents::incrementNoTrace(ProfileEvents::CASMountReleaseSkippedForeignOccupant); emitMountEvent( event_sink, CasEventType::MountConflict, srid, "foreign_writer", ¤t, diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h index 448ec650493c..991465e07ee4 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h @@ -83,7 +83,7 @@ struct MountRenewRequestEvent struct MountRenewOperationEnvironment { std::function boot_ms; - /// Facts the mount fence cannot see (park requested, pool no longer live, shutdown). FALSE ends + /// Facts the mount fence cannot see (a remount request, a pool no longer live, shutdown). FALSE ends /// the renewal exactly as a lost fence does; the engine does not need to know which refused. std::function live; /// Sampled ONCE before the write. A renewal refused before its first attempt is reported as @@ -551,8 +551,8 @@ bool isCreatorFenceTerminal(CasOperation & op, const Layout & layout, const Stri /// /// PLANES. A bounded renewal is admitted under the mount fence, because it writes under the authority /// the fence tracks. The worker's renewal (`UntilDefinitive`) runs on a plane with no lease budget: it -/// keeps trying after the lease expired, and a stop, a park or a terminal lifecycle reaches it through -/// its liveness. The claim and the farewell are admitted off the fence: a self-remount claims with the +/// keeps trying after the lease expired, and a stop, a remount request or a terminal lifecycle reaches +/// it through its liveness. The claim and the farewell are admitted off the fence: a self-remount claims with the /// fence already latched lost, so a claim gated on the fence could never reclaim, and a farewell /// refused because the fence has run down would leave the slot looking live until GC fences it out. /// Neither is unguarded: a claim's safety is its own conditional write, and a caller that has shutdown diff --git a/src/Disks/tests/gtest_cas_forget.cpp b/src/Disks/tests/gtest_cas_forget.cpp index db7ba970d24b..f8e9f566f03c 100644 --- a/src/Disks/tests/gtest_cas_forget.cpp +++ b/src/Disks/tests/gtest_cas_forget.cpp @@ -359,14 +359,14 @@ TEST(CASForget, ForgetCleanFarewellGatedOnDrain) } } -/// (b1) BOUNDED COMPLETION: FORGET racing an ACTIVE persistent remount worker joins it without deadlock. Here the -/// faulting backend keeps every attempt at `StayTransient` (it never reaches `armMountFence`), so this +/// (b1) BOUNDED COMPLETION: FORGET racing an ACTIVE reclaim on the lease thread joins it without deadlock. Here the +/// faulting backend keeps every attempt at `StayTransient` (it never reaches the arm), so this /// isolates the join/no-deadlock property; the fence re-arm path is covered by (b2) below. Uses a /// `std::future` timeout wait (never a sleep) — the timeout only fires on a genuine deadlock regression. TEST(CASForget, ForgetRacingActiveRemountThreadCompletesBounded) { auto backend = std::make_shared(); - /// `background_watermark = true` so the persistent recovery worker exists (mirrors + /// `background_watermark = true` so the lease thread exists (mirrors /// gtest_cas_pool.cpp's ShutdownGuardRefusesToArmRemount setup). auto store = DB::Cas::Pool::open(backend, DB::Cas::PoolConfig{.pool_prefix = "p", .server_root_id = "test", .background_watermark = true}); @@ -375,9 +375,9 @@ TEST(CASForget, ForgetRacingActiveRemountThreadCompletesBounded) /// trip the fence and latch a recovery generation — the worker now loops `tryRemountOnce` against the fault. backend->fail.store(true); store->tripMountLost(); - ASSERT_TRUE(store->scheduleRemountForTest()) << "the recovery worker must accept the request and run"; + ASSERT_TRUE(store->scheduleRemountForTest()) << "the lease thread must accept the request and run"; - /// FORGET from ANOTHER thread must join the active remount worker and finish in bounded time. + /// FORGET from ANOTHER thread must join the lease thread and finish in bounded time. std::promise done; auto fut = done.get_future(); std::thread forgetter([&] diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp index 0bf56f36b193..c8c362087277 100644 --- a/src/Disks/tests/gtest_cas_pool.cpp +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -3400,109 +3400,6 @@ TEST(CASPool, CachedSourceDecodeLetsAdoptionCommitAnAbsentBlobThatFsckReports) #define EXPECT_RUNTIME_STATE_REJECTION(statement) EXPECT_THROW(statement, DB::Exception) #endif -TEST(CASPoolRemount, DirectRenewCannotRaceWorkerStartOrRenewerReplacement) -{ - auto backend = std::make_shared(); - const Layout layout("runtime-direct"); - uint64_t wall_ms = 1000; - uint64_t boot_ms = 100; - const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); - CasEventSink sink; - RuntimeUnderTest runtime_holder( - backend, layout, MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), - .boot_ms_fn = [&] { return boot_ms; }}, - "test", sink, runtimeRenewBudget(), [] { return false; }); - CasMountRuntime & runtime = *runtime_holder; - runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); - - DB::Cas::tests::ManualBarrier barrier; - backend->barrier = &barrier; - backend->fault = RuntimeRenewBackend::Fault::BlockThenDelegate; - auto direct = std::async(std::launch::async, [&] { runtime.renewWatermarkOnce(); }); - barrier.waitUntilArrived(); - EXPECT_RUNTIME_STATE_REJECTION(runtime.startBackgroundWorkers(std::chrono::milliseconds(10))); - EXPECT_RUNTIME_STATE_REJECTION(runtime.installRenewer(uuid, 2, [&] { return wall_ms; })); - EXPECT_RUNTIME_STATE_REJECTION(runtime.renewerReset()); - barrier.release(); - EXPECT_NO_THROW(direct.get()); - runtime.finishTeardown(true); -} - -TEST(CASPoolRemount, DueWorkerAdmissionIsReservedBeforeParkRequest) -{ - auto backend = std::make_shared(); - const Layout layout("runtime-admission-park"); - uint64_t wall_ms = 1000; - uint64_t boot_ms = 100; - const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); - DB::Cas::tests::ManualBarrier admitted; - DB::Cas::tests::ManualBarrier remount; - CasEventSink sink; - RuntimeUnderTest runtime_holder( - backend, layout, - MountConfig{ - .mount_lease_ttl_ms = std::chrono::milliseconds(1000), - .background_watermark = true, - .boot_ms_fn = [&] { return boot_ms; }, - .renewal_admitted_hook_for_test = [&] { admitted.arriveAndWait(); }}, - "test", sink, runtimeRenewBudget(), [&] - { - remount.arriveAndWait(); - return false; - }); - CasMountRuntime & runtime = *runtime_holder; - runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); - runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); - admitted.waitUntilArrived(); - runtime.tripMountLost(); - runtime.scheduleRemount(); - EXPECT_EQ(runtime.renewalDriverStateForTest(), RenewalDriverState::ParkRequested); - admitted.release(); - remount.waitUntilArrived(); - EXPECT_EQ(runtime.renewalDriverStateForTest(), RenewalDriverState::Parked); - remount.release(); - runtime.stopBackgroundWorkers(); - runtime.finishTeardown(false); -} - -TEST(CASPoolRemount, DueWorkerAdmissionIsReservedBeforeStop) -{ - auto backend = std::make_shared(); - const Layout layout("runtime-admission-stop"); - uint64_t wall_ms = 1000; - uint64_t boot_ms = 100; - const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); - DB::Cas::tests::ManualBarrier admitted; - CasEventSink sink; - RuntimeUnderTest runtime_holder( - backend, layout, - MountConfig{ - .mount_lease_ttl_ms = std::chrono::milliseconds(1000), - .background_watermark = true, - .boot_ms_fn = [&] { return boot_ms; }, - .renewal_admitted_hook_for_test = [&] { admitted.arriveAndWait(); }}, - "test", sink, runtimeRenewBudget(), [] { return false; }); - CasMountRuntime & runtime = *runtime_holder; - runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); - runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); - admitted.waitUntilArrived(); - auto stop = std::async(std::launch::async, [&] { runtime.stopBackgroundWorkers(); }); - runtime.waitForRenewalDriverStateForTest(RenewalDriverState::Stopping); - admitted.release(); - EXPECT_NO_THROW(stop.get()); - EXPECT_EQ(runtime.renewalDriverStateForTest(), RenewalDriverState::Dormant); - runtime.finishTeardown(false); -} - TEST(CASPoolRemount, DirectRenewIsRefusedForBackgroundConfiguredRuntimeAfterStop) { auto backend = std::make_shared(); @@ -3529,10 +3426,10 @@ TEST(CASPoolRemount, DirectRenewIsRefusedForBackgroundConfiguredRuntimeAfterStop #undef EXPECT_RUNTIME_STATE_REJECTION -TEST(CASPoolRemount, RemountWaitsForRenewalParkedBeforeReplacement) +TEST(CASPoolRemount, ReclaimWaitsForTheRenewalToEnd) { auto backend = std::make_shared(); - const Layout layout("runtime-park"); + const Layout layout("runtime-reclaim-waits"); uint64_t wall_ms = 1000; uint64_t boot_ms = 10'000; const UInt128 uuid{1}; @@ -3561,17 +3458,16 @@ TEST(CASPoolRemount, RemountWaitsForRenewalParkedBeforeReplacement) renewal_barrier.waitUntilArrived(); runtime.tripMountLost(); runtime.scheduleRemount(); - runtime.waitForRenewalDriverStateForTest(RenewalDriverState::ParkRequested); - EXPECT_EQ(remount_calls.load(), 0u) << "replacement callback must wait until renewal has parked"; + EXPECT_EQ(remount_calls.load(), 0u) << "the reclaim must wait until the renewal in flight ends"; renewal_barrier.release(); remount_barrier.waitUntilArrived(); - EXPECT_EQ(runtime.renewalDriverStateForTest(), RenewalDriverState::Parked); + EXPECT_EQ(remount_calls.load(), 1u); remount_barrier.release(); runtime.stopBackgroundWorkers(); runtime.finishTeardown(false); } -TEST(CASPoolRemount, TeardownJoinsBothWorkersBeforeRelease) +TEST(CASPoolRemount, TeardownJoinsTheLeaseThreadBeforeRelease) { auto backend = std::make_shared(); const Layout layout("runtime-join"); @@ -3600,7 +3496,7 @@ TEST(CASPoolRemount, TeardownJoinsBothWorkersBeforeRelease) runtime.armMountFence(uuid, 1, anchor + 1000); runtime.startBackgroundWorkers(std::chrono::hours(1)); runtime.stopBackgroundWorkers(); - EXPECT_EQ(worker_exits.load(), 2u); + EXPECT_EQ(worker_exits.load(), 1u); runtime.finishTeardown(true); EXPECT_EQ(decodeMountLease(readObj(*backend, layout.mountKey("test"))->bytes).min_active_build_sequence, std::numeric_limits::max()); @@ -3743,7 +3639,7 @@ TEST(CASMountRuntime, MemoryLimitDoesNotEndTheLeaseThread) runtime.finishTeardown(false); } -TEST(CASPoolRemount, NaturalTerminalTransitionMakesBothPersistentWorkersSelfExit) +TEST(CASPoolRemount, NaturalTerminalTransitionMakesTheLeaseThreadExit) { for (PoolLifecycle terminal : {PoolLifecycle::IdentityLost, PoolLifecycle::VanishedReplaced}) { @@ -3793,124 +3689,115 @@ TEST(CASPoolRemount, NaturalTerminalTransitionMakesBothPersistentWorkersSelfExit runtime.scheduleRemount(); transitioned.waitUntilArrived(); transitioned.release(); - const bool both_exited_without_stop = exits.waitForAtLeast(2); + const bool exited_without_stop = exits.waitForAtLeast(1); runtime.stopBackgroundWorkers(); - EXPECT_TRUE(both_exited_without_stop); - EXPECT_EQ(exits.count(), 2u); + EXPECT_TRUE(exited_without_stop); + EXPECT_EQ(exits.count(), 1u); runtime.finishTeardown(false); } } -TEST(CASPoolRemount, ParkedRenewalCannotMissNaturalTerminalPublication) +/// A terminal publication that races a wait of the lease loop is serialized by `driver_mutex`: it +/// cannot land between the wait predicate's sample and the wait, so the thread exits without a stop. +/// Two waits: the cadence wait and the reclaim backoff. +TEST(CASPoolRemount, ALeaseWaitCannotMissNaturalTerminalPublication) { - for (PoolLifecycle terminal : {PoolLifecycle::IdentityLost, PoolLifecycle::VanishedReplaced}) + enum class Wait : uint8_t { Cadence, ReclaimBackoff }; + for (Wait wait : {Wait::Cadence, Wait::ReclaimBackoff}) { - auto backend = std::make_shared(); - const Layout layout(terminal == PoolLifecycle::IdentityLost - ? "runtime-parked-terminal-identity-lost" - : "runtime-parked-terminal-vanished"); - uint64_t wall_ms = 1000; - uint64_t boot_ms = 100; - const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); - WorkerExitLatch exits; - std::latch renewal_before_driver_lock{1}; - std::latch release_renewal{1}; - std::once_flag pause_renewal_once; - std::latch parked_predicate_sampled_false{1}; - std::latch release_parked_predicate{1}; - std::latch terminal_pre_lock_reached{1}; - std::latch terminal_post_lock_reached{1}; - std::once_flag release_once; - std::atomic renewal_holds_driver_mutex{false}; - std::atomic terminal_reached_post_lock_while_renewal_held{false}; - const auto release_parked = [&] - { - std::call_once(release_once, [&] { release_parked_predicate.count_down(); }); - }; - RuntimeWorkerFactory factory = [&](std::function worker_body) + for (PoolLifecycle terminal : {PoolLifecycle::IdentityLost, PoolLifecycle::VanishedReplaced}) { - return ThreadFromGlobalPool([&, body = std::move(worker_body)] + auto backend = std::make_shared(); + const Layout layout(fmt::format( + "runtime-wait-terminal-{}-{}", + wait == Wait::Cadence ? "cadence" : "backoff", + terminal == PoolLifecycle::IdentityLost ? "identity-lost" : "vanished")); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, + MountClaimResult::Claimed); + WorkerExitLatch exits; + std::once_flag pause_once; + std::latch predicate_sampled_false{1}; + std::latch release_predicate{1}; + std::once_flag release_once; + std::atomic waiter_holds_driver_mutex{false}; + std::atomic publication_entered_while_held{false}; + const auto release_waiter = [&] { - body(); - exits.recordExit(); - }); - }; - CasMountRuntime * runtime_ptr = nullptr; - CasEventSink sink; - RuntimeUnderTest runtime_holder( - backend, layout, - MountConfig{ - .mount_lease_ttl_ms = std::chrono::milliseconds(1000), - .background_watermark = true, - .boot_ms_fn = [&] { return boot_ms; }, - .worker_factory = factory, - .remount_parked_hook_for_test = [&] - { - release_renewal.count_down(); - }, - .renewal_before_driver_lock_hook_for_test = [&] + std::call_once(release_once, [&] { release_predicate.count_down(); }); + }; + RuntimeWorkerFactory factory = [&](std::function worker_body) + { + return ThreadFromGlobalPool([&, body = std::move(worker_body)] { - std::call_once(pause_renewal_once, [&] + body(); + exits.recordExit(); + }); + }; + CasEventSink sink; + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{ + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }, + .worker_factory = factory, + .lease_wait_predicate_false_hook_for_test = [&] { - renewal_before_driver_lock.count_down(); - release_renewal.wait(); - }); - }, - .renewal_parked_predicate_false_hook_for_test = [&] - { - renewal_holds_driver_mutex.store(true, std::memory_order_release); - parked_predicate_sampled_false.count_down(); - release_parked_predicate.wait(); - renewal_holds_driver_mutex.store(false, std::memory_order_release); - }, - .terminal_publication_waiting_for_driver_lock_hook_for_test = [&] - { - terminal_pre_lock_reached.count_down(); - }, - .terminal_publication_driver_lock_contended_hook_for_test = [&] - { - release_parked(); - }, - .terminal_publication_driver_lock_acquired_hook_for_test = [&] - { - if (renewal_holds_driver_mutex.load(std::memory_order_acquire)) - terminal_reached_post_lock_while_renewal_held.store(true, std::memory_order_release); - release_parked(); - terminal_post_lock_reached.count_down(); - }}, - "test", sink, runtimeRenewBudget(), [&] + std::call_once(pause_once, [&] + { + waiter_holds_driver_mutex.store(true, std::memory_order_release); + predicate_sampled_false.count_down(); + release_predicate.wait(); + waiter_holds_driver_mutex.store(false, std::memory_order_release); + }); + }, + .terminal_publication_driver_lock_contended_hook_for_test = [&] + { + release_waiter(); + }, + .terminal_publication_driver_lock_acquired_hook_for_test = [&] + { + if (waiter_holds_driver_mutex.load(std::memory_order_acquire)) + publication_entered_while_held.store(true, std::memory_order_release); + release_waiter(); + }}, + "test", sink, runtimeRenewBudget(), [] { return false; }); + /// Declared after the runtime so it runs first: a failed expectation must not leave the + /// lease thread inside the hook while the runtime's destructor joins it. + SCOPE_EXIT({ release_waiter(); }); + CasMountRuntime & runtime = *runtime_holder; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + runtime.armMountFence(uuid, 1, anchor + 1000); + if (wait == Wait::ReclaimBackoff) { - parked_predicate_sampled_false.wait(); - if (terminal == PoolLifecycle::IdentityLost) - runtime_ptr->enterIdentityLost(); - else - runtime_ptr->enterVanished(PoolLifecycle::VanishedReplaced, "injected parked-wait replacement"); - /// Before the fix, terminal publication does not wait for `driver_mutex`, so it reaches - /// this release only after its notification has raced ahead of the renewal worker's wait. - /// After the fix, the pre-lock hook above releases the waiter before publication blocks. - release_parked(); - return false; - }); - CasMountRuntime & runtime = *runtime_holder; - runtime_ptr = &runtime; - runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); - runtime.startBackgroundWorkers(std::chrono::hours(1)); - renewal_before_driver_lock.wait(); - runtime.tripMountLost(); - runtime.scheduleRemount(); - terminal_pre_lock_reached.wait(); - terminal_post_lock_reached.wait(); - const bool violated_serialization - = terminal_reached_post_lock_while_renewal_held.load(std::memory_order_acquire); - const bool both_exited_without_stop = violated_serialization ? false : exits.waitForAtLeast(2); - runtime.stopBackgroundWorkers(); - EXPECT_FALSE(violated_serialization); - EXPECT_TRUE(both_exited_without_stop); - EXPECT_EQ(exits.count(), 2u); - runtime.finishTeardown(false); + /// A request before the start makes the first pass a reclaim, which fails and backs off. + runtime.tripMountLost(); + runtime.scheduleRemount(); + } + runtime.startBackgroundWorkers(std::chrono::hours(1)); + + predicate_sampled_false.wait(); + if (terminal == PoolLifecycle::IdentityLost) + { + runtime.tripMountLost(); + runtime.enterIdentityLost(); + } + else + { + runtime.enterVanished(PoolLifecycle::VanishedReplaced, "injected replacement during a lease wait"); + } + const bool exited_without_stop = exits.waitForAtLeast(1); + runtime.stopBackgroundWorkers(); + EXPECT_FALSE(publication_entered_while_held.load()) + << "the publication must not take driver_mutex while the waiter holds it"; + EXPECT_TRUE(exited_without_stop); + EXPECT_EQ(exits.count(), 1u); + runtime.finishTeardown(false); + } } } @@ -3967,46 +3854,40 @@ TEST(CASPoolRemount, VanishedReasonPreparationFailureLeavesTerminalTransitionRet runtime.enterVanished(PoolLifecycle::VanishedForgotten, "must-remain-ignored"); EXPECT_EQ(runtime.lifecycle(), PoolLifecycle::VanishedReplaced); EXPECT_EQ(runtime.vanishedReason(), "retry-completed"); - const bool both_exited_without_stop = exits.waitForAtLeast(2); + const bool exited_without_stop = exits.waitForAtLeast(1); runtime.stopBackgroundWorkers(); - EXPECT_TRUE(both_exited_without_stop); - EXPECT_EQ(exits.count(), 2u); + EXPECT_TRUE(exited_without_stop); + EXPECT_EQ(exits.count(), 1u); EXPECT_EQ(preparation_calls.load(), 2u); runtime.finishTeardown(false); } TEST(CASPoolRemount, WorkerConstructionRollbackFailsOpenClosed) { - for (uint64_t throw_on : {1u, 2u}) + auto backend = std::make_shared(); + const Layout layout("runtime-worker-failure"); + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + RuntimeWorkerFactory factory = [](std::function) -> ThreadFromGlobalPool { - auto backend = std::make_shared(); - const Layout layout("runtime-worker-failure-" + std::to_string(throw_on)); - uint64_t wall_ms = 1000; - uint64_t boot_ms = 100; - uint64_t factory_calls = 0; - RuntimeWorkerFactory factory = [&](std::function fn) - { - if (++factory_calls == throw_on) - throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "injected runtime worker construction failure"); - return ThreadFromGlobalPool(std::move(fn)); - }; - const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); - CasEventSink sink; - RuntimeUnderTest runtime_holder( - backend, layout, - MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, - .boot_ms_fn = [&] { return boot_ms; }, .worker_factory = factory}, - "test", sink, runtimeRenewBudget(), [] { return false; }); - CasMountRuntime & runtime = *runtime_holder; - runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); - EXPECT_THROW(runtime.startBackgroundWorkers(std::chrono::milliseconds(10)), DB::Exception); - EXPECT_FALSE(runtime.mayMutate()); - EXPECT_FALSE(runtime.workersRunningForTest()); - runtime.finishTeardown(false); - } + throw DB::Exception(DB::ErrorCodes::NETWORK_ERROR, "injected runtime worker construction failure"); + }; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + CasEventSink sink; + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }, .worker_factory = factory}, + "test", sink, runtimeRenewBudget(), [] { return false; }); + CasMountRuntime & runtime = *runtime_holder; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + runtime.armMountFence(uuid, 1, anchor + 1000); + EXPECT_THROW(runtime.startBackgroundWorkers(std::chrono::milliseconds(10)), DB::Exception); + EXPECT_FALSE(runtime.mayMutate()); + EXPECT_FALSE(runtime.workersRunningForTest()); + runtime.finishTeardown(false); } TEST(CASPoolRemount, ExternalLossDuringRenewalUsesOneRecoveryGeneration) @@ -4032,6 +3913,7 @@ TEST(CASPoolRemount, ExternalLossDuringRenewalUsesOneRecoveryGeneration) "test", sink, runtimeRenewBudget(), [&] { ++remount_calls; + runtime_ptr->beginReclaim(); ++fresh_epochs; fenceOutMount(*backend, layout.mountKey("test")); const MountClaimResult fresh = claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 2, wall_ms, 1000); @@ -4042,8 +3924,7 @@ TEST(CASPoolRemount, ExternalLossDuringRenewalUsesOneRecoveryGeneration) const uint64_t fresh_anchor = runtime_ptr->startRenewer(); runtime_ptr->setProcessEpoch(2, std::memory_order_release); runtime_ptr->setLiveWriterEpoch(2); - runtime_ptr->armMountFence(uuid, 2, fresh_anchor + 1000); - runtime_ptr->noteRemounted(); + EXPECT_TRUE(runtime_ptr->armIfAdmissible(fresh_anchor + 1000)); remount_barrier.arriveAndWait(); return true; }); @@ -4070,55 +3951,6 @@ TEST(CASPoolRemount, ExternalLossDuringRenewalUsesOneRecoveryGeneration) runtime.finishTeardown(false); } -TEST(CASPoolRemount, TerminalDepositionDoesNotTouchRenewerAfterReplacement) -{ - auto backend = std::make_shared(); - const Layout layout("runtime-terminal-replacement"); - uint64_t wall_ms = 1000; - uint64_t boot_ms = 100; - const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); - DB::Cas::tests::ManualBarrier terminal_deposited; - DB::Cas::tests::ManualBarrier remount; - std::atomic replaced{false}; - CasMountRuntime * runtime_ptr = nullptr; - CasEventSink sink; - RuntimeUnderTest runtime_holder( - backend, layout, - MountConfig{ - .mount_lease_ttl_ms = std::chrono::milliseconds(1000), - .background_watermark = true, - .boot_ms_fn = [&] { return boot_ms; }, - .renewal_terminal_deposited_hook_for_test = [&] - { - runtime_ptr->renewerReset(); - runtime_ptr->installRenewer(uuid, 2, [&] { return wall_ms; }); - runtime_ptr->renewerReset(); - replaced.store(true, std::memory_order_release); - terminal_deposited.arriveAndWait(); - }}, - "test", sink, runtimeRenewBudget(), [&] - { - remount.arriveAndWait(); - return false; - }); - CasMountRuntime & runtime = *runtime_holder; - runtime_ptr = &runtime; - runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); - /// A definitive answer ends the worker's renewal; a transient fault would only be retried. - fenceOutMount(*backend, layout.mountKey("test")); - runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); - terminal_deposited.waitUntilArrived(); - EXPECT_TRUE(replaced.load(std::memory_order_acquire)); - terminal_deposited.release(); - remount.waitUntilArrived(); - remount.release(); - runtime.stopBackgroundWorkers(); - runtime.finishTeardown(false); -} - TEST(CASPoolRemount, ConcurrentRemountRequestIsProcessedAfterActiveGeneration) { auto backend = std::make_shared(); @@ -4183,6 +4015,7 @@ TEST(CASPoolRemount, ImmediatePostRemountRenewalFailureIsNotDropped) const uint64_t call = ++calls; if (call == 1) { + runtime_ptr->beginReclaim(); fenceOutMount(*backend, layout.mountKey("test")); const MountClaimResult fresh = claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 2, wall_ms, 10'000); EXPECT_EQ(fresh.kind, MountClaimResult::Claimed); @@ -4190,8 +4023,7 @@ TEST(CASPoolRemount, ImmediatePostRemountRenewalFailureIsNotDropped) return false; runtime_ptr->installRenewer(uuid, 2, [&] { return wall_ms; }); const uint64_t fresh_anchor = runtime_ptr->startRenewer(); - runtime_ptr->armMountFence(uuid, 2, fresh_anchor + 10'000); - runtime_ptr->noteRemounted(); + EXPECT_TRUE(runtime_ptr->armIfAdmissible(fresh_anchor + 10'000)); boot_ms = 2'000; /// A definitive answer for the fresh incarnation's first worker renewal; a transient /// fault would only be retried. @@ -4634,83 +4466,8 @@ TEST(CASPool, DeterministicWorkerFailureFencesWithoutWaitingForCadence) runtime.finishTeardown(false); } -/// A remount request ends a worker renewal that is retrying past its lease, inside the wait or the -/// request it is in: no request and no wait starts after it. -TEST(CASMountRuntime, ParkEndsAnUnboundedRenewal) -{ - enum class ParkDuring : uint8_t { Wait, Request }; - const auto run = [](ParkDuring park_during) - { - const Layout layout(park_during == ParkDuring::Wait ? "unbounded-park-in-wait" : "unbounded-park-in-request"); - const UInt128 uuid{1}; - uint64_t wall_ms = 1000; - std::atomic boot_ms{100}; - std::atomic past_the_lease{std::numeric_limits::max()}; - std::atomic held{false}; - std::atomic waits{0}; - std::atomic writes_at_remount{0}; - std::atomic waits_at_remount{0}; - DB::Cas::tests::ManualBarrier holding; - DB::Cas::tests::ManualBarrier remount_entered; - const auto hold_once_past_the_lease = [&] - { - if (boot_ms.load() >= past_the_lease.load() && !held.exchange(true)) - holding.arriveAndWait(); - }; - /// Declared after the locals its hooks capture. - auto backend = std::make_shared(); - ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, - MountClaimResult::Claimed); - CasEventSink sink; - RuntimeUnderTest runtime_holder( - backend, layout, - MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, - .boot_ms_fn = [&] { return boot_ms.load(); }}, - "test", sink, runtimeRenewBudget(), [&] - { - writes_at_remount = backend->outage_writes.load(); - waits_at_remount = waits.load(); - remount_entered.arriveAndWait(); - return false; - }); - CasMountRuntime & runtime = *runtime_holder; - runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); - /// Five lease lengths on: a renewal bounded by its lease ended long before. - past_the_lease = anchor + 5'000; - runtime_holder.setRetrySleepForTest([&](uint64_t ms) - { - ++waits; - boot_ms += ms; - if (park_during == ParkDuring::Wait) - hold_once_past_the_lease(); - }); - if (park_during == ParkDuring::Request) - backend->on_outage_write = hold_once_past_the_lease; - backend->outage = [] { return true; }; - runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); - - holding.waitUntilArrived(); - const uint64_t writes_at_park = backend->outage_writes.load(); - const uint64_t waits_at_park = waits.load(); - runtime.scheduleRemount(); - EXPECT_EQ(runtime.renewalDriverStateForTest(), RenewalDriverState::ParkRequested); - holding.release(); - remount_entered.waitUntilArrived(); - EXPECT_EQ(runtime.renewalDriverStateForTest(), RenewalDriverState::Parked); - EXPECT_EQ(writes_at_remount.load(), writes_at_park) << "no request starts after the park"; - EXPECT_EQ(waits_at_remount.load(), waits_at_park) << "no wait starts after the park"; - remount_entered.release(); - runtime.stopBackgroundWorkers(); - runtime.finishTeardown(false); - }; - run(ParkDuring::Wait); - run(ParkDuring::Request); -} - /// FORGET while the worker's renewal retries past its lease: the intent then the trip, in the order -/// `Pool::forgetDisk` uses, end the renewal inside the wait it is in, both workers exit, and no remount +/// `Pool::forgetDisk` uses, end the renewal inside the wait it is in, the lease thread exits, and no remount /// generation is raised. TEST(CASMountRuntime, ForgetEndsAnUnboundedRenewal) { @@ -4763,8 +4520,8 @@ TEST(CASMountRuntime, ForgetEndsAnUnboundedRenewal) runtime.tripMountLost(); holding.release(); - const bool both_exited = exits.waitForAtLeast(2); - EXPECT_TRUE(both_exited) << "the worker loops must exit on the published intent"; + const bool exited = exits.waitForAtLeast(1); + EXPECT_TRUE(exited) << "the lease loop must exit on the published intent"; EXPECT_EQ(backend->outage_writes.load(), writes_at_forget) << "no request starts after FORGET"; EXPECT_EQ(waits.load(), waits_at_forget) << "no wait starts after FORGET"; EXPECT_EQ(runtime.remountRequestedGenerationForTest(), generation_at_forget); @@ -5193,6 +4950,251 @@ TEST(CASMountRuntime, StopDuringAReclaimJoinsTheThread) << "no farewell for a slot this runtime did not claim back"; } +/// An interference report while the renewal retries past its lease ends it inside the wait or the +/// request it is in: no request and no wait starts after the report. The thread that renewed runs the +/// reclaim and then renews under the new epoch. +TEST(CASMountRuntime, ARemountRequestEndsTheRenewalAndTheSameThreadReclaims) +{ + enum class RequestDuring : uint8_t { Wait, Request }; + const auto run = [](RequestDuring request_during) + { + const Layout layout(request_during == RequestDuring::Wait ? "same-thread-reclaim-in-wait" : "same-thread-reclaim-in-request"); + const UInt128 uuid{1}; + uint64_t wall_ms = 1000; + std::atomic boot_ms{100}; + std::atomic past_the_lease{std::numeric_limits::max()}; + std::atomic held{false}; + std::atomic outage_on{true}; + std::atomic waits{0}; + std::atomic writes_at_reclaim{0}; + std::atomic waits_at_reclaim{0}; + std::atomic worker_threads{0}; + std::atomic reclaimed{false}; + std::atomic renewed_after_reclaim{false}; + std::mutex thread_ids_mutex; + std::thread::id renewing_thread; + std::thread::id reclaiming_thread; + std::thread::id renewing_after_reclaim_thread; + DB::Cas::tests::ManualBarrier holding; + DB::Cas::tests::ManualBarrier renewed; + CasMountRuntime * runtime_ptr = nullptr; + const auto record_thread = [&](std::thread::id & slot) + { + std::lock_guard lock(thread_ids_mutex); + slot = std::this_thread::get_id(); + }; + const auto hold_once_past_the_lease = [&] + { + if (boot_ms.load() >= past_the_lease.load() && !held.exchange(true)) + { + record_thread(renewing_thread); + holding.arriveAndWait(); + } + }; + RuntimeWorkerFactory factory = [&](std::function worker_body) + { + ++worker_threads; + return ThreadFromGlobalPool(std::move(worker_body)); + }; + /// Declared after the locals its hooks capture. + auto backend = std::make_shared(); + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, + MountClaimResult::Claimed); + CasEventSink sink; + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms.load(); }, .worker_factory = factory}, + "test", sink, runtimeRenewBudget(), [&] + { + CasMountRuntime & reclaiming = *runtime_ptr; + writes_at_reclaim = backend->outage_writes.load(); + waits_at_reclaim = waits.load(); + record_thread(reclaiming_thread); + outage_on = false; + reclaiming.beginReclaim(); + fenceOutMount(*backend, layout.mountKey("test")); + const MountClaimResult fresh + = claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 2, wall_ms, 1000); + EXPECT_EQ(fresh.kind, MountClaimResult::Claimed); + if (fresh.kind != MountClaimResult::Claimed) + return false; + reclaiming.installRenewer(uuid, 2, [&] { return wall_ms; }); + const uint64_t fresh_anchor = reclaiming.startRenewer(); + reclaiming.setLiveWriterEpoch(2); + EXPECT_TRUE(reclaiming.armIfAdmissible(fresh_anchor + 1000)); + reclaimed = true; + return true; + }); + SCOPE_EXIT({ + holding.release(); + renewed.release(); + }); + CasMountRuntime & runtime = *runtime_holder; + runtime_ptr = &runtime; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + runtime.armMountFence(uuid, 1, anchor + 1000); + /// Five lease lengths on: a renewal bounded by its lease ended long before. + past_the_lease = anchor + 5'000; + runtime_holder.setRetrySleepForTest([&](uint64_t ms) + { + ++waits; + boot_ms += ms; + if (request_during == RequestDuring::Wait) + hold_once_past_the_lease(); + }); + if (request_during == RequestDuring::Request) + backend->on_outage_write = hold_once_past_the_lease; + backend->outage = [&] { return outage_on.load(); }; + backend->after_commit = [&] + { + if (reclaimed.load() && !renewed_after_reclaim.exchange(true)) + { + record_thread(renewing_after_reclaim_thread); + renewed.arriveAndWait(); + } + }; + runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + + holding.waitUntilArrived(); + const uint64_t writes_at_request = backend->outage_writes.load(); + const uint64_t waits_at_request = waits.load(); + /// An interference report while the renewal retries past its lease. + runtime.tripMountLost(); + runtime.scheduleRemount(); + holding.release(); + renewed.waitUntilArrived(); + + EXPECT_EQ(worker_threads.load(), 1u) << "one thread renews and reclaims"; + { + std::lock_guard lock(thread_ids_mutex); + EXPECT_EQ(reclaiming_thread, renewing_thread) << "the reclaim runs on the thread that renewed"; + EXPECT_EQ(renewing_after_reclaim_thread, renewing_thread) << "the same thread renews after the reclaim"; + } + EXPECT_EQ(writes_at_reclaim.load(), writes_at_request) << "no request starts after the remount request"; + EXPECT_EQ(waits_at_reclaim.load(), waits_at_request) << "no wait starts after the remount request"; + EXPECT_EQ(decodeMountLease(readObj(*backend, layout.mountKey("test"))->bytes).writer_epoch, 2u) + << "the renewal after the reclaim runs under the new epoch"; + EXPECT_EQ(runtime.remountRequestedGenerationForTest(), 1u); + renewed.release(); + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); + }; + run(RequestDuring::Wait); + run(RequestDuring::Request); +} + +/// A renewal whose write committed before a remount request is consumed before the reclaim runs, so +/// its deadline cannot overwrite the one the reclaim arms. The renewal's last liveness check comes +/// after its write committed: the request is raised there, so the renewal still returns `Committed`. +TEST(CASMountRuntime, ACommitConsumedAfterARemountRequestCannotOverwriteTheReclaimedDeadline) +{ + const Layout layout("commit-consumed-after-request"); + const UInt128 uuid{1}; + uint64_t wall_ms = 1000; + std::atomic boot_ms{100}; + std::atomic worker_threads{0}; + std::atomic first_worker_thread{}; + std::atomic committed{false}; + std::atomic requested{false}; + std::atomic held{false}; + std::atomic hold_timed_out{false}; + std::atomic armed_by_reclaim{false}; + std::atomic admissions{0}; + std::promise reclaim_armed; + std::shared_future reclaim_armed_future = reclaim_armed.get_future().share(); + DB::Cas::tests::ManualBarrier second_admission; + CasMountRuntime * runtime_ptr = nullptr; + RuntimeWorkerFactory factory = [&](std::function worker_body) + { + const bool first = worker_threads.fetch_add(1) == 0; + return ThreadFromGlobalPool([&, first, body = std::move(worker_body)] + { + if (first) + first_worker_thread = std::this_thread::get_id(); + body(); + }); + }; + /// With a second thread the reclaim can run before the renewing thread consumes its result. Hold + /// that consumption at its first clock read until the reclaim armed. With one thread this never + /// waits: the consumption comes before the reclaim on the same thread. + const auto hold_consumption_until_armed = [&] + { + if (worker_threads.load() == 2 && requested.load() + && std::this_thread::get_id() == first_worker_thread.load() && !held.exchange(true)) + { + if (reclaim_armed_future.wait_for(std::chrono::seconds(20)) != std::future_status::ready) + hold_timed_out = true; + } + }; + /// Declared after the locals its hooks capture. + auto backend = std::make_shared(); + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, + MountClaimResult::Claimed); + CasEventSink sink; + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{ + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .background_watermark = true, + .boot_ms_fn = [&] + { + hold_consumption_until_armed(); + return boot_ms.load(); + }, + .worker_factory = factory, + .renewal_admitted_hook_for_test = [&] + { + if (++admissions == 2) + second_admission.arriveAndWait(); + }, + .renewal_live_for_test = [&] + { + if (committed.load() && !requested.exchange(true)) + runtime_ptr->scheduleRemount(); + return true; + }}, + "test", sink, runtimeRenewBudget(), [&] + { + CasMountRuntime & reclaiming = *runtime_ptr; + reclaiming.beginReclaim(); + fenceOutMount(*backend, layout.mountKey("test")); + const MountClaimResult fresh + = claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 2, wall_ms, 1000); + EXPECT_EQ(fresh.kind, MountClaimResult::Claimed); + if (fresh.kind != MountClaimResult::Claimed) + return false; + reclaiming.installRenewer(uuid, 2, [&] { return wall_ms; }); + boot_ms = 500; + const uint64_t fresh_anchor = reclaiming.startRenewer(); + reclaiming.setLiveWriterEpoch(2); + armed_by_reclaim = reclaiming.armIfAdmissible(fresh_anchor + 1000); + reclaim_armed.set_value(); + return true; + }); + SCOPE_EXIT({ second_admission.release(); }); + CasMountRuntime & runtime = *runtime_holder; + runtime_ptr = &runtime; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + runtime.armMountFence(uuid, 1, anchor + 1000); + /// Set after the setup's own writes: only the loop's renewal may raise the request. + backend->after_commit = [&] { committed = true; }; + runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + + second_admission.waitUntilArrived(); + EXPECT_FALSE(hold_timed_out.load()); + ASSERT_TRUE(armed_by_reclaim.load()); + /// Between the first renewal's deadline (100 + 1000) and the reclaim's (500 + 1000). + boot_ms = 1200; + EXPECT_TRUE(runtime.mayMutate()) + << "the deadline is the reclaim's, not the one of the renewal consumed after the request"; + second_admission.release(); + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); +} + TEST(CASPool, RenewWatermarkOnceRefreshesFenceAndDepositsOneFailure) { auto backend = std::make_shared(); From ecd7d59323d1c0964ef48e498a0b5e2ceb89b3bf Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 05:39:20 +0200 Subject: [PATCH 04/37] Let only the CAS lease thread drive the renewer while it runs installRenewer, renewerReset, startRenewer and every renewal check ownership under driver_mutex: while a lease thread runs, a call from any other thread is a LOGICAL_ERROR. Before the thread starts and after it is joined, the caller owns the renewer. Co-Authored-By: Claude Opus 5.5 --- .../ContentAddressed/Pool/CasMountRuntime.cpp | 19 ++++++ .../ContentAddressed/Pool/CasMountRuntime.h | 5 ++ src/Disks/tests/gtest_cas_pool.cpp | 68 +++++++++++++++++++ 3 files changed, 92 insertions(+) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index 47e6e990ff42..20be9cde8b07 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -303,6 +303,14 @@ bool CasMountRuntime::canArm(uint64_t /*deadline_boot_ms*/) const && remount_requested_generation <= remount_handled_generation; } +void CasMountRuntime::checkRenewerOwner() const +{ + if (workers_started && lease_thread_id != std::this_thread::get_id()) + throw Exception( + ErrorCodes::LOGICAL_ERROR, + "CAS mount runtime: only the lease thread may drive the renewer while it runs"); +} + uint64_t CasMountRuntime::minActive() { std::lock_guard lk(builds_mutex); @@ -396,6 +404,7 @@ void CasMountRuntime::installRenewer( /// The previous renewer is destroyed after the unlock. std::lock_guard lock(driver_mutex); + checkRenewerOwner(); std::swap(mount_renewer, replaced); } @@ -404,6 +413,7 @@ uint64_t CasMountRuntime::startRenewer() MountLeaseRenewer * renewer = nullptr; { std::lock_guard lock(driver_mutex); + checkRenewerOwner(); if (!mount_renewer) throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: startRenewer without a renewer"); if (mount_renewer->state() != MountLeaseRenewerState::New) @@ -572,6 +582,7 @@ MountRenewResult CasMountRuntime::renewRenewerOnce(RenewCaller caller) throw Exception( ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: direct renewal is disabled when background ownership is configured"); + checkRenewerOwner(); if (!mount_renewer) throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: renewal without a renewer"); if (mount_renewer->state() != MountLeaseRenewerState::Active) @@ -604,6 +615,7 @@ void CasMountRuntime::renewerReset() { std::unique_ptr released; std::lock_guard lock(driver_mutex); + checkRenewerOwner(); std::swap(mount_renewer, released); } @@ -624,6 +636,7 @@ void CasMountRuntime::startBackgroundWorkers(std::chrono::milliseconds period) throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: the lease thread requires an Active renewer"); workers_started = true; workers_stop_requested = false; + lease_thread_id = {}; renewal_period = period; } @@ -639,6 +652,7 @@ void CasMountRuntime::startBackgroundWorkers(std::chrono::milliseconds period) std::lock_guard lock(driver_mutex); workers_stop_requested = true; workers_started = false; + lease_thread_id = {}; driver_cv.notify_all(); } tripFenceWithoutOperationalLoss(); @@ -655,6 +669,10 @@ void CasMountRuntime::renewalLoop() /// The tracker still counts this thread's allocations but never throws on it: a renewal failed by a /// memory limit costs the mount, and an exception outside the request ends this thread. LockMemoryExceptionInThread memory_exception_lock(VariableContext::Global); + { + std::lock_guard lock(driver_mutex); + lease_thread_id = std::this_thread::get_id(); + } /// The wake condition of every wait in this loop; the reclaim backoff ignores a request, which is /// the one it is retrying. @@ -775,6 +793,7 @@ void CasMountRuntime::stopBackgroundWorkers() { std::lock_guard lock(driver_mutex); workers_started = false; + lease_thread_id = {}; driver_cv.notify_all(); } } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index 6d51e96af28f..fdef370e68e1 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -18,6 +18,7 @@ #include #include #include +#include namespace DB::Cas { @@ -454,6 +455,8 @@ class CasMountRuntime /// terminal. Requires `driver_mutex`, the mutex that a stop, a request and a terminal publication /// take, so the check and the arm are one step. bool canArm(uint64_t deadline_boot_ms) const; + /// Throws `LOGICAL_ERROR` when a lease thread runs and the caller is not it. Requires `driver_mutex`. + void checkRenewerOwner() const; std::unique_lock lockTerminalPublication(); /// ---- injected environment (no `Pool` back-reference); initialized first, in this order ---- @@ -503,6 +506,8 @@ class CasMountRuntime mutable std::condition_variable driver_cv; bool workers_started = false; bool workers_stop_requested = false; + /// The lease thread, from its first statement until `stopBackgroundWorkers` joins it. + std::thread::id lease_thread_id; std::chrono::milliseconds renewal_period{0}; uint64_t remount_requested_generation = 0; uint64_t remount_handled_generation = 0; diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp index c8c362087277..1753b1abd9db 100644 --- a/src/Disks/tests/gtest_cas_pool.cpp +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -14,6 +14,7 @@ #include #include #include +#include #include #include #include @@ -21,6 +22,7 @@ #include #include #include +#include #include #include #include @@ -42,6 +44,7 @@ extern const int FILE_DOESNT_EXIST; extern const int UNKNOWN_EXCEPTION; extern const int NETWORK_ERROR; extern const int MEMORY_LIMIT_EXCEEDED; +extern const int LOGICAL_ERROR; } namespace ProfileEvents @@ -5195,6 +5198,71 @@ TEST(CASMountRuntime, ACommitConsumedAfterARemountRequestCannotOverwriteTheRecla runtime.finishTeardown(false); } +/// While the lease thread runs, no other thread may replace, reset or start the renewer. Constructing +/// a `LOGICAL_ERROR` exception aborts under a debug or sanitizer build, so there the same contract is +/// a death expectation. +#ifndef DEBUG_OR_SANITIZER_BUILD +TEST(CASMountRuntime, OnlyTheLeaseThreadReplacesTheRenewerWhileItRuns) +#else +TEST(CASMountRuntimeDeathTest, OnlyTheLeaseThreadReplacesTheRenewerWhileItRuns) +#endif +{ + auto backend = std::make_shared(); + const Layout layout("runtime-renewer-owner"); + uint64_t wall_ms = 1000; + const uint64_t boot_ms = 100; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, + MountClaimResult::Claimed); + DB::Cas::tests::ManualBarrier admitted; + CasEventSink sink; + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }, + .renewal_admitted_hook_for_test = [&] { admitted.arriveAndWait(); }}, + "test", sink, runtimeRenewBudget(), [] { return false; }); + SCOPE_EXIT({ admitted.release(); }); + CasMountRuntime & runtime = *runtime_holder; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + /// The lease thread is held between its decision to renew and the renewal, with no lock held. + admitted.waitUntilArrived(); + +#ifndef DEBUG_OR_SANITIZER_BUILD + const auto expect_owner_refusal = [](const char * what, const std::function & call) + { + try + { + call(); + ADD_FAILURE() << what << " from a test thread must be refused while the lease thread runs"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::LOGICAL_ERROR) << what; + EXPECT_NE(e.message().find("only the lease thread"), String::npos) << what << ": " << e.message(); + } + }; + expect_owner_refusal("installRenewer", [&] { runtime.installRenewer(uuid, 2, [&] { return wall_ms; }); }); + expect_owner_refusal("renewerReset", [&] { runtime.renewerReset(); }); + expect_owner_refusal("startRenewer", [&] { (void)runtime.startRenewer(); }); +#else + /// The child exits through `std::_Exit` if the call does not abort: running exit handlers in a + /// forked child can block on a thread pool mutex a vanished thread held. + EXPECT_DEATH({ runtime.installRenewer(uuid, 2, [&] { return wall_ms; }); std::_Exit(0); }, "only the lease thread"); + EXPECT_DEATH({ runtime.renewerReset(); std::_Exit(0); }, "only the lease thread"); + EXPECT_DEATH({ (void)runtime.startRenewer(); std::_Exit(0); }, "only the lease thread"); +#endif + + admitted.release(); + runtime.stopBackgroundWorkers(); + /// With the thread joined the caller drives the renewer again. + EXPECT_NO_THROW(runtime.renewerReset()); + runtime.finishTeardown(false); +} + TEST(CASPool, RenewWatermarkOnceRefreshesFenceAndDepositsOneFailure) { auto backend = std::make_shared(); From fd39d11a85ae56804deb236f6087932958d0c641 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 05:39:43 +0200 Subject: [PATCH 05/37] Describe the single CAS lease thread in the mounts and leases doc Replace the text about separate renewal and remount workers and parking with the lease thread, the reclaim's latch and arming rule, and the generations. Co-Authored-By: Claude Opus 5.5 --- .../cas/architecture/mounts-and-leases.md | 39 +++++++++++-------- 1 file changed, 22 insertions(+), 17 deletions(-) diff --git a/docs/en/antalya/cas/architecture/mounts-and-leases.md b/docs/en/antalya/cas/architecture/mounts-and-leases.md index 1f2878286615..1125f258b8c2 100644 --- a/docs/en/antalya/cas/architecture/mounts-and-leases.md +++ b/docs/en/antalya/cas/architecture/mounts-and-leases.md @@ -251,29 +251,34 @@ stateDiagram-v2 VanishedForgotten --> [*] ``` -`IdentityLost`, `VanishedReplaced` and `VanishedForgotten` are terminal and absorbing: the remount -and GC threads self-exit, and there is deliberately no auto-revive — an identity disappearing +`IdentityLost`, `VanishedReplaced` and `VanishedForgotten` are terminal and absorbing: the lease +thread and the GC threads exit, and there is deliberately no auto-revive — an identity disappearing under a live mount is an operator-level event. ## Mount, unmount, crash {#mount-lifecycle} **Writable open** runs in a strict order: bootstrap-residual proof, capability probe under a random per-mount prefix, pool-meta create-or-validate, `validateServerRootId`, owner claim, -`allocateWriterEpoch`, mount claim and synchronous renewer start, arm the fence, then create and -release the runtime-owned renewal and remount workers before the writable pool becomes externally -visible. If the claim consumed the TTL, one fresh synchronous renewal re-anchors the deadline -before the fence is armed. -Failure to construct either worker joins the partial pair, closes the fence, and fails the writable -open. No incident path constructs a thread. - -The renewal and remount workers are separate and long-lived under one stable `CasMountRuntime`. -`scheduleRemount` increments a requested-generation latch and wakes the persistent remount worker, -including while an older generation is active. Before renewer replacement, remount requests -`ParkRequested` and waits for the renewal driver to report `Parked`, which proves that no renewer call -is in flight. A successful remount handles only its snapshotted generation; a newer request is -processed before renewal resumes. - -**Clean unmount:** request stop and join both persistent workers, drain the ref lanes, and only if +`allocateWriterEpoch`, mount claim and synchronous renewer start, arm the fence, then start the +runtime-owned lease thread before the writable pool becomes externally visible. If the claim consumed +the TTL, one fresh synchronous renewal re-anchors the deadline before the fence is armed. +Failure to start the lease thread closes the fence and fails the writable open. No incident path +constructs a thread. + +One lease thread per writable mount renews the lease and runs the self-remount, one after the other. +`scheduleRemount` increments a requested-generation latch and wakes the thread; a pending request also +ends a renewal in progress. While the thread runs, only it replaces, starts or resets the renewer. + +- A reclaim latches the fence first. +- It arms the fence, and reports `Live`, only when no newer request is pending, no stop is requested + and the lifecycle is not terminal. The check and the arm are one step under the runtime's mutex, + which a stop, a request and a FORGET intent also take. +- A reclaim acknowledges only the generation it served, so a request raised during a reclaim is served + by the next reclaim before renewal resumes. +- A renewal's result is published before the thread looks at the requests again, so a renewal that + finished before a request cannot overwrite the deadline of the reclaim that follows. + +**Clean unmount:** request stop and join the lease thread, drain the ref lanes, and only if the drain *certified* quiescence call `MountLeaseRenewer::release` on an `Active` renewer to write the terminal farewell (`expires_at_ms` already expired, `min_active_build_sequence = UINT64_MAX`). That sentinel is what lets a successor reclaim instantly. A `RenewalTerminal` renewer, an unresolved ref write, or a sent From bb3817921e2c5a0241ed2cc997aa31600dd5387d Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 06:04:59 +0200 Subject: [PATCH 06/37] Trip and request the remount of a CAS interference report in one step Pool::reportImpossibleInterference calls CasMountRuntime::tripAndRequestRemount, which trips the fence and raises the remount generation under one driver_mutex section, so a reclaim's arm can no longer clear a report's trip before its request lands. A request that a reclaim has already started serving does not cover the report and a new generation is raised. Tests: the reclaim latch, the arm racing a report, a stop on a Live pool and the serialized terminal publication now fail rather than pass vacuously or hang when the behaviour regresses. Co-Authored-By: Claude Opus 5.5 --- .../ContentAddressed/Pool/CasMountRuntime.cpp | 25 ++- .../ContentAddressed/Pool/CasMountRuntime.h | 11 +- .../ContentAddressed/Pool/CasPool.cpp | 3 +- src/Disks/tests/gtest_cas_pool.cpp | 163 +++++++++++++++--- 4 files changed, 172 insertions(+), 30 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index 20be9cde8b07..464d747dbcb4 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -674,8 +674,8 @@ void CasMountRuntime::renewalLoop() lease_thread_id = std::this_thread::get_id(); } - /// The wake condition of every wait in this loop; the reclaim backoff ignores a request, which is - /// the one it is retrying. + /// The wake condition of every wait in this loop. The reclaim backoff ignores requests: a newer one + /// waits for the backoff too. const auto woken = [this](bool by_request) { const bool wake = workers_stop_requested @@ -994,11 +994,11 @@ void CasMountRuntime::enterVanished(PoolLifecycle which, const String & reason) void CasMountRuntime::publishVanishedIntent() { - /// spec §5 step 1: publish the terminal-intent latch WITHOUT settling the state. `scheduleRemount` + /// FORGET's first step: publish the terminal-intent latch WITHOUT settling the state. `scheduleRemount` /// and the lease loop consult `vanished_intent` at their step boundaries, so this stops new remount /// scheduling and makes the lease loop exit at its next step — bounding FORGET's join of the lease - /// thread to one step + one backend timeout. The state store + WARN follow in `enterVanished` - /// (step 6). Idempotent. + /// thread to one step + one backend timeout. The state store + WARN follow in `enterVanished`. + /// Idempotent. { auto lock = lockTerminalPublication(); vanished_intent.store(true, std::memory_order_release); @@ -1016,6 +1016,21 @@ void CasMountRuntime::scheduleRemount() driver_cv.notify_all(); } +void CasMountRuntime::tripAndRequestRemount() +{ + schedule_remount_calls_for_test.fetch_add(1, std::memory_order_relaxed); + std::lock_guard lock(driver_mutex); + /// Atomics only. + tripMountLost(); + if (workers_stop_requested || remountTerminal()) + return; + /// A request in `reclaim_generation` is already being served by a reclaim that latched before this + /// trip, so it does not cover the trip. + if (remount_requested_generation <= std::max(remount_handled_generation, reclaim_generation)) + ++remount_requested_generation; + driver_cv.notify_all(); +} + bool CasMountRuntime::scheduleRemountForTest() { scheduleRemount(); diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index fdef370e68e1..c049cf2e91b3 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -34,7 +34,7 @@ using PartWriteTxnPtr = std::shared_ptr; /// - `TransientNotLive` — the lease was lost; access is uncertain and a self-remount retries. The §2 /// `Present`+identity-match recovery rule fires only from here (or `Live`). /// - `IdentityLost` — the pool sentinels are authoritatively absent (both KeyAbsent): -/// fail-loud and TERMINAL (rev.8). The lease and GC threads exit; +/// fail-loud and TERMINAL. The lease and GC threads exit; /// matching-sentinel reappearance does NOT auto-revive it ([D3]); recovery is a /// restart or `SYSTEM CAS FORGET`. /// - `Vanished*` — fully terminal truth: the data root was replaced by a foreign pool, or the @@ -155,6 +155,7 @@ class CasMountRuntime uint64_t peekNextBuildSeq(); /// Renew the merged mount heartbeat once, including its build-watermark floor. A read-only runtime /// has no renewer and fails with a logical exception rather than fabricating a heartbeat. + /// A test seam: it is not protected against a concurrent replacement of the renewer. void renewWatermarkOnce(); /// ---- local write fence ---- @@ -246,10 +247,10 @@ class CasMountRuntime void setLifecycleForTest(PoolLifecycle lc); /// Publish the terminal-intent latch (`vanished_intent`) WITHOUT settling the lifecycle state. This is - /// spec §5 step 1 of `SYSTEM CAS FORGET`: publishing the latch FIRST makes the runtime stop latching + /// the first step of `SYSTEM CAS FORGET`: publishing the latch FIRST makes the runtime stop latching /// remount generations and the lease loop exit at its next step boundary, so FORGET's join of the /// lease thread is bounded by one step and one backend timeout. The state store + WARN happen - /// later, in `enterVanished` at step 6. Idempotent. Publication is serialized by `driver_mutex` and + /// later, in `enterVanished`. Idempotent. Publication is serialized by `driver_mutex` and /// followed by a condition-variable notification, so the lease loop cannot miss the terminal edge /// between its predicate sample and wait. A natural /// terminal transition does NOT call this — its `enterVanished` publishes the latch itself. @@ -380,6 +381,10 @@ class CasMountRuntime void stopBackgroundWorkers(); /// Latch a recovery generation for the lease thread. It never constructs a thread. void scheduleRemount(); + /// One interference report: trip the fence and request a remount in one `driver_mutex` step, so no + /// reclaim can arm between the two. Raises no generation when a request that no reclaim has started + /// serving is pending: the reclaim that serves it latches after this trip. + void tripAndRequestRemount(); bool scheduleRemountForTest(); void beginShutdownForTest(); /// Return how many times `scheduleRemount` was entered, including calls refused by the background diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp index 396e03a28f2e..8a5707eb9cca 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp @@ -1789,8 +1789,7 @@ void Pool::reportImpossibleInterference(const String & key, const String & reaso /// Incidental-only detection has the same fail-closed reaction as a foreign/superseded lease /// renewal. The fence and the lease thread, which runs the self-remount, live on `mount_runtime`. - mount_runtime.tripMountLost(); - mount_runtime.scheduleRemount(); + mount_runtime.tripAndRequestRemount(); /// Diagnosis off the critical path: a background task may spend a FEW /// requests -- never the caller's thread, and never blocking this call's own return. diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp index 1753b1abd9db..7de42d441825 100644 --- a/src/Disks/tests/gtest_cas_pool.cpp +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -24,7 +24,6 @@ #include #include #include -#include #include #include #include @@ -3440,6 +3439,8 @@ TEST(CASPoolRemount, ReclaimWaitsForTheRenewalToEnd) DB::Cas::tests::ManualBarrier renewal_barrier; DB::Cas::tests::ManualBarrier remount_barrier; std::atomic remount_calls{0}; + std::atomic renewal_request_returned{false}; + std::atomic reclaim_saw_the_request_returned{false}; CasEventSink sink; RuntimeUnderTest runtime_holder( backend, layout, @@ -3448,6 +3449,7 @@ TEST(CASPoolRemount, ReclaimWaitsForTheRenewalToEnd) "test", sink, runtimeRenewBudget(), [&] { ++remount_calls; + reclaim_saw_the_request_returned = renewal_request_returned.load(); remount_barrier.arriveAndWait(); return false; }); @@ -3457,6 +3459,8 @@ TEST(CASPoolRemount, ReclaimWaitsForTheRenewalToEnd) runtime.armMountFence(uuid, 1, anchor + 1000); backend->barrier = &renewal_barrier; backend->fault = RuntimeRenewBackend::Fault::BlockThenDelegate; + /// Set after the setup's own writes: only the held renewal request sets it. + backend->after_commit = [&] { renewal_request_returned = true; }; runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); renewal_barrier.waitUntilArrived(); runtime.tripMountLost(); @@ -3465,6 +3469,8 @@ TEST(CASPoolRemount, ReclaimWaitsForTheRenewalToEnd) renewal_barrier.release(); remount_barrier.waitUntilArrived(); EXPECT_EQ(remount_calls.load(), 1u); + EXPECT_TRUE(reclaim_saw_the_request_returned.load()) + << "the reclaim must start only after the renewal's request in flight returned"; remount_barrier.release(); runtime.stopBackgroundWorkers(); runtime.finishTeardown(false); @@ -3722,15 +3728,12 @@ TEST(CASPoolRemount, ALeaseWaitCannotMissNaturalTerminalPublication) MountClaimResult::Claimed); WorkerExitLatch exits; std::once_flag pause_once; - std::latch predicate_sampled_false{1}; - std::latch release_predicate{1}; - std::once_flag release_once; + /// Bounded: a regression fails after the barrier's timeout instead of hanging the gate. + DB::Cas::tests::ManualBarrier waiter; + std::atomic waiter_timed_out{false}; std::atomic waiter_holds_driver_mutex{false}; std::atomic publication_entered_while_held{false}; - const auto release_waiter = [&] - { - std::call_once(release_once, [&] { release_predicate.count_down(); }); - }; + const auto release_waiter = [&] { waiter.release(); }; RuntimeWorkerFactory factory = [&](std::function worker_body) { return ThreadFromGlobalPool([&, body = std::move(worker_body)] @@ -3752,8 +3755,14 @@ TEST(CASPoolRemount, ALeaseWaitCannotMissNaturalTerminalPublication) std::call_once(pause_once, [&] { waiter_holds_driver_mutex.store(true, std::memory_order_release); - predicate_sampled_false.count_down(); - release_predicate.wait(); + try + { + waiter.arriveAndWait(); + } + catch (...) + { + waiter_timed_out = true; + } waiter_holds_driver_mutex.store(false, std::memory_order_release); }); }, @@ -3783,7 +3792,7 @@ TEST(CASPoolRemount, ALeaseWaitCannotMissNaturalTerminalPublication) } runtime.startBackgroundWorkers(std::chrono::hours(1)); - predicate_sampled_false.wait(); + waiter.waitUntilArrived(); if (terminal == PoolLifecycle::IdentityLost) { runtime.tripMountLost(); @@ -3793,8 +3802,12 @@ TEST(CASPoolRemount, ALeaseWaitCannotMissNaturalTerminalPublication) { runtime.enterVanished(PoolLifecycle::VanishedReplaced, "injected replacement during a lease wait"); } + /// A publication that took no `driver_mutex` released nobody; release the waiter here so it + /// waits, misses the edge, and the expectation below fails instead of hanging. + release_waiter(); const bool exited_without_stop = exits.waitForAtLeast(1); runtime.stopBackgroundWorkers(); + EXPECT_FALSE(waiter_timed_out.load()); EXPECT_FALSE(publication_entered_while_held.load()) << "the publication must not take driver_mutex while the waiter holds it"; EXPECT_TRUE(exited_without_stop); @@ -4571,11 +4584,17 @@ TEST(CASMountRuntime, StopWakesTheRetryWaitOfAnUnboundedRenewal) std::chrono::steady_clock::now() - started).count(); }); backend->outage = [] { return true; }; + const uint64_t lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); ASSERT_EQ(wait_requested.wait_for(std::chrono::seconds(20)), std::future_status::ready); const uint64_t requested_ms = wait_requested.get(); runtime.stopBackgroundWorkers(); + /// The renewal the stop ended had sent a request, so it ends terminal on a `Live` pool. + EXPECT_FALSE(runtime.mayMutate()) << "the stop latches the fence"; + EXPECT_EQ(runtime.lifecycle(), PoolLifecycle::Live); + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(), lost_before) + << "the stop counts no lease loss"; EXPECT_GE(requested_ms, kMountRenewRetrySpacingMs * 8 / 10); EXPECT_LE(requested_ms, kMountRenewRetrySpacingMs * 12 / 10); @@ -4666,8 +4685,9 @@ TEST(CASMountRuntime, AStaleSuccessIsFollowedAtOnceByTheNextRenewal) runtime.finishTeardown(false); } -/// An interference report raised while a reclaim runs: that reclaim arms nothing, the next one starts -/// with the fence latched, and no write is admitted between the two. +/// A remount request raised on an armed fence: the reclaim latches the fence before anything else and +/// counts one loss. An interference report raised while that reclaim runs: it arms nothing, the next +/// one starts with the fence latched, and no write is admitted between the two. TEST(CASMountRuntime, NothingArmsWhileARemountRequestIsPending) { auto backend = std::make_shared(); @@ -4678,6 +4698,10 @@ TEST(CASMountRuntime, NothingArmsWhileARemountRequestIsPending) ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); std::atomic calls{0}; + bool may_mutate_before_first_latch = false; + bool may_mutate_after_first_latch = true; + PoolLifecycle lifecycle_after_first_latch = PoolLifecycle::Live; + uint64_t lost_after_first_latch = 0; bool first_armed = true; bool may_mutate_after_first = true; bool may_mutate_at_second_entry = true; @@ -4697,10 +4721,13 @@ TEST(CASMountRuntime, NothingArmsWhileARemountRequestIsPending) CasMountRuntime & reclaiming = *runtime_ptr; if (++calls == 1) { + may_mutate_before_first_latch = reclaiming.mayMutate(); reclaiming.beginReclaim(); + may_mutate_after_first_latch = reclaiming.mayMutate(); + lifecycle_after_first_latch = reclaiming.lifecycle(); + lost_after_first_latch = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); /// An interference report while this reclaim runs. - reclaiming.tripMountLost(); - reclaiming.scheduleRemount(); + reclaiming.tripAndRequestRemount(); first_armed = reclaiming.armIfAdmissible(boot_ms + 1000); may_mutate_after_first = reclaiming.mayMutate(); return true; @@ -4722,11 +4749,16 @@ TEST(CASMountRuntime, NothingArmsWhileARemountRequestIsPending) runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); + const uint64_t lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); runtime.startBackgroundWorkers(std::chrono::hours(1)); - runtime.tripMountLost(); + /// A request without a trip: the fence is still armed when the reclaim starts. runtime.scheduleRemount(); second_done.waitUntilArrived(); + EXPECT_TRUE(may_mutate_before_first_latch) << "the request alone must leave the fence armed"; + EXPECT_FALSE(may_mutate_after_first_latch) << "a reclaim starts with the fence latched"; + EXPECT_EQ(lifecycle_after_first_latch, PoolLifecycle::TransientNotLive); + EXPECT_EQ(lost_after_first_latch, lost_before + 1) << "the latch counts the one loss"; EXPECT_FALSE(first_armed) << "a reclaim must not arm while a newer remount request is pending"; EXPECT_FALSE(may_mutate_after_first); EXPECT_FALSE(may_mutate_at_second_entry) << "no write may be admitted between the two reclaims"; @@ -4740,6 +4772,100 @@ TEST(CASMountRuntime, NothingArmsWhileARemountRequestIsPending) runtime.finishTeardown(false); } +/// An interference report that lands while a reclaim runs, before its arm or while the arm holds +/// `driver_mutex`: the arm does not override it. The fence ends latched with one request pending, the +/// report counts a loss only on a `Live` pool, and the next reclaim serves it. +TEST(CASMountRuntime, AnInterferenceReportDuringAReclaimIsServedByTheNextReclaim) +{ + enum class ReportAt : uint8_t { BeforeTheArm, InsideTheArm }; + const auto run = [](ReportAt report_at) + { + auto backend = std::make_shared(); + const Layout layout(report_at == ReportAt::BeforeTheArm ? "report-before-arm" : "report-inside-arm"); + uint64_t wall_ms = 1000; + const uint64_t boot_ms = 100; + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, + MountClaimResult::Claimed); + std::atomic calls{0}; + bool first_armed = false; + bool may_mutate_after_first = true; + uint64_t generation_after_first = 0; + bool may_mutate_at_second_entry = true; + bool second_armed = false; + std::future report; + DB::Cas::tests::ManualBarrier second_done; + CasMountRuntime * runtime_ptr = nullptr; + CasEventSink sink; + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }}, + "test", sink, runtimeRenewBudget(), [&] + { + CasMountRuntime & reclaiming = *runtime_ptr; + if (++calls == 1) + { + reclaiming.beginReclaim(); + if (report_at == ReportAt::BeforeTheArm) + { + reclaiming.tripAndRequestRemount(); + } + else + { + /// The hook runs inside the arm with `driver_mutex` held, so the report starts + /// there and can finish only after the arm's section. + reclaiming.setArmMountFenceInterpositionHookForTest([&] + { + report = std::async(std::launch::async, [&] { runtime_ptr->tripAndRequestRemount(); }); + }); + } + first_armed = reclaiming.armIfAdmissible(boot_ms + 1000); + reclaiming.setArmMountFenceInterpositionHookForTest({}); + if (report.valid()) + report.get(); + may_mutate_after_first = reclaiming.mayMutate(); + generation_after_first = reclaiming.remountRequestedGenerationForTest(); + return true; + } + may_mutate_at_second_entry = reclaiming.mayMutate(); + reclaiming.beginReclaim(); + second_armed = reclaiming.armIfAdmissible(boot_ms + 1000); + second_done.arriveAndWait(); + return true; + }); + SCOPE_EXIT({ second_done.release(); }); + CasMountRuntime & runtime = *runtime_holder; + runtime_ptr = &runtime; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + runtime.armMountFence(uuid, 1, anchor + 1000); + const uint64_t lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); + runtime.startBackgroundWorkers(std::chrono::hours(1)); + runtime.scheduleRemount(); + + second_done.waitUntilArrived(); + if (report_at == ReportAt::BeforeTheArm) + EXPECT_FALSE(first_armed) << "a report raised during the reclaim must stop its arm"; + else + EXPECT_TRUE(first_armed) << "the report waits for the arm's section"; + EXPECT_FALSE(may_mutate_after_first) << "the arm must not override the report's trip"; + EXPECT_EQ(generation_after_first, 2u) << "the report raises one generation"; + EXPECT_FALSE(may_mutate_at_second_entry); + EXPECT_TRUE(second_armed) << "the next reclaim serves the report"; + EXPECT_EQ(calls.load(), 2u); + EXPECT_EQ(runtime.remountRequestedGenerationForTest(), 2u); + /// The first latch counts one loss; the report counts one more only when the arm made the pool `Live`. + EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(), + lost_before + (report_at == ReportAt::BeforeTheArm ? 1 : 2)); + second_done.release(); + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); + }; + run(ReportAt::BeforeTheArm); + run(ReportAt::InsideTheArm); +} + /// One interference report produces one remount generation, one epoch change and one lease-loss /// count, also when the reclaim fails twice before it succeeds. Two failures cost one and two seconds /// of the loop's real backoff. @@ -4871,8 +4997,7 @@ TEST(CASMountRuntime, AReclaimFinishedAfterTheForgetIntentArmsNothing) } /// A stop requested during a reclaim: the join returns once the attempt returns, the reclaim arms -/// nothing, the stop counts no lease loss, and no farewell is written for the slot this runtime did -/// not claim back. +/// nothing, and no farewell is written for the slot this runtime did not claim back. TEST(CASMountRuntime, StopDuringAReclaimJoinsTheThread) { auto backend = std::make_shared(); @@ -4942,8 +5067,6 @@ TEST(CASMountRuntime, StopDuringAReclaimJoinsTheThread) EXPECT_FALSE(armed) << "a reclaim that finishes after a stop must not arm the fence"; EXPECT_FALSE(runtime.mayMutate()); EXPECT_EQ(lost_at_latch, lost_before + 1) << "the fenced-out renewal counts the one loss"; - EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(), lost_at_latch) - << "the stop counts no lease loss"; const auto slot_before_teardown = readObj(*backend, key); ASSERT_TRUE(slot_before_teardown.has_value()); runtime.finishTeardown(true); From 6897788966987a356cd6b9c459dc7f48f0f2472f Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 06:18:21 +0200 Subject: [PATCH 07/37] Pin the one-step CAS interference report with a deterministic test The arm interposition hook now waits until the report's request count rises, so a report split back into an unlocked trip and a separate request fails the test. The loop's terminal consume step and tripAndRequestRemount share lossNeedsNewRequest. Comments say what the request counter counts and what the backoff case of the terminal-publication test does not catch. Co-Authored-By: Claude Opus 5.5 --- .../ContentAddressed/Pool/CasMountRuntime.cpp | 14 +++++++++----- .../ContentAddressed/Pool/CasMountRuntime.h | 8 ++++++-- src/Disks/tests/gtest_cas_pool.cpp | 19 ++++++++++++++++++- 3 files changed, 33 insertions(+), 8 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index 464d747dbcb4..baa2e5789157 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -303,6 +303,13 @@ bool CasMountRuntime::canArm(uint64_t /*deadline_boot_ms*/) const && remount_requested_generation <= remount_handled_generation; } +bool CasMountRuntime::lossNeedsNewRequest() const +{ + /// A request up to `reclaim_generation` was snapshotted by a reclaim that latched before this loss, + /// so it does not cover the loss. + return remount_requested_generation <= std::max(remount_handled_generation, reclaim_generation); +} + void CasMountRuntime::checkRenewerOwner() const { if (workers_started && lease_thread_id != std::this_thread::get_id()) @@ -511,8 +518,7 @@ void CasMountRuntime::consumeRenewResult(const MountRenewResult & result, RenewC } tripMountLost(); schedule_remount_calls_for_test.fetch_add(1, std::memory_order_relaxed); - /// A pending request already covers this loss. - if (!remountTerminal() && remount_requested_generation == remount_handled_generation) + if (!remountTerminal() && lossNeedsNewRequest()) ++remount_requested_generation; break; case RenewCaller::Direct: @@ -1024,9 +1030,7 @@ void CasMountRuntime::tripAndRequestRemount() tripMountLost(); if (workers_stop_requested || remountTerminal()) return; - /// A request in `reclaim_generation` is already being served by a reclaim that latched before this - /// trip, so it does not cover the trip. - if (remount_requested_generation <= std::max(remount_handled_generation, reclaim_generation)) + if (lossNeedsNewRequest()) ++remount_requested_generation; driver_cv.notify_all(); } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index c049cf2e91b3..fad49e80b63e 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -387,8 +387,8 @@ class CasMountRuntime void tripAndRequestRemount(); bool scheduleRemountForTest(); void beginShutdownForTest(); - /// Return how many times `scheduleRemount` was entered, including calls refused by the background - /// setting. This is useful for testing the renewer's loss callback without starting a real recovery. + /// Return how many remount requests were attempted, refused ones included: `scheduleRemount`, + /// `tripAndRequestRemount` and the loop's terminal renewals. This is useful for testing the renewer's loss callback without starting a real recovery. uint64_t scheduleRemountCallCountForTest() const { return schedule_remount_calls_for_test.load(std::memory_order_relaxed); @@ -460,6 +460,10 @@ class CasMountRuntime /// terminal. Requires `driver_mutex`, the mutex that a stop, a request and a terminal publication /// take, so the check and the arm are one step. bool canArm(uint64_t deadline_boot_ms) const; + /// Whether a new loss needs a new remount generation. Requires `driver_mutex`. False only while a + /// request that no reclaim has snapshotted is pending: the reclaim that serves it latches after the + /// loss. + bool lossNeedsNewRequest() const; /// Throws `LOGICAL_ERROR` when a lease thread runs and the caller is not it. Requires `driver_mutex`. void checkRenewerOwner() const; std::unique_lock lockTerminalPublication(); diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp index 7de42d441825..f5b3b3bb6aea 100644 --- a/src/Disks/tests/gtest_cas_pool.cpp +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -3708,7 +3708,8 @@ TEST(CASPoolRemount, NaturalTerminalTransitionMakesTheLeaseThreadExit) /// A terminal publication that races a wait of the lease loop is serialized by `driver_mutex`: it /// cannot land between the wait predicate's sample and the wait, so the thread exits without a stop. -/// Two waits: the cadence wait and the reclaim backoff. +/// Two waits: the cadence wait and the reclaim backoff. The backoff case checks only the exit: a missed +/// edge there costs one backoff, at most 1 s here, so it does not catch an unserialized publication. TEST(CASPoolRemount, ALeaseWaitCannotMissNaturalTerminalPublication) { enum class Wait : uint8_t { Cadence, ReclaimBackoff }; @@ -4794,6 +4795,7 @@ TEST(CASMountRuntime, AnInterferenceReportDuringAReclaimIsServedByTheNextReclaim bool may_mutate_at_second_entry = true; bool second_armed = false; std::future report; + std::atomic report_never_started{false}; DB::Cas::tests::ManualBarrier second_done; CasMountRuntime * runtime_ptr = nullptr; CasEventSink sink; @@ -4817,7 +4819,21 @@ TEST(CASMountRuntime, AnInterferenceReportDuringAReclaimIsServedByTheNextReclaim /// there and can finish only after the arm's section. reclaiming.setArmMountFenceInterpositionHookForTest([&] { + const uint64_t requests_before = runtime_ptr->scheduleRemountCallCountForTest(); report = std::async(std::launch::async, [&] { runtime_ptr->tripAndRequestRemount(); }); + /// One step: the count rises before the lock, so no trip has landed yet. Two + /// steps: the unlocked trip lands before the count rises, and the arm clears it. + /// The bound only turns a hang into a failure. + const auto until = std::chrono::steady_clock::now() + std::chrono::seconds(20); + while (runtime_ptr->scheduleRemountCallCountForTest() == requests_before) + { + if (std::chrono::steady_clock::now() >= until) + { + report_never_started = true; + break; + } + std::this_thread::yield(); + } }); } first_armed = reclaiming.armIfAdmissible(boot_ms + 1000); @@ -4845,6 +4861,7 @@ TEST(CASMountRuntime, AnInterferenceReportDuringAReclaimIsServedByTheNextReclaim runtime.scheduleRemount(); second_done.waitUntilArrived(); + EXPECT_FALSE(report_never_started.load()) << "the report did not start within the bound"; if (report_at == ReportAt::BeforeTheArm) EXPECT_FALSE(first_armed) << "a report raised during the reclaim must stop its arm"; else From a9af278a91b3936e4d579f496f6305bb49d5e914 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 06:31:31 +0200 Subject: [PATCH 08/37] Arm a latched CAS mount fence from a renewal with room for a ref append The arming rule gets its deadline term: a claim or a renewal arms the fence only when its deadline admits a ref append now. `armIfAdmissible` latches the fence when it does not arm, and a committed renewal arms a latched fence under the same rule. A loop renewal ends on a stop, a pending remount request or a terminal lifecycle; a lost fence alone no longer ends it. Co-Authored-By: Claude Opus 5.5 --- .../ContentAddressed/Pool/CasMountRuntime.cpp | 57 +- .../ContentAddressed/Pool/CasMountRuntime.h | 22 +- src/Disks/tests/gtest_cas_pool.cpp | 514 ++++++++++++++++++ 3 files changed, 566 insertions(+), 27 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index baa2e5789157..dba958a667cb 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -148,26 +148,31 @@ Fence::Admit CasMountRuntime::admit(uint64_t admitted_generation, uint64_t neede if (mount_fence.lost.load(std::memory_order_acquire) || fenceGeneration() != admitted_generation) return Fence::Admit::LostOrRearmed; const uint64_t now = bootMsNow(); - const uint64_t deadline = mount_fence.deadline_boot_ms.load(std::memory_order_acquire); - if (now >= deadline) + return budgetAdmits(mount_fence.deadline_boot_ms.load(std::memory_order_acquire), now, needed_ms); +} + +Fence::Admit CasMountRuntime::budgetAdmits(uint64_t deadline_boot_ms, uint64_t now_boot_ms, uint64_t needed_ms) const +{ + if (now_boot_ms >= deadline_boot_ms) return Fence::Admit::NoBudget; /// Compared by subtraction rather than as the sum `needed_ms + margin`, which can wrap for an /// absurd configuration and then read as if there were room. - const uint64_t remaining = deadline - now; + const uint64_t remaining = deadline_boot_ms - now_boot_ms; if (needed_ms >= remaining || cas_request_budget.lease_safety_margin_ms >= remaining - needed_ms) return Fence::Admit::NoBudget; return Fence::Admit::Ok; } -bool CasMountRuntime::refAppendFenceOk() const +uint64_t CasMountRuntime::refAppendReservationMs() const { - /// Two envelopes' worth of room under the live generation -- a write and its settlement read, which - /// is what `writeLoop` reserves -- so a ref-log attempt is not started when it cannot plausibly - /// finish, safety margin included, before the lease expires. + /// Two envelopes: a write and its settlement read, which is what `writeLoop` reserves. const uint64_t envelope_ms = cas_request_budget.attemptEnvelopeMs(); - const uint64_t needed_ms = envelope_ms > std::numeric_limits::max() / 2 - ? std::numeric_limits::max() : 2 * envelope_ms; - return admit(fenceGeneration(), needed_ms) == Fence::Admit::Ok; + return envelope_ms > std::numeric_limits::max() / 2 ? std::numeric_limits::max() : 2 * envelope_ms; +} + +bool CasMountRuntime::refAppendFenceOk() const +{ + return admit(fenceGeneration(), refAppendReservationMs()) == Fence::Admit::Ok; } std::optional CasMountRuntime::leaseExpiredAt(uint64_t now_boot_ms) const @@ -289,18 +294,29 @@ bool CasMountRuntime::armIfAdmissible(uint64_t deadline_boot_ms) remount_handled_generation = std::max(remount_handled_generation, reclaim_generation); const bool arm = canArm(deadline_boot_ms); if (arm) + { armFence(deadline_boot_ms, /*report_live=*/true); + } else + { + /// An open starts unarmed, which admits writes; a claim that does not admit a ref append must + /// not leave it so. Latched before the deadline is published, so no reader sees the claim's + /// deadline on an open fence. A reclaim arrives here already latched by `beginReclaim`. + if (!mount_fence.lost.load(std::memory_order_acquire)) + tripFenceWithoutOperationalLoss(); setMountDeadline(deadline_boot_ms); + } driver_cv.notify_all(); return arm; } -bool CasMountRuntime::canArm(uint64_t /*deadline_boot_ms*/) const +bool CasMountRuntime::canArm(uint64_t deadline_boot_ms) const { + /// The clock is read last and only when every other term holds. return !workers_stop_requested && !remountTerminal() - && remount_requested_generation <= remount_handled_generation; + && remount_requested_generation <= remount_handled_generation + && budgetAdmits(deadline_boot_ms, bootMsNow(), refAppendReservationMs()) == Fence::Admit::Ok; } bool CasMountRuntime::lossNeedsNewRequest() const @@ -458,9 +474,9 @@ bool CasMountRuntime::renewalLive(RenewCaller caller) const return false; if (caller != RenewCaller::Loop) return true; - return remount_requested_generation <= remount_handled_generation - && lifecycle() == PoolLifecycle::Live - && !mount_fence.lost.load(std::memory_order_acquire); + /// A lost fence alone does not end it: the renewal that makes an open or a reclaim ready runs under + /// one, and every trip that must end it comes with a request, an intent or a stop. + return remount_requested_generation <= remount_handled_generation && !remountTerminal(); } bool CasMountRuntime::renewalCancelled() const @@ -501,10 +517,13 @@ void CasMountRuntime::consumeRenewResult(const MountRenewResult & result, RenewC if (result.outcome == MountRenewOutcome::Committed) { const uint64_t ttl_ms = static_cast(config.mount_lease_ttl_ms.count()); - restored = publishRenewedDeadline( - result.attempt_start_boot_ms > std::numeric_limits::max() - ttl_ms - ? std::numeric_limits::max() - : result.attempt_start_boot_ms + ttl_ms); + const uint64_t deadline_boot_ms = result.attempt_start_boot_ms > std::numeric_limits::max() - ttl_ms + ? std::numeric_limits::max() + : result.attempt_start_boot_ms + ttl_ms; + restored = publishRenewedDeadline(deadline_boot_ms); + /// A latched fence is armed by the first renewal whose own deadline leaves room for a ref append. + if (mount_fence.lost.load(std::memory_order_acquire) && canArm(deadline_boot_ms)) + armFence(deadline_boot_ms, /*report_live=*/true); } else if (result.outcome == MountRenewOutcome::Terminal) { diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index fad49e80b63e..05ab88f24cf0 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -87,7 +87,7 @@ struct MountConfig /// Deterministic failure injection at the vanished-reason preparation boundary. std::function vanished_reason_prepare_hook_for_test = {}; /// Test-only extra liveness condition, for exact pre/post-send gate interleavings. It is ANDed with - /// the ordinary predicate, so a stop still ends the renewal. FALSE ends it exactly as a lost fence does. + /// the ordinary predicate, so a stop still ends the renewal. FALSE ends the renewal early. std::function renewal_live_for_test = {}; }; @@ -161,7 +161,9 @@ class CasMountRuntime /// ---- local write fence ---- /// Return whether a mutable operation may start under the locally observed lease state. bool mayMutate() const; - /// Permanently latch the local fence as lost for this runtime incarnation. + /// Latch the fence lost and count one lease loss. Every production caller pairs the trip with a + /// remount request, a published FORGET intent or a stop. A trip that comes alone is re-armed by the + /// next renewal that commits with room for a ref append. void tripMountLost(); /// Publish the BOOTTIME deadline from a successful lease renewal. void setMountDeadline(uint64_t deadline_boot_ms); @@ -173,7 +175,7 @@ class CasMountRuntime void beginReclaim(); /// End of a reclaim that claimed. One step under `driver_mutex`: acknowledge the generation /// `beginReclaim` recorded, publish the deadline, and arm the fence and report `Live` if the arming - /// rule holds. Returns whether it armed. + /// rule holds; otherwise latch the fence. Returns whether it armed. bool armIfAdmissible(uint64_t deadline_boot_ms); /// Test-only interposition at the publication boundary between the re-armed generation and the /// live fence. A caller admitted from this hook must be refused: the old generation is already @@ -445,9 +447,9 @@ class CasMountRuntime void consumeRenewResult(const MountRenewResult & result, RenewCaller caller); void renewalLoop(); ThreadFromGlobalPool makeWorker(std::function body); - /// The renewal's liveness: no stop requested and, for the loop, no pending remount request, a pool - /// that is `Live` and a fence that is not lost -- the loop's plane has no fence of its own. FALSE - /// ends the renewal. + /// The renewal's liveness. Every caller ends on a stop. A loop renewal also ends on a pending remount + /// request and on a terminal lifecycle, which includes a published FORGET intent; a lost fence alone + /// does not end it. FALSE ends the renewal. bool renewalLive(RenewCaller caller) const; /// Whether this node has already been asked to stop. Sampled ONCE, before the write, so a refusal /// caused by the stop cannot be mistaken for one that preceded it. @@ -457,9 +459,13 @@ class CasMountRuntime /// `Live` is published before the fence opens, so a reader that sees the fence armed reads `Live`. void armFence(uint64_t deadline_boot_ms, bool report_live); /// The arming rule: no remount request pending, no stop requested, a lifecycle that is not - /// terminal. Requires `driver_mutex`, the mutex that a stop, a request and a terminal publication - /// take, so the check and the arm are one step. + /// terminal, and a deadline that admits a ref append now. Requires `driver_mutex`, the mutex that a + /// stop, a request and a terminal publication take, so the check and the arm are one step. bool canArm(uint64_t deadline_boot_ms) const; + /// `admit`'s budget verdict for a lease that ends at `deadline_boot_ms`, at `now_boot_ms`. + Fence::Admit budgetAdmits(uint64_t deadline_boot_ms, uint64_t now_boot_ms, uint64_t needed_ms) const; + /// What a ref append reserves: a write and the read that settles it, two attempt envelopes. + uint64_t refAppendReservationMs() const; /// Whether a new loss needs a new remount generation. Requires `driver_mutex`. False only while a /// request that no reclaim has snapshotted is pending: the reclaim that serves it latches after the /// loss. diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp index f5b3b3bb6aea..b6591451516e 100644 --- a/src/Disks/tests/gtest_cas_pool.cpp +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -6307,3 +6307,517 @@ TEST(CASMountRuntime, AnExpiryEndedByAFenceIsNotCountedAsARestore) runtime.stopBackgroundWorkers(); runtime.finishTeardown(false); } + +namespace +{ +/// Clock values of the runtime readiness cases. With `runtimeRenewBudget` a ref append needs +/// 2 x 10 + 20 = 40 ms of lease. The claim starts at 100 s and the open decides with 30 ms left. +constexpr uint64_t kReadinessDecideBootMs = kExpiryDeadlineBootMs - 30; +/// The first readiness renewal fails 24 times, 1.3 s each, and commits with a start more than a TTL ago. +constexpr uint64_t kReadinessStaleCommitBootMs = kReadinessDecideBootMs + kExpiryFailedPuts * kExpiryFailedPutMs; +static_assert(kReadinessStaleCommitBootMs >= kReadinessDecideBootMs + kExpiryTtlMs); +/// The second commits with 35 ms of its lease left: in the future, but under the 40 ms a ref append needs. +constexpr uint64_t kReadinessShortCommitBootMs = kReadinessStaleCommitBootMs + kExpiryTtlMs - 35; +/// Far above any request count these tests reach, so only a regression hits it. +constexpr uint32_t kReadinessRequestBound = 100; + +struct ReadinessView +{ + const char * admit = ""; + bool may_mutate = true; + PoolLifecycle lifecycle = PoolLifecycle::IdentityLost; + bool expired = true; + String failure; + uint64_t generation = 0; + bool ref_append_ok = false; +}; + +ReadinessView readinessViewOf(CasMountRuntime & runtime) +{ + return ReadinessView{ + .admit = expiryAdmitName(runtime.admit(runtime.fenceGeneration(), 0)), + .may_mutate = runtime.mayMutate(), + .lifecycle = runtime.lifecycle(), + .expired = runtime.leaseExpiredSinceBootMs().has_value(), + .failure = runtime.lastRenewFailure(), + .generation = runtime.fenceGeneration(), + .ref_append_ok = runtime.refAppendFenceOk(), + }; +} +} + +/// The open's closed state, at the runtime level: a claim with too little lease for a ref append leaves +/// the fence latched, and only a loop renewal whose own deadline leaves that room arms it. A stale success +/// and a success with too little room arm nothing, and each is followed at once by the next renewal. +TEST(CASMountRuntime, AReadinessRenewalArmsOnlyWithRoomForARefAppend) +{ + /// Everything the hooks capture is declared before the backend and the runtime that store them. + const Layout layout("runtime-readiness-room"); + const UInt128 uuid{1}; + uint64_t wall_ms = 1000; + std::atomic boot_ms{kExpiryClaimBootMs}; + std::atomic puts{0}; + std::atomic remount_calls{0}; + std::atomic request_bound_hit{false}; + uint32_t loop_passes = 0; + std::vector admitted_at; + ReadinessView during_first_put; + ReadinessView after_stale; + ReadinessView after_short; + ReadinessView after_arm; + DB::Cas::tests::ManualBarrier armed_pass; + CasMountRuntime * runtime_ptr = nullptr; + const uint64_t lost_before = eventCount(ProfileEvents::CASMountLeaseLost); + const uint64_t expired_before = eventCount(ProfileEvents::CASMountLeaseExpired); + CasEventSink sink; + auto backend = std::make_shared(); + + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, kExpiryTtlMs).kind, + MountClaimResult::Claimed); + + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{ + .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), + .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms.load(); }, + .renewal_before_driver_lock_hook_for_test = [&] + { + ++loop_passes; + if (loop_passes == 2) + after_stale = readinessViewOf(*runtime_ptr); + else if (loop_passes == 3) + after_short = readinessViewOf(*runtime_ptr); + else if (loop_passes == 4) + { + after_arm = readinessViewOf(*runtime_ptr); + armed_pass.arriveAndWait(); + } + }, + .renewal_admitted_hook_for_test = [&] { admitted_at.push_back(boot_ms.load()); }, + .renewal_live_for_test = [&] + { + if (puts.load() < kReadinessRequestBound) + return true; + request_bound_hit = true; + return false; + }}, + "test", sink, runtimeRenewBudget(), [&] + { + ++remount_calls; + return false; + }); + CasMountRuntime & runtime = *runtime_holder; + runtime_ptr = &runtime; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + ASSERT_EQ(anchor, kExpiryClaimBootMs); + + /// Case 5, before the thread starts: the claim leaves 30 ms of its lease, a ref append needs 40. + boot_ms = kReadinessDecideBootMs; + const uint64_t generation_before = runtime.fenceGeneration(); + EXPECT_FALSE(runtime.armIfAdmissible(anchor + kExpiryTtlMs)) << "a claim without room for a ref append arms nothing"; + const uint64_t latched_generation = runtime.fenceGeneration(); + EXPECT_EQ(latched_generation, generation_before + 1) << "the latch ends the unarmed incarnation"; + EXPECT_STREQ(expiryAdmitName(runtime.admit(latched_generation, 0)), "LostOrRearmed"); + EXPECT_FALSE(runtime.mayMutate()); + EXPECT_EQ(runtime.lifecycle(), PoolLifecycle::Live) << "the open's closed state; nothing was lost"; + EXPECT_FALSE(runtime.leaseExpiredSinceBootMs().has_value()) << "a latched fence is not an expiry"; + EXPECT_EQ(eventCount(ProfileEvents::CASMountLeaseLost), lost_before) << "the latch counts no lease loss"; + + backend->on_put = [&](uint32_t put_no) + { + ++puts; + if (put_no == 1) + during_first_put = readinessViewOf(*runtime_ptr); + if (put_no <= kExpiryFailedPuts) + { + boot_ms.fetch_add(kExpiryFailedPutMs); + return true; + } + /// Put 25 commits the first renewal, stale. Put 26, the second renewal, lands with 35 ms left. + if (put_no == kExpiryFailedPuts + 2) + boot_ms.store(kReadinessShortCommitBootMs); + return false; + }; + runtime.startBackgroundWorkers(std::chrono::milliseconds(kExpiryPeriodMs)); + + bool arrived = true; + try + { + armed_pass.waitUntilArrived(); + } + catch (const DB::Exception &) + { + arrived = false; + } + if (!arrived) + { + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); + FAIL() << "no renewal armed the fence; request bound hit: " << request_bound_hit.load(); + } + + /// Case 5, while the readiness renewal itself is sent. + EXPECT_STREQ(during_first_put.admit, "LostOrRearmed") << "the latched fence refuses a write while the renewal is sent"; + EXPECT_FALSE(during_first_put.may_mutate); + EXPECT_FALSE(during_first_put.expired); + + /// Case 3: the first renewal started at 129.97 s and committed at 161.17 s, past its own start + TTL. + EXPECT_STREQ(after_stale.admit, "LostOrRearmed") << "a stale success arms nothing"; + EXPECT_EQ(after_stale.lifecycle, PoolLifecycle::Live); + EXPECT_EQ(after_stale.generation, latched_generation); + EXPECT_FALSE(after_stale.expired) << "a latched fence reports no expiry, also after a stale success"; + EXPECT_NE(after_stale.failure.find("injected renewal timeout"), String::npos) + << "a stale success does not end the run of trouble: " << after_stale.failure; + ASSERT_EQ(admitted_at.size(), 3u); + EXPECT_EQ(admitted_at[0], kReadinessDecideBootMs) << "the first renewal is due at once"; + EXPECT_EQ(admitted_at[1], kReadinessStaleCommitBootMs) << "a stale success is followed at once"; + + /// Case 4: the second renewal started at 161.17 s and committed at 191.135 s with 35 ms left. + EXPECT_STREQ(after_short.admit, "LostOrRearmed") << "a success with less lease than a ref append needs arms nothing"; + EXPECT_EQ(after_short.generation, latched_generation); + EXPECT_TRUE(after_short.failure.empty()) << "a deadline in the future ends the run of trouble: " << after_short.failure; + EXPECT_FALSE(after_short.expired); + EXPECT_EQ(admitted_at[2], kReadinessShortCommitBootMs) << "a short success is followed at once"; + + /// The third renewal started at 191.135 s with the whole TTL ahead: it arms, from its own start. + EXPECT_STREQ(after_arm.admit, "Ok"); + EXPECT_TRUE(after_arm.ref_append_ok) << "the arm leaves room for a ref append"; + EXPECT_EQ(after_arm.lifecycle, PoolLifecycle::Live); + EXPECT_EQ(after_arm.generation, latched_generation + 1) << "one arm, one new generation"; + EXPECT_FALSE(after_arm.expired); + EXPECT_EQ(eventCount(ProfileEvents::CASMountLeaseLost), lost_before) << "readiness is not a lease loss"; + EXPECT_EQ(eventCount(ProfileEvents::CASMountLeaseExpired), expired_before) << "an arm is not a restore"; + EXPECT_EQ(puts.load(), kExpiryFailedPuts + 3); + EXPECT_FALSE(request_bound_hit.load()); + EXPECT_EQ(remount_calls.load(), 0u); + + armed_pass.release(); + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); +} + +/// A trip that comes with no remount request, no intent and no stop does not end a loop renewal: its +/// requests keep going out under the latched fence until one commits. +TEST(CASMountRuntime, ARenewalIsSentUnderALatchedFence) +{ + const Layout layout("runtime-renew-under-latch"); + const UInt128 uuid{1}; + uint64_t wall_ms = 1000; + std::atomic boot_ms{kExpiryClaimBootMs}; + std::atomic puts{0}; + std::atomic remount_calls{0}; + std::atomic request_bound_hit{false}; + uint32_t loop_passes = 0; + std::vector admit_at_put; + DB::Cas::tests::ManualBarrier third_put; + DB::Cas::tests::ManualBarrier second_pass; + CasMountRuntime * runtime_ptr = nullptr; + CasEventSink sink; + auto backend = std::make_shared(); + + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, kExpiryTtlMs).kind, + MountClaimResult::Claimed); + + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{ + .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), + .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms.load(); }, + .renewal_before_driver_lock_hook_for_test = [&] + { + ++loop_passes; + if (loop_passes == 1) + boot_ms.store(kExpiryFirstStartBootMs); + else if (loop_passes == 2) + second_pass.arriveAndWait(); + }, + .renewal_live_for_test = [&] + { + if (puts.load() < kReadinessRequestBound) + return true; + request_bound_hit = true; + return false; + }}, + "test", sink, runtimeRenewBudget(), [&] + { + ++remount_calls; + return false; + }); + CasMountRuntime & runtime = *runtime_holder; + runtime_ptr = &runtime; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + runtime.armMountFence(uuid, 1, anchor + kExpiryTtlMs); + + /// Puts 1 to 4 fail, 1.3 s each; put 5 commits at 115.2 s, inside the lease of the 100 s claim. + backend->on_put = [&](uint32_t put_no) + { + ++puts; + admit_at_put.push_back(expiryAdmitName(runtime_ptr->admit(runtime_ptr->fenceGeneration(), 0))); + if (put_no == 3) + third_put.arriveAndWait(); + if (put_no <= 4) + { + boot_ms.fetch_add(kExpiryFailedPutMs); + return true; + } + return false; + }; + runtime.startBackgroundWorkers(std::chrono::milliseconds(kExpiryPeriodMs)); + + third_put.waitUntilArrived(); + runtime.tripMountLost(); + third_put.release(); + bool arrived = true; + try + { + second_pass.waitUntilArrived(); + } + catch (const DB::Exception &) + { + arrived = false; + } + if (!arrived) + { + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); + FAIL() << "the renewal never finished; request bound hit: " << request_bound_hit.load(); + } + + EXPECT_EQ(puts.load(), 5u) << "the renewal kept sending after the trip until it committed"; + EXPECT_EQ(admit_at_put, (std::vector{"Ok", "Ok", "Ok", "LostOrRearmed", "LostOrRearmed"})) + << "puts 4 and 5 went out under the latched fence"; + EXPECT_EQ(runtime.remountRequestedGenerationForTest(), 0u) << "a lost fence alone requests no remount"; + EXPECT_EQ(remount_calls.load(), 0u); + EXPECT_FALSE(request_bound_hit.load()); + + second_pass.release(); + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); +} + +/// A FORGET intent published while the renewal retries ends it at its next admission check, with no trip +/// and no remount request; the loop exits. +TEST(CASMountRuntime, TheForgetIntentAloneEndsARenewal) +{ + const Layout layout("runtime-intent-alone"); + const UInt128 uuid{1}; + uint64_t wall_ms = 1000; + std::atomic boot_ms{kExpiryClaimBootMs}; + std::atomic puts{0}; + std::atomic remount_calls{0}; + std::atomic request_bound_hit{false}; + uint32_t loop_passes = 0; + DB::Cas::tests::ManualBarrier third_put; + WorkerExitLatch exits; + RuntimeWorkerFactory factory = [&](std::function worker_body) + { + return ThreadFromGlobalPool([&, body = std::move(worker_body)] + { + body(); + exits.recordExit(); + }); + }; + const uint64_t lost_before = eventCount(ProfileEvents::CASMountLeaseLost); + CasEventSink sink; + auto backend = std::make_shared(); + + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, kExpiryTtlMs).kind, + MountClaimResult::Claimed); + + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{ + .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), + .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms.load(); }, + .worker_factory = factory, + .renewal_before_driver_lock_hook_for_test = [&] + { + if (++loop_passes == 1) + boot_ms.store(kExpiryFirstStartBootMs); + }, + .renewal_live_for_test = [&] + { + if (puts.load() < kReadinessRequestBound) + return true; + request_bound_hit = true; + return false; + }}, + "test", sink, runtimeRenewBudget(), [&] + { + ++remount_calls; + return false; + }); + CasMountRuntime & runtime = *runtime_holder; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + runtime.armMountFence(uuid, 1, anchor + kExpiryTtlMs); + + /// Every put fails; put 3 is held until the intent is published. + backend->on_put = [&](uint32_t put_no) + { + ++puts; + if (put_no == 3) + third_put.arriveAndWait(); + boot_ms.fetch_add(kExpiryFailedPutMs); + return true; + }; + runtime.startBackgroundWorkers(std::chrono::milliseconds(kExpiryPeriodMs)); + + third_put.waitUntilArrived(); + const uint64_t generation_at_intent = runtime.remountRequestedGenerationForTest(); + runtime.publishVanishedIntent(); + third_put.release(); + + const bool exited = exits.waitForAtLeast(1); + EXPECT_TRUE(exited) << "the loop exits on the published intent"; + EXPECT_EQ(puts.load(), 3u) << "no request after the intent"; + EXPECT_FALSE(request_bound_hit.load()) << "the intent, not the request bound, ended the renewal"; + EXPECT_EQ(runtime.remountRequestedGenerationForTest(), generation_at_intent); + EXPECT_EQ(remount_calls.load(), 0u); + EXPECT_FALSE(runtime.mayMutate()) << "the early end trips the fence"; + EXPECT_EQ(eventCount(ProfileEvents::CASMountLeaseLost), lost_before) << "a FORGET is not a lease loss"; + + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); +} + +/// A trip with no request, no intent and no stop: the next renewal that commits with room for a ref +/// append arms the fence again under the same epoch and reports `Live`. +TEST(CASMountRuntime, ATripAloneIsRearmedByTheNextRenewal) +{ + const Layout layout("runtime-trip-alone"); + const UInt128 uuid{1}; + uint64_t wall_ms = 1000; + std::atomic boot_ms{kExpiryClaimBootMs}; + std::atomic remount_calls{0}; + uint32_t loop_passes = 0; + uint64_t generation_after_trip = 0; + ReadinessView after_renewal; + DB::Cas::tests::ManualBarrier second_pass; + CasMountRuntime * runtime_ptr = nullptr; + const uint64_t lost_before = eventCount(ProfileEvents::CASMountLeaseLost); + CasEventSink sink; + auto backend = std::make_shared(); + + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, kExpiryTtlMs).kind, + MountClaimResult::Claimed); + + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{ + .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), + .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms.load(); }, + .renewal_before_driver_lock_hook_for_test = [&] + { + ++loop_passes; + if (loop_passes == 1) + { + boot_ms.store(kExpiryFirstStartBootMs); + runtime_ptr->tripMountLost(); + generation_after_trip = runtime_ptr->fenceGeneration(); + } + else if (loop_passes == 2) + { + after_renewal = readinessViewOf(*runtime_ptr); + second_pass.arriveAndWait(); + } + }}, + "test", sink, runtimeRenewBudget(), [&] + { + ++remount_calls; + return false; + }); + CasMountRuntime & runtime = *runtime_holder; + runtime_ptr = &runtime; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + runtime.armMountFence(uuid, 1, anchor + kExpiryTtlMs); + runtime.startBackgroundWorkers(std::chrono::milliseconds(kExpiryPeriodMs)); + + second_pass.waitUntilArrived(); + EXPECT_STREQ(after_renewal.admit, "Ok") << "the renewal at 110 s re-armed the fence"; + EXPECT_EQ(after_renewal.lifecycle, PoolLifecycle::Live); + EXPECT_EQ(after_renewal.generation, generation_after_trip + 1); + EXPECT_EQ(runtime.liveWriterEpoch(), 0u) << "no reclaim ran: the epoch the test never published is unchanged"; + EXPECT_EQ(runtime.remountRequestedGenerationForTest(), 0u); + EXPECT_EQ(remount_calls.load(), 0u); + EXPECT_EQ(eventCount(ProfileEvents::CASMountLeaseLost), lost_before + 1) << "the trip counted its loss once"; + second_pass.release(); + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); +} + +/// A renewal that commits while a remount request is pending does not arm a latched fence, whatever room +/// its deadline leaves: the reclaim that follows starts from a fence that refuses writes. The request is +/// raised from the renewal's liveness check after its write landed, the last check before the commit is +/// consumed; a request raised earlier ends the renewal instead. +TEST(CASMountRuntime, ARenewalArmsNothingWhileARequestIsPending) +{ + const Layout layout("runtime-no-arm-while-pending"); + const UInt128 uuid{1}; + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + String admit_at_reclaim = "unobserved"; + bool may_mutate_at_reclaim = true; + PoolLifecycle lifecycle_at_reclaim = PoolLifecycle::Live; + std::atomic renewal_landed{false}; + bool request_raised = false; + DB::Cas::tests::ManualBarrier reclaim_entered; + CasMountRuntime * runtime_ptr = nullptr; + CasEventSink sink; + auto backend = std::make_shared(); + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, + MountClaimResult::Claimed); + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{ + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms; }, + .renewal_live_for_test = [&] + { + if (renewal_landed.load() && !request_raised) + { + request_raised = true; + runtime_ptr->tripMountLost(); + runtime_ptr->scheduleRemount(); + } + return true; + }}, + "test", sink, runtimeRenewBudget(), [&] + { + admit_at_reclaim = expiryAdmitName(runtime_ptr->admit(runtime_ptr->fenceGeneration(), 0)); + may_mutate_at_reclaim = runtime_ptr->mayMutate(); + lifecycle_at_reclaim = runtime_ptr->lifecycle(); + reclaim_entered.arriveAndWait(); + return false; + }); + CasMountRuntime & runtime = *runtime_holder; + runtime_ptr = &runtime; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + runtime.armMountFence(uuid, 1, anchor + 1000); + const String key = layout.mountKey("test"); + const uint64_t writes_before = backend->putOverwriteCount(key); + const uint64_t requests_before = runtime.scheduleRemountCallCountForTest(); + backend->after_commit = [&] { renewal_landed = true; }; + runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + + reclaim_entered.waitUntilArrived(); + EXPECT_TRUE(request_raised); + EXPECT_EQ(backend->putOverwriteCount(key), writes_before + 1) << "the renewal committed, with 1000 ms of lease ahead"; + EXPECT_EQ(runtime.scheduleRemountCallCountForTest(), requests_before + 1) + << "the commit was consumed as one: a terminal loop renewal would have counted a request of its own"; + EXPECT_EQ(admit_at_reclaim, "LostOrRearmed") << "a commit consumed under a pending request arms nothing"; + EXPECT_FALSE(may_mutate_at_reclaim); + EXPECT_EQ(lifecycle_at_reclaim, PoolLifecycle::TransientNotLive); + reclaim_entered.release(); + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); +} From ccda27bb6236175d198274a2ca705790ae8f52c4 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 06:37:03 +0200 Subject: [PATCH 09/37] Say that writes resume when a renewal restores the CAS mount lease A renewal can commit with a start more than a lease ago and leave the lease expired, so a success alone does not resume writes. Co-Authored-By: Claude Opus 5.5 --- .../MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp | 2 +- src/Disks/tests/gtest_cas_mount_runtime.cpp | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index dba958a667cb..cb88e80b9232 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -195,7 +195,7 @@ std::optional CasMountRuntime::leaseExpiredRefusal(uint64_t admitted_gen if (fenceGeneration() != admitted_generation || !leaseExpiredSinceBootMs()) return std::nullopt; return String("the mount lease expired and no renewal has restored it yet; " - "writes resume when a renewal succeeds"); + "writes resume when a renewal restores it"); } String CasMountRuntime::lastRenewFailure() const diff --git a/src/Disks/tests/gtest_cas_mount_runtime.cpp b/src/Disks/tests/gtest_cas_mount_runtime.cpp index 7a2ccbbd8830..50f1f30ee31d 100644 --- a/src/Disks/tests/gtest_cas_mount_runtime.cpp +++ b/src/Disks/tests/gtest_cas_mount_runtime.cpp @@ -237,7 +237,7 @@ TEST(CASMountRuntime, ExpiredLeaseRefusalSaysWritesResume) const String expired = refusalText([&] { f->checkFenceOrThrow(generation); }); EXPECT_NE(expired.find("lease expired"), String::npos) << expired; - EXPECT_NE(expired.find("writes resume when a renewal succeeds"), String::npos) << expired; + EXPECT_NE(expired.find("writes resume when a renewal restores it"), String::npos) << expired; /// A re-arm moves the generation while the lease stays expired: the caller's incarnation is gone. f->armMountFence(kUuid, 1, /*deadline_boot_ms=*/1'100); From 123d67c7b4e18baa63431ebf66e2ec0fbca799fb Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 07:00:36 +0200 Subject: [PATCH 10/37] Make a CAS mount ready through the lease thread instead of a re-anchor An open whose claim does not admit a ref append latches the fence, starts the lease thread and waits up to one lease for a renewal to arm it; then it stops and joins the thread and fails with ABORTED. A writable open without a lease thread fails at once in that case. A reclaim whose claim is too old reports success at step claimed_not_armed and the next renewal arms the fence. The startup and remount re-anchors have no production caller left. Co-Authored-By: Claude Opus 5.5 --- .../ContentAddressed/Pool/CasMountRuntime.cpp | 36 +- .../ContentAddressed/Pool/CasMountRuntime.h | 13 +- .../ContentAddressed/Pool/CasPool.cpp | 141 +-- src/Disks/tests/gtest_cas_pool.cpp | 1082 +++++++++++------ 4 files changed, 767 insertions(+), 505 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index cb88e80b9232..b23e58fa1201 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -215,8 +215,16 @@ void CasMountRuntime::noteRenewRequest(const MountRenewRequestEvent & event) noe } try { - std::lock_guard lock(renew_failure_mutex); - last_renew_failure = event.failure_text; + { + std::lock_guard lock(renew_failure_mutex); + last_renew_failure = event.failure_text; + } + /// `waitUntilArmed` re-reads the fence clock on this wake-up. Passing through `driver_mutex` orders + /// the notification after its predicate check, so the wake-up is not lost. + { + std::lock_guard lock(driver_mutex); + } + driver_cv.notify_all(); } catch (...) // NOLINT(bugprone-empty-catch) { @@ -310,6 +318,28 @@ bool CasMountRuntime::armIfAdmissible(uint64_t deadline_boot_ms) return arm; } +bool CasMountRuntime::waitUntilArmed(uint64_t timeout_ms) const +{ + const uint64_t started = bootMsNow(); + const uint64_t give_up = started > std::numeric_limits::max() - timeout_ms + ? std::numeric_limits::max() + : started + timeout_ms; + std::unique_lock lock(driver_mutex); + while (true) + { + if (!mount_fence.lost.load(std::memory_order_acquire)) + return true; + if (workers_stop_requested || remountTerminal() || !workers_started) + return false; + const uint64_t now = bootMsNow(); + if (now >= give_up) + return false; + /// The fence clock can be injected and move with no real time passing, so every wake-up re-reads + /// it: an arm, a failed renewal request, a stop and a terminal publication all notify. + driver_cv.wait_for(lock, std::chrono::milliseconds(give_up - now)); + } +} + bool CasMountRuntime::canArm(uint64_t deadline_boot_ms) const { /// The clock is read last and only when every other term holds. @@ -475,7 +505,7 @@ bool CasMountRuntime::renewalLive(RenewCaller caller) const if (caller != RenewCaller::Loop) return true; /// A lost fence alone does not end it: the renewal that makes an open or a reclaim ready runs under - /// one, and every trip that must end it comes with a request, an intent or a stop. + /// one, and every trip that must end it comes with a request, a terminal lifecycle or a stop. return remount_requested_generation <= remount_handled_generation && !remountTerminal(); } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index 05ab88f24cf0..93b30c1a9cf4 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -161,9 +161,10 @@ class CasMountRuntime /// ---- local write fence ---- /// Return whether a mutable operation may start under the locally observed lease state. bool mayMutate() const; - /// Latch the fence lost and count one lease loss. Every production caller pairs the trip with a - /// remount request, a published FORGET intent or a stop. A trip that comes alone is re-armed by the - /// next renewal that commits with room for a ref append. + /// Latch the fence lost and count one lease loss. A trip that comes alone is re-armed by the next + /// renewal that commits with room for a ref append, so every production caller pairs it with a + /// remount request, a terminal lifecycle (a published FORGET intent included) or a stop, or trips + /// where no renewal follows: the lease loop's error exit and a direct renewal. void tripMountLost(); /// Publish the BOOTTIME deadline from a successful lease renewal. void setMountDeadline(uint64_t deadline_boot_ms); @@ -177,6 +178,9 @@ class CasMountRuntime /// `beginReclaim` recorded, publish the deadline, and arm the fence and report `Live` if the arming /// rule holds; otherwise latch the fence. Returns whether it armed. bool armIfAdmissible(uint64_t deadline_boot_ms); + /// Blocks until the fence is armed. Gives up after `timeout_ms` on the fence clock, on a stop, on a + /// terminal lifecycle, or when no lease thread runs. Returns whether the fence is armed. + bool waitUntilArmed(uint64_t timeout_ms) const; /// Test-only interposition at the publication boundary between the re-armed generation and the /// live fence. A caller admitted from this hook must be refused: the old generation is already /// dead, while the new generation is not live until `lost` is cleared. Through `armIfAdmissible` @@ -317,7 +321,8 @@ class CasMountRuntime /// because the lease expired; empty when the refusal has any other cause or there is none. std::optional leaseExpiredRefusal(uint64_t admitted_generation) const; /// Counts each `PUT` of the worker's renewal as it is sent and keeps the text of every failed - /// `PUT` or resolve read. Runs on the renewing thread. + /// `PUT` or resolve read, and wakes `waitUntilArmed` on each failure. Runs on the renewing thread, + /// never under `driver_mutex`. void noteRenewRequest(const MountRenewRequestEvent & event) noexcept; /// TRUE once the pool has reached — or is being driven toward — a state on which the lease thread diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp index 8a5707eb9cca..4e8d25a481eb 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp @@ -710,8 +710,7 @@ void Pool::mountWritable(PoolPtr & store, UInt128 our_uuid, MountClaimPolicy pol /// loop to classify (and log) an unclean reclaim. MountPriorState claimed_prior = MountPriorState::None; /// The pre-I/O boot-clock instant of the claim attempt FINALLY adopted below -- survives the - /// `break` so the arm below can detect a claim that consumed the lease TTL and re-anchor before - /// arming (rev.4 Phase B, round-3 finding 2). + /// `break` so the arm below computes the claim's deadline from it. uint64_t claim_anchor_boot_ms = 0; constexpr int max_fence_recoveries = 3; for (int fence_recovery = 0; ; ++fence_recovery) @@ -862,64 +861,39 @@ void Pool::mountWritable(PoolPtr & store, UInt128 our_uuid, MountClaimPolicy pol : ""); } - /// Arm the local write fence: cache (uuid, epoch) and set the boottime deadline at the claim - /// attempt's anchor + ttl (NOT `bootMsNow()` here -- arming from a post-I/O instant would authorize - /// mutations under a deadline the durable lease never actually backs). From here ordinary ref - /// mutations (appendRefOps) are fence-gated via mayMutate. + /// The claim's deadline is its attempt's start plus the TTL, never a later instant: a post-I/O instant + /// would authorize writes under a deadline the durable lease does not back. The fence is armed from it + /// only if it still admits a ref append; otherwise it stays latched until a renewal arms it. const uint64_t ttl_ms_u = static_cast(store->config.mount_lease_ttl_ms.count()); - const uint64_t safety_ms = store->config.cas_request_budget.lease_safety_margin_ms; - const uint64_t safe_deadline = claim_anchor_boot_ms > std::numeric_limits::max() - ttl_ms_u - ? std::numeric_limits::max() - safety_ms - : claim_anchor_boot_ms + ttl_ms_u - safety_ms; - const uint64_t now_boot_ms = store->bootMsNow(); - const uint64_t period_ms = static_cast(store->config.mount_renew_period.count()); - const uint64_t envelope_ms = store->config.cas_request_budget.attemptEnvelopeMs(); - const uint64_t two_envelopes_ms = envelope_ms > std::numeric_limits::max() / 2 - ? std::numeric_limits::max() : 2 * envelope_ms; - const uint64_t renewal_window_ms = store->config.background_watermark - ? (two_envelopes_ms > std::numeric_limits::max() - period_ms - ? std::numeric_limits::max() : period_ms + two_envelopes_ms) - : two_envelopes_ms; - /// Preserve one ordinary cadence followed by one physical renewal attempt (a write and its - /// settlement read: two envelopes) inside the safe lease window. If that publication horizon was - /// consumed, re-anchor synchronously before opening the fence; the renewer independently retains - /// its per-request deadline checks. STRICT, like `CasMountRuntime::admit`: a horizon that fits - /// exactly still starts a renewal the fence would then refuse. - const bool renewal_window_fits = now_boot_ms <= safe_deadline - && renewal_window_ms < safe_deadline - now_boot_ms; - if (!renewal_window_fits) + const uint64_t claim_deadline_boot_ms = claim_anchor_boot_ms > std::numeric_limits::max() - ttl_ms_u + ? std::numeric_limits::max() + : claim_anchor_boot_ms + ttl_ms_u; + store->mount_runtime.setLiveWriterEpoch(writer_epoch); + const bool armed = store->mount_runtime.armIfAdmissible(claim_deadline_boot_ms); + if (!store->config.background_watermark) { - /// The claim path outlived the lease TTL: its anchor can no longer authorize an armed fence (a - /// successor may have legally started reclaiming). Re-anchor with ONE fresh conditional lease - /// write -- it fails closed (Phase A classification) if anything took the slot meanwhile -- and - /// arm from the new attempt's anchor (rev.4 Phase B, round-3 finding 2). - /// - /// The unbounded operator-configured wait this guard was written for (`T_mat`) is gone, so - /// reaching it now means the claim's adoption write outran the safe lease window - /// TTL, which `validateCasRequestBudget` already refuses to configure. It stays because a stalled - /// socket can still outlive a budget, and its recovery is one conditional write that fails closed; - /// it is LOUD rather than fatal because a slow open under a healthy protocol is not a reason to - /// refuse to start. - LOG_WARNING(getLogger("CasPool"), - "Content-addressed mount {}: the mount claim consumed the lease TTL ({} ms) before the write " - "fence could be armed; re-writing the lease first", srid, ttl_ms_u); - claim_anchor_boot_ms = store->mount_runtime.renewRenewerForStartupOnce(); + /// Nothing would renew this claim, so a latched fence would refuse writes for good. + if (!armed) + throw Exception(ErrorCodes::ABORTED, + "CAS mount '{}': the mount claim left too little of the {} ms lease to admit a write, and this " + "mount has no lease thread to renew it; retry the open", srid, ttl_ms_u); + return; } - store->mount_runtime.setLiveWriterEpoch(writer_epoch); - store->armMountFence( - our_uuid, - writer_epoch, - claim_anchor_boot_ms > std::numeric_limits::max() - ttl_ms_u - ? std::numeric_limits::max() - : claim_anchor_boot_ms + ttl_ms_u); - /// Gate the lease thread with `background_watermark`: it runs only in production - /// (`background_watermark` = context != nullptr && !read_only), never in unit tests — which - /// drive `renewWatermarkOnce` explicitly and rely on the armed sub-TTL deadline, never on a loop. - /// The synchronous renewer is still started above (it must adopt the mount and arm the fence on - /// every writable open); only the lease thread is conditional. The merged - /// heartbeat renews at `mount_renew_period` — one beat now renews the lease and the floor. - if (store->config.background_watermark) - store->mount_runtime.startBackgroundWorkers(store->config.mount_renew_period); + store->mount_runtime.startBackgroundWorkers(store->config.mount_renew_period); + if (armed) + return; + LOG_WARNING(getLogger("CasPool"), + "Content-addressed mount {}: the mount claim left too little of the {} ms lease to admit a write; " + "the open waits for the lease thread's first renewal", srid, ttl_ms_u); + if (store->mount_runtime.waitUntilArmed(ttl_ms_u)) + return; + /// Joined before the failure propagates, so no renewal of this open runs after it. + store->mount_runtime.stopBackgroundWorkers(); + const String last_failure = store->mount_runtime.lastRenewFailure(); + throw Exception(ErrorCodes::ABORTED, + "CAS mount '{}': no renewal left enough of the {} ms lease to admit a write within one lease after " + "the mount claim; last failed renewal request: {}", + srid, ttl_ms_u, last_failure.empty() ? String("none") : last_failure); } PoolPtr Pool::openForDecommission(BackendPtr backend, PoolConfig config, const String & victim_srid) @@ -1256,6 +1230,7 @@ bool Pool::tryRemountOnce() const String & srid = config.server_root_id; std::string_view step = "entry"; bool succeeded = false; + bool armed = false; uint64_t result_writer_epoch = 0; String error; SCOPE_EXIT( @@ -1277,9 +1252,10 @@ bool Pool::tryRemountOnce() CasEvent event; event.type = CasEventType::MountRemount; event.outcome = succeeded ? "ok" : "failed"; - event.reason = succeeded - ? "whole-chain remount restored Live under a fresh mount incarnation" - : "whole-chain remount returned without restoring Live"; + event.reason = !succeeded + ? "whole-chain remount returned without restoring Live" + : (armed ? "whole-chain remount restored Live under a fresh mount incarnation" + : "whole-chain remount claimed a fresh mount incarnation; the next renewal arms the fence"); event.detail = { {"attempt_no", std::to_string(attempt_no)}, {"step", String{step}}, @@ -1508,7 +1484,7 @@ bool Pool::tryRemountOnce() step = "renewer_install"; mount_runtime.installRenewer(our_uuid, writer_epoch, now_ms); step = "renewer_start"; - uint64_t remount_anchor_boot_ms = mount_runtime.startRenewer(); + const uint64_t remount_anchor_boot_ms = mount_runtime.startRenewer(); /// Re-establish the ref-protocol incarnation BEFORE re-arming the fence. Order is load-bearing: /// Starting the renewer does NOT clear `lost`, so the fence stays closed here and no append/publish can race the @@ -1536,42 +1512,15 @@ bool Pool::tryRemountOnce() if (config.remount_quiesce_hook_for_test) config.remount_quiesce_hook_for_test(); - /// Quiescence may consume most of the new lease. The same renewal-window gate used at startup - /// admits one synchronous re-anchor while at least one physical attempt still fits the old - /// authority window and before the fence is armed. - const uint64_t safety_ms = config.cas_request_budget.lease_safety_margin_ms; - const uint64_t safe_deadline = remount_anchor_boot_ms > std::numeric_limits::max() - ttl_ms - ? std::numeric_limits::max() - safety_ms - : remount_anchor_boot_ms + ttl_ms - safety_ms; - const uint64_t now_boot_ms = mount_runtime.bootMsNow(); - const uint64_t period_ms = static_cast(config.mount_renew_period.count()); - const uint64_t envelope_ms = config.cas_request_budget.attemptEnvelopeMs(); - const uint64_t two_envelopes_ms = envelope_ms > std::numeric_limits::max() / 2 - ? std::numeric_limits::max() : 2 * envelope_ms; - const uint64_t renewal_window_ms = config.background_watermark - ? (two_envelopes_ms > std::numeric_limits::max() - period_ms - ? std::numeric_limits::max() : period_ms + two_envelopes_ms) - : two_envelopes_ms; - /// STRICT, like the open path and `CasMountRuntime::admit`: a horizon that fits exactly still - /// starts a renewal the fence would then refuse. - const bool renewal_window_fits = now_boot_ms <= safe_deadline - && renewal_window_ms < safe_deadline - now_boot_ms; - if (!renewal_window_fits) - { - step = "renewer_redo"; - remount_anchor_boot_ms = mount_runtime.renewRenewerForRemountOnce(); - } - - /// No-throw commit section: arm only after epoch, renewer, recovery cancellation, and ref-runtime - /// quiescence are complete, and only under the arming rule. A reclaim that claimed and did not - /// arm still returns true: its request is acknowledged, and a false would make the loop back off - /// and claim yet another epoch. + /// No-throw commit section. The fence and the lifecycle are published only after epoch, renewer, + /// recovery cancellation and ref-runtime quiescence are complete. A claim whose deadline no longer + /// admits a ref append leaves the fence latched; the lease thread's next renewal arms it. step = "arm_fence"; - const uint64_t deadline_boot_ms = remount_anchor_boot_ms > std::numeric_limits::max() - ttl_ms - ? std::numeric_limits::max() - : remount_anchor_boot_ms + ttl_ms; - if (mount_runtime.armIfAdmissible(deadline_boot_ms)) - step = "publish_live"; + armed = mount_runtime.armIfAdmissible( + remount_anchor_boot_ms > std::numeric_limits::max() - ttl_ms + ? std::numeric_limits::max() + : remount_anchor_boot_ms + ttl_ms); + step = armed ? "publish_live" : "claimed_not_armed"; succeeded = true; return true; } diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp index b6591451516e..566431f4fd76 100644 --- a/src/Disks/tests/gtest_cas_pool.cpp +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -2338,36 +2338,6 @@ class ScopedRemountLogCapture int old_level; }; -class ScopedParkedRenewalLogCapture -{ -public: - ScopedParkedRenewalLogCapture() - : logger(getLogger("CasMountLeaseRenewer")) - , channel(new Poco::StreamChannel(stream)) - , old_channel(logger->getChannel(), /*shared=*/true) - , old_level(logger->getLevel()) - { - logger->setChannel(channel.get()); - logger->setLevel("information"); - } - - ~ScopedParkedRenewalLogCapture() - { - logger->setChannel(old_channel); - logger->setLevel(old_level); - } - - String captured() const { return stream.str(); } - -private: - LoggerPtr logger; - std::ostringstream stream; // STYLE_CHECK_ALLOW_STD_STRING_STREAM - Poco::AutoPtr channel; - /// A real reference (shared=true), so the parked previous channel cannot die while ours is installed. - Poco::AutoPtr old_channel; - int old_level; -}; - size_t countRemountFinalLogs(const String & output) { constexpr std::string_view needle = "CAS whole-chain remount attempt"; @@ -2757,90 +2727,38 @@ TEST(CASMountOpenWaits, FencedPriorReclaimsWithoutAnyWait) << "a certified-dead predecessor needs neither the observation window nor any grace period"; } -/// The open-time publication horizon must reserve TWO attempt envelopes (connect cap included), not -/// two bare attempt timeouts -- a slow connect could otherwise overrun the reservation the horizon -/// check was guarding. `background_watermark` defaults false (not set below), so `CasPool.cpp`'s -/// `renewal_window_ms` ternary takes its no-period branch: `2 * attemptEnvelopeMs()`. The check is also -/// STRICT (refuses equality), matching `CasMountRuntime::admit`. -TEST(CASMountOpenWaits, PublicationHorizonUsesTheEnvelope) -{ - /// Opens with a boot clock costing `per_call_ms` per read (models a faster or slower claim) and - /// returns how many times the mount key was written. attempt 100, cap 100: envelope = - /// 100 + 2*100 = 300, so 2*envelope = 600; the old code reserved 2*attempt = 200. Empirically the - /// claim path's own anchor read and the horizon check's own `now_boot_ms` read are five reads apart, - /// so `remaining = safe_deadline(TTL 1000 - margin 50 = 950) - now = 950 - 5 * per_call_ms`. - const auto mountWriteCount = [](uint64_t per_call_ms) -> uint64_t - { - auto b = std::make_shared(); - Layout l{"p"}; - DB::Cas::tests::seedPoolMetaForRestart(*b); - /// Held in a shared atomic, not a plain local: the hooks below mutate it, and the Pool can - /// outlive this lambda's own stack frame (a background publish holds `shared_from_this()`), so - /// a by-reference capture of a local would dangle. - auto fake_boot = std::make_shared>(0); - PoolPtr store; - store = Pool::open(b, PoolConfig{ - .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "test", - .mount_lease_ttl_ms = std::chrono::milliseconds(1000), - .cas_request_budget = CasRequestBudget{.attempt_timeout_ms = 100, .lease_safety_margin_ms = 50, .connect_timeout_cap_ms = 100}, - .boot_ms_fn = [fake_boot, per_call_ms] - { - return fake_boot->fetch_add(per_call_ms); - }, - .wait_sleep_fn = [fake_boot](uint64_t ms) - { - *fake_boot += ms; - }, - }); - if (!store) - return 0; - return b->putOverwriteCount(l.mountKey("test")) + b->putCount(l.mountKey("test")); - }; - - /// remaining = 945 (per_call_ms=1): both 2*attempt(200) and 2*envelope(600) fit -- two writes (the - /// claim's own reclaim, then the renewer's adopt) and no re-anchor. - EXPECT_EQ(mountWriteCount(1), 2u) << "a horizon that fits both windows must not re-anchor"; - /// remaining = 450 (per_call_ms=100): 2*attempt(200) fits, 2*envelope(600) does not -- the - /// re-anchor costs one extra write. This is the discriminator: reverting the reservation to - /// 2*attempt would make this case behave like the one above (two writes). - EXPECT_EQ(mountWriteCount(100), 3u) << "the old 2*attempt window fit here; only the envelope window must redo"; - /// remaining = 600 (per_call_ms=70) exactly equals 2*envelope: STRICT ("<", not "<=") refuses - /// equality too, so this must also redo -- reverting the strict comparison to "<=" would make this - /// case behave like the fits-both case (two writes). - EXPECT_EQ(mountWriteCount(70), 3u) << "an exact boundary (renewal_window_ms == remaining) must be refused, not accepted"; -} - -/// Same reservation change as `PublicationHorizonUsesTheEnvelope` above, exercised through the remount -/// path's own `renewer_redo` step (`CasPool.cpp` ~1503). Modelled directly on -/// `CASPoolRemount.TheRenewerRedoRenewsOnTheOpenPlane` above: the step's admission refuses a driver that -/// was never parked by a persistent renewal worker, so `background_watermark` must be true and the -/// remount must be driven through `scheduleRemountForTest` (which parks the worker before running it), -/// never through a bare `tryRemountOnce` with no workers -- the direct-driven attempt deadlocks in -/// exactly the way that test's own comment describes. +/// A reclaim arms the fence only when its claim still admits a ref append: more than two envelopes plus +/// the margin of lease, strictly, with no period term. With attempt 100 and cap 100 that is 650 ms of the +/// 1000 ms lease. A quiescence that leaves less reports success at step `claimed_not_armed` with the pool +/// not `Live`; the lease thread's next renewal arms it. TEST(CASPoolRemount, RemountRenewerRedoUsesTheEnvelope) { - /// One successful self-remount whose quiescence costs `quiesce_ms`; returns the conditional - /// mount-slot writes it issued, counted while the remount worker is still held inside the event - /// sink that reported the result (so the renewal worker it un-parks cannot add one). - const auto remountConditionalMountWrites = [](uint64_t quiesce_ms) -> uint64_t + struct AtRemountEvent + { + String step; + PoolLifecycle lifecycle = PoolLifecycle::IdentityLost; + bool may_mutate = true; + }; + /// One reclaim whose quiescence costs `quiesce_ms`; returns what the MountRemount event and the pool + /// show while the lease thread is held inside the sink that reported it. + const auto remountWithQuiesce = [](uint64_t quiesce_ms) -> AtRemountEvent { auto backend = std::make_shared(); - /// Held in a shared atomic, not a plain local: the hooks below mutate it, and the Pool can - /// outlive this lambda's own stack frame (a background publish holds `shared_from_this()`), so - /// a by-reference capture of a local would dangle. + /// Held in shared state: the hooks below mutate it, and the Pool can outlive this lambda's frame. auto fake_boot = std::make_shared>(1'000'000); - /// Heap-owned, not a plain local: declaration order relative to `store` only protects against an - /// ordinary same-thread unwind, not a detached background completion that holds an extra - /// `shared_from_this()` and can still be running on another thread after this call returns. auto committed = std::make_shared(); + auto step = std::make_shared(); auto store = Pool::open(backend, PoolConfig{ .pool_prefix = "remount-renewer-redo-envelope", .server_root_id = "test", .background_watermark = true, - .event_sink = [committed](const CasEvent & event) + .event_sink = [committed, step](const CasEvent & event) { if (event.type == CasEventType::MountRemount && event.outcome == "ok") + { + *step = event.detail.at("step"); committed->arriveAndWait(); + } }, .mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::milliseconds(100), @@ -2858,42 +2776,39 @@ TEST(CASPoolRemount, RemountRenewerRedoUsesTheEnvelope) *fake_boot += quiesce_ms; }, }); - const String mount_key = store->layout().mountKey("test"); - - fenceOutMount(*backend, mount_key); - const uint64_t before = backend->putOverwriteCount(mount_key); - EXPECT_TRUE(store->scheduleRemountForTest()) - << "the remount must be latched with quiesce_ms=" << quiesce_ms; + fenceOutMount(*backend, store->layout().mountKey("test")); + EXPECT_TRUE(store->scheduleRemountForTest()) << "the remount must be latched with quiesce_ms=" << quiesce_ms; committed->waitUntilArrived(); - const uint64_t writes = backend->putOverwriteCount(mount_key) - before; + AtRemountEvent seen{.step = *step, .lifecycle = store->lifecycle(), .may_mutate = store->mayMutate()}; committed->release(); - return writes; + return seen; }; - /// attempt 100, cap 100: envelope = 100 + 2*100 = 300, so period(100) + 2*envelope(600) = 700. A - /// 450 ms quiescence leaves remaining = TTL(1000) - margin(50) - 450 = 500: the old - /// period + 2*attempt (300) window fit that, but the new period + 2*envelope (700) window does not - /// -- so only the envelope-based check must redo (validation: 100 + 600 + 50 = 750 < 1000). - const uint64_t control = remountConditionalMountWrites(0); - EXPECT_GT(remountConditionalMountWrites(450), control) - << "a quiescence that fits the old attempt-only window but not the envelope window must still cost the redo"; - /// A 250 ms quiescence leaves remaining = 950 - 250 = 700, exactly equal to - /// period + 2*envelope (700): STRICT ("<", not "<=") refuses equality too, so this must also - /// redo -- reverting the strict comparison to "<=" would make this case behave like the control. - EXPECT_GT(remountConditionalMountWrites(250), control) - << "an exact boundary (renewal_window_ms == remaining) must be refused, not accepted"; + for (uint64_t quiesce_ms : {0, 300}) + { + const AtRemountEvent seen = remountWithQuiesce(quiesce_ms); + EXPECT_EQ(seen.step, "publish_live") << "quiesce_ms=" << quiesce_ms; + EXPECT_EQ(seen.lifecycle, PoolLifecycle::Live) << "quiesce_ms=" << quiesce_ms; + EXPECT_TRUE(seen.may_mutate) << "quiesce_ms=" << quiesce_ms; + } + /// 350: 650 ms left, exactly the reservation, refused. 450: 550 ms left; two bare attempts plus the + /// margin (250) would fit, two envelopes do not. + for (uint64_t quiesce_ms : {350, 450}) + { + const AtRemountEvent seen = remountWithQuiesce(quiesce_ms); + EXPECT_EQ(seen.step, "claimed_not_armed") << "quiesce_ms=" << quiesce_ms; + EXPECT_EQ(seen.lifecycle, PoolLifecycle::TransientNotLive) << "quiesce_ms=" << quiesce_ms; + EXPECT_FALSE(seen.may_mutate) << "quiesce_ms=" << quiesce_ms; + } } namespace { -/// Stalls the CLAIM ITSELF past the lease TTL, and counts what the open writes afterwards. +/// Ages the claim on its own I/O, and counts what the open writes afterwards. /// -/// The mount key is written twice before the write fence arms: once by `claimMount`'s reclaim, then -/// once by the renewer's adopt -- and the fence's anchor is taken BETWEEN them. So advancing the -/// injected boot clock on the SECOND write models exactly the thing the Phase B redo exists for: the -/// claim's own I/O outliving the lease it is about to arm a fence under. (This used to be modelled by -/// a materialization grace long enough to consume the TTL; that wait is retired, and the guard it -/// motivated is not -- a stalled socket can still outlive a validated request budget.) +/// The mount key is written twice before the open decides whether to arm: once by `claimMount`'s +/// reclaim, then once by the renewer's adopt, and the claim's start is taken BETWEEN them. A hook on +/// the SECOND write therefore ages the claim the open arms from. class StalledMountClaimBackend final : public DB::Cas::InMemoryBackend { public: @@ -2922,9 +2837,8 @@ class StalledMountClaimBackend final : public DB::Cas::InMemoryBackend }; } -/// Phase B startup-arm (spec rev.4, codex round-3 finding 2): a claim path that consumed the lease TTL -/// must force ONE fresh conditional lease write before arming — the fence must never arm from an anchor -/// that has already expired (a successor could have legally reclaimed meanwhile). +/// A claim path that consumed most of the lease must not arm the fence from that claim: the open waits for +/// the lease thread's first renewal, which writes the lease once more and arms from its own start. TEST(CASPool, StartupArmRedoesLeaseWriteWhenTheClaimConsumesTtl) { auto backend = std::make_shared(); @@ -2969,9 +2883,8 @@ TEST(CASPool, StartupArmRedoesLeaseWriteWhenTheClaimConsumesTtl) { return fake_boot_ms->load(); }; - /// The renewer's adopt write stalls for 15 s of boot clock. That consumes the publication horizon - /// (one 10 s cadence plus one 5 s attempt) while leaving one physical attempt admissible inside - /// the old lease's safety window, so the synchronous redo can safely re-anchor. + /// The renewer's adopt write stalls for 15 s of boot clock: 15 s of the lease are left, a ref append + /// needs 2 x 7 s + 2 s = 16 s, so the open waits; the loop's renewal is due at once (10 s cadence). backend->on_second_mount_write = [fake_boot_ms] { *fake_boot_ms += 15'000; @@ -2981,12 +2894,71 @@ TEST(CASPool, StartupArmRedoesLeaseWriteWhenTheClaimConsumesTtl) ASSERT_NE(store, nullptr); ASSERT_EQ(backend->mount_writes.load(), 3) - << "the fixture assumes exactly two mount writes before the redo (the reclaim and the renewer's " - "adopt, with the fence anchor between them); a different sequence would make the stall land " - "somewhere else and this test would stop testing the redo"; + << "the fixture assumes exactly two mount writes before the open decides (the reclaim and the " + "renewer's adopt, with the claim's start between them), then the loop's renewal"; EXPECT_EQ(backend->mount_writes_after_stall.load(), 1) - << "a TTL-consuming claim must be followed by exactly ONE fresh conditional lease write " - "(the re-anchoring redo) before the write fence arms"; + << "a claim that consumed the lease is followed by exactly one renewal before the open returns"; + EXPECT_TRUE(store->mayMutate()) << "the open returns with the fence armed"; +} + +/// The arming rule at an open reserves what a ref append needs: two attempt envelopes, connect cap +/// included, plus the margin, strictly, and no renewal period. With attempt 100 and cap 100 the envelope is +/// 300 ms, so a claim arms the fence only while more than 2 x 300 + 50 = 650 ms of its 1000 ms lease remain. +/// The 340 ms period keeps the lease thread from renewing an armed claim before the count is read. +TEST(CASMountOpenWaits, PublicationHorizonUsesTheEnvelope) +{ + /// Opens over a fenced predecessor whose adopt write ages the claim by `claim_age_ms`, and returns the + /// conditional writes of the mount key made before the open returned. + const auto mountWritesAtOpen = [](uint64_t claim_age_ms) -> int + { + auto backend = std::make_shared(); + DB::Cas::Layout layout("pool"); + DB::Cas::tests::seedPoolMetaForRestart(*backend, "pool"); + backend->mount_key = layout.mountKey("s"); + MountLease prior; + prior.server_uuid = DB::UInt128(0x42); + prior.writer_epoch = 7; + prior.seq = 7; + prior.expires_at_ms = 1; + prior.gc_fenced = true; + prior.write_attempt_id = DB::UInt128{7}; + createObj(*backend, layout.mountKey("s"), encodeMountLease(prior)); + createObj(*backend, layout.epochKey("s"), encodeServerEpoch(ServerEpoch{.next_writer_epoch = 8})); + /// Held in a shared atomic: the Pool can outlive this lambda's frame. + auto fake_boot = std::make_shared>(10'000); + backend->on_second_mount_write = [fake_boot, claim_age_ms] + { + *fake_boot += claim_age_ms; + }; + DB::Cas::PoolConfig cfg; + cfg.pool_prefix = "pool"; + cfg.server_id = DB::UInt128(0x42); + cfg.server_root_id = "s"; + cfg.background_watermark = true; + cfg.mount_lease_ttl_ms = std::chrono::milliseconds(1000); + cfg.mount_renew_period = std::chrono::milliseconds(340); + cfg.cas_request_budget = CasRequestBudget{.attempt_timeout_ms = 100, .lease_safety_margin_ms = 50, .connect_timeout_cap_ms = 100}; + cfg.boot_ms_fn = [fake_boot] + { + return fake_boot->load(); + }; + cfg.wait_sleep_fn = [fake_boot](uint64_t ms) + { + *fake_boot += ms; + }; + auto store = Pool::open(backend, cfg); + EXPECT_TRUE(store && store->mayMutate()) << "claim_age_ms=" << claim_age_ms; + return backend->mount_writes.load(); + }; + + /// 1000 ms left: the claim arms; the reclaim and the adopt only. + EXPECT_EQ(mountWritesAtOpen(0), 2); + /// 700 ms left: the claim arms. A window with the period in it (340 + 600 = 940) would not fit here. + EXPECT_EQ(mountWritesAtOpen(300), 2) << "the arming rule has no period term"; + /// 650 ms left, exactly the reservation: refused, strict like `admit`; the loop's renewal arms. + EXPECT_EQ(mountWritesAtOpen(350), 3) << "an exact boundary must be refused"; + /// 600 ms left: two bare attempts plus the margin (250) would fit, two envelopes (650) do not. + EXPECT_EQ(mountWritesAtOpen(400), 3) << "the reservation counts envelopes, not bare attempts"; } /// ==== What a self-remount may block on ==== @@ -4069,200 +4041,6 @@ TEST(CASPoolRemount, ImmediatePostRemountRenewalFailureIsNotDropped) runtime.finishTeardown(false); } -TEST(CASPoolRemount, StaleRemountAnchorPerformsParkedRedo) -{ - auto backend = std::make_shared(); - /// Held in a shared atomic, not a plain local: `remount_quiesce_hook_for_test` below mutates it, - /// and the Pool can outlive this stack frame (a background publish holds `shared_from_this()`), so - /// a by-reference capture of a local would dangle. - auto fake_boot = std::make_shared>(100); - /// Heap-owned, not a plain local: declaration order relative to the Pool below only protects - /// against an ordinary same-thread unwind, not a detached background completion that holds an - /// extra `shared_from_this()` and can still be running on another thread after this frame returns. - auto committed = std::make_shared(); - PoolConfig config{ - .pool_prefix = "stale-remount-anchor", - .server_root_id = "test", - .background_watermark = true, - .event_sink = [committed](const CasEvent & event) - { - if (event.type == CasEventType::MountRemount && event.outcome == "ok") - committed->arriveAndWait(); - }, - .mount_lease_ttl_ms = std::chrono::milliseconds(1000), - .mount_renew_period = std::chrono::milliseconds(100), - .cas_request_budget = runtimeRenewBudget(), - .boot_ms_fn = [fake_boot] - { - return fake_boot->load(); - }, - .remount_quiesce_hook_for_test = [fake_boot] - { - *fake_boot += 900; - }, - }; - auto store = Pool::open(backend, config); - const String key = store->layout().mountKey("test"); - fenceOutMount(*backend, key); - const uint64_t writes_before = backend->putOverwriteCount(key); - ASSERT_TRUE(store->scheduleRemountForTest()); - committed->waitUntilArrived(); - EXPECT_GE(backend->putOverwriteCount(key), writes_before + 3) - << "claim, renewer start, and the stale-anchor parked redo must all write"; - committed->release(); -} - -TEST(CASPoolRemount, ParkedRedoRecoveryObservabilityPrecedesRemountResult) -{ - auto backend = std::make_shared(); - /// Held in a shared atomic, not a plain local: `remount_quiesce_hook_for_test` below mutates it, - /// and the Pool can outlive this stack frame (a background publish holds `shared_from_this()`), so - /// a by-reference capture of a local would dangle. - auto fake_boot = std::make_shared>(100); - /// Heap-owned, not plain locals: `event_sink` below mutates them, and the Pool can outlive this - /// stack frame (a background publish holds `shared_from_this()`), so a by-reference capture of a - /// local -- including a non-copyable `std::promise` -- would dangle. - auto result_observed = std::make_shared>(); - std::future result_future = result_observed->get_future(); - auto result_published = std::make_shared>(false); - auto events = std::make_shared(); - PoolConfig config{ - .pool_prefix = "parked-redo-recovered-observability", - .server_root_id = "test", - .background_watermark = true, - .event_sink = [result_observed, result_published, events](CasEvent event) - { - const bool final_remount = event.type == CasEventType::MountRemount && event.outcome == "ok"; - events->push(std::move(event)); - if (final_remount && !result_published->exchange(true)) - result_observed->set_value(); - }, - .mount_lease_ttl_ms = std::chrono::milliseconds(1000), - /// 500 with a 700 ms quiescence, so the redo's window (period + attempt timeout = 510) does not - /// fit the 280 ms of safe lease left -- and the reissue the ambiguity needs still does, whatever - /// the engine's jittered backoff draws from its first-reissue range of at most 200 ms. - .mount_renew_period = std::chrono::milliseconds(500), - .cas_request_budget = runtimeRenewBudget(), - .boot_ms_fn = [fake_boot] - { - return fake_boot->load(); - }, - /// `backend` is captured BY VALUE (a copy of the shared_ptr, not the stack slot holding it): - /// the Pool can outlive this frame, so a by-reference capture of the local `shared_ptr` itself - /// would dangle even though the pointee it owns is heap-allocated. - .remount_quiesce_hook_for_test = [fake_boot, backend] - { - *fake_boot += 700; - backend->fault = RuntimeRenewBackend::Fault::ThrowBefore; - }, - }; - auto store = Pool::open(backend, config); - std::weak_ptr store_lifetime = store; - ScopedParkedRenewalLogCapture renewal_logs; - fenceOutMount(*backend, store->layout().mountKey("test")); - ASSERT_TRUE(store->scheduleRemountForTest()); - ASSERT_EQ(result_future.wait_for(std::chrono::seconds(20)), std::future_status::ready); - - const std::vector observed = events->snapshot(); - const auto recovered = std::find_if(observed.begin(), observed.end(), [](const CasEvent & event) - { - return event.type == CasEventType::WatermarkRenew && event.outcome == "recovered"; - }); - const auto remounted = std::find_if(observed.begin(), observed.end(), [](const CasEvent & event) - { - return event.type == CasEventType::MountRemount && event.outcome == "ok"; - }); - ASSERT_NE(recovered, observed.end()); - ASSERT_NE(remounted, observed.end()); - EXPECT_LT(std::distance(observed.begin(), recovered), std::distance(observed.begin(), remounted)); - EXPECT_EQ(recovered->detail.at("remount_attempt_no"), remounted->detail.at("attempt_no")); - EXPECT_EQ(recovered->detail.at("classification"), "committed_after_retry"); - /// The physical retry itself: the ambiguous attempt and the reissue that committed. - EXPECT_EQ(recovered->detail.at("attempts_sent"), "2"); - EXPECT_NE(renewal_logs.captured().find("CAS mount renewal 'test' recovered"), String::npos); - - /// `~Pool` stops and joins both persistent runtime workers. Make that quiescence boundary part of - /// the test, before any event/log capture state referenced by those workers can leave scope. - store.reset(); - EXPECT_TRUE(store_lifetime.expired()); -} - -TEST(CASPoolRemount, ParkedRedoFailureObservabilityPrecedesRemountResult) -{ - auto backend = std::make_shared(); - /// Held in a shared atomic, not a plain local: the hooks below mutate it, and the Pool can outlive - /// this stack frame (a background publish holds `shared_from_this()`), so a by-reference capture - /// of a local would dangle. - auto fake_boot = std::make_shared>(100); - /// Heap-owned, not plain locals: `event_sink` below mutates them, and the Pool can outlive this - /// stack frame (a background publish holds `shared_from_this()`), so a by-reference capture of a - /// local -- including a non-copyable `std::promise` -- would dangle. - auto result_observed = std::make_shared>(); - std::future result_future = result_observed->get_future(); - auto result_published = std::make_shared>(false); - auto events = std::make_shared(); - PoolConfig config{ - .pool_prefix = "parked-redo-failed-observability", - .server_root_id = "test", - .background_watermark = true, - .event_sink = [result_observed, result_published, events](CasEvent event) - { - const bool final_remount = event.type == CasEventType::MountRemount && event.outcome == "failed"; - events->push(std::move(event)); - if (final_remount && !result_published->exchange(true)) - result_observed->set_value(); - }, - .mount_lease_ttl_ms = std::chrono::milliseconds(1000), - .mount_renew_period = std::chrono::milliseconds(100), - .cas_request_budget = runtimeRenewBudget(), - .boot_ms_fn = [fake_boot] - { - return fake_boot->load(); - }, - /// `backend` is captured BY VALUE (a copy of the shared_ptr): the Pool can outlive this frame, - /// so a by-reference capture of the local `shared_ptr` itself would dangle. - .remount_quiesce_hook_for_test = [fake_boot, backend] - { - *fake_boot += 900; - backend->fault = RuntimeRenewBackend::Fault::ThrowBefore; - /// The attempt is admitted 80 ms before its lease-safe bound; spending 90 inside it puts the - /// resolve read past that bound, so the ambiguity is refused instead of reissued. - backend->before_throw = [fake_boot] - { - *fake_boot += 90; - }; - }, - }; - auto store = Pool::open(backend, config); - std::weak_ptr store_lifetime = store; - ScopedParkedRenewalLogCapture renewal_logs; - fenceOutMount(*backend, store->layout().mountKey("test")); - ASSERT_TRUE(store->scheduleRemountForTest()); - ASSERT_EQ(result_future.wait_for(std::chrono::seconds(20)), std::future_status::ready); - - const std::vector observed = events->snapshot(); - const auto failed_renew = std::find_if(observed.begin(), observed.end(), [](const CasEvent & event) - { - return event.type == CasEventType::WatermarkRenew && event.outcome == "failed"; - }); - const auto failed_remount = std::find_if(observed.begin(), observed.end(), [](const CasEvent & event) - { - return event.type == CasEventType::MountRemount && event.outcome == "failed"; - }); - ASSERT_NE(failed_renew, observed.end()); - ASSERT_NE(failed_remount, observed.end()); - EXPECT_LT(std::distance(observed.begin(), failed_renew), std::distance(observed.begin(), failed_remount)); - EXPECT_EQ(failed_renew->detail.at("remount_attempt_no"), failed_remount->detail.at("attempt_no")); - EXPECT_EQ(failed_renew->detail.at("attempts_sent"), "1"); - EXPECT_EQ(failed_renew->detail.at("classification"), "external_lease_deadline"); - EXPECT_NE(renewal_logs.captured().find("CAS mount renewal 'test' fenced"), String::npos); - - /// A ready final-result future proves publication order; destruction additionally proves the - /// background renewal/remount threads are joined before the fixture's captured state is destroyed. - store.reset(); - EXPECT_TRUE(store_lifetime.expired()); -} - TEST(CASPoolRemount, ThrowingEventSinkAfterCommitLeavesRuntimeLive) { auto backend = std::make_shared(); @@ -5487,75 +5265,6 @@ TEST(CASPoolRemount, WholeChainResultsAreNumberedAndStepLabelled) EXPECT_EQ(countRemountFinalLogs(logs.captured()), 2u) << logs.captured(); } -/// The remount's `renewer_redo` step re-anchors the lease BEFORE `armMountFence`, so it runs with the -/// fence still latched lost. Admitted on the mount plane it could only ever give up, and every remount -/// that reached the step would fail -- so it renews on the renewer's open plane instead. -/// -/// Driven the way production reaches the step, which is the only way it CAN be reached: the persistent -/// renewal worker runs, `scheduleRemount` parks it, and the redo is the parked driver's one call. A -/// remount driven directly with no workers leaves that driver dormant, and the step's admission refuses -/// a dormant driver rather than renewing. -/// -/// The step is reached only when quiescence has eaten most of the new lease: with a fresh anchor the -/// renewal window fits and the step is skipped entirely. So the quiesce hook advances the injected boot -/// clock to just inside the safety margin, and the paired run with no quiesce cost is the control that -/// proves the step was reached rather than skipped. -TEST(CASPoolRemount, TheRenewerRedoRenewsOnTheOpenPlane) -{ - /// One successful self-remount whose quiescence costs `quiesce_ms`; returns the conditional - /// mount-slot writes it issued. Counted while the remount worker is still held inside the event - /// sink that reported the result, so the renewal worker it un-parks cannot add one. - const auto remountConditionalMountWrites = [](uint64_t quiesce_ms) -> uint64_t - { - auto backend = std::make_shared(); - /// Held in a shared atomic, not a plain local: the hooks below mutate it, and the Pool can - /// outlive this lambda's own stack frame (a background publish holds `shared_from_this()`), so - /// a by-reference capture of a local would dangle. - auto fake_boot = std::make_shared>(1'000'000); - /// Heap-owned, not a plain local: declaration order relative to `store` only protects against an - /// ordinary same-thread unwind, not a detached background completion that holds an extra - /// `shared_from_this()` and can still be running on another thread after this call returns. - auto committed = std::make_shared(); - auto store = Pool::open(backend, PoolConfig{ - .pool_prefix = "remount-renewer-redo", - .server_root_id = "test", - .background_watermark = true, - .event_sink = [committed](const CasEvent & event) - { - if (event.type == CasEventType::MountRemount && event.outcome == "ok") - committed->arriveAndWait(); - }, - .boot_ms_fn = [fake_boot] - { - return fake_boot->load(); - }, - .wait_sleep_fn = [fake_boot](uint64_t ms) - { - *fake_boot += ms; - }, - .remount_quiesce_hook_for_test = [fake_boot, quiesce_ms] - { - *fake_boot += quiesce_ms; - }, - }); - const String mount_key = store->layout().mountKey("test"); - - fenceOutMount(*backend, mount_key); - const uint64_t before = backend->putOverwriteCount(mount_key); - EXPECT_TRUE(store->scheduleRemountForTest()) - << "the remount must be latched with quiesce_ms=" << quiesce_ms; - committed->waitUntilArrived(); - const uint64_t writes = backend->putOverwriteCount(mount_key) - before; - committed->release(); - return writes; - }; - - /// 27 s of a 30 s lease, against a 2 s safety margin and a window of one renewal period plus one - /// attempt (15 s, since the renewal worker runs here): the window no longer fits. - EXPECT_GT(remountConditionalMountWrites(27'000), remountConditionalMountWrites(0)) - << "a quiescence that consumed the lease must cost one extra lease write -- the redo"; -} - TEST(CASPoolRemount, LeaseLossHasOneOperationalOwner) { auto backend = std::make_shared(); @@ -6821,3 +6530,572 @@ TEST(CASMountRuntime, ARenewalArmsNothingWhileARequestIsPending) runtime.stopBackgroundWorkers(); runtime.finishTeardown(false); } + +namespace +{ +/// Fails the conditional writes of the mount key that `on_mount_write` says to fail, with a timeout raised +/// before the store applied anything. The hook gets the 1-based number of the conditional write of the +/// mount key, counting every one: reclaims, adopts, renewals and a test's own fence-out. +class MountWriteScriptBackend final : public DB::Cas::InMemoryBackend +{ +public: + String mount_key; + std::function on_mount_write; + std::atomic mount_writes{0}; + + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, DB::Cas::TransportAccess & access) override + { + if (key == mount_key && expected_value) + { + const uint32_t write_no = ++mount_writes; + if (on_mount_write && on_mount_write(write_no)) + throw Poco::TimeoutException("injected readiness renewal timeout"); + } + return InMemoryBackend::write(key, bytes, expected_value, access); + } +}; + +const UInt128 kReadyUuid{0x42}; +constexpr uint64_t kReadyClaimBootMs = 10'000; +/// The adopt write or the quiesce takes 20 s: the claim leaves 10 s of the default 30 s lease, and a ref +/// append needs 2 x 7 s + 2 s = 16 s with the default budget. +constexpr uint64_t kReadyClaimAgeMs = 20'000; +constexpr uint64_t kReadyPeriodMs = 10'000; + +/// Pool meta, and a fenced, long-expired predecessor under epoch 7 with the epoch object that minted it. +/// An open as `kReadyUuid` reclaims it at once and writes the mount key exactly twice (the reclaim, then +/// the renewer's adopt) before it decides whether to arm. +void seedFencedPredecessor(DB::Cas::Backend & backend, const String & pool_prefix) +{ + DB::Cas::tests::seedPoolMetaForRestart(backend, pool_prefix); + const Layout layout(pool_prefix); + MountLease prior; + prior.server_uuid = kReadyUuid; + prior.writer_epoch = 7; + prior.seq = 7; + prior.expires_at_ms = 1; + prior.gc_fenced = true; + prior.write_attempt_id = UInt128{7}; + createObj(backend, layout.mountKey("s"), encodeMountLease(prior)); + createObj(backend, layout.epochKey("s"), encodeServerEpoch(ServerEpoch{.next_writer_epoch = 8})); +} + +/// A writable pool with a lease thread, the default lease, period and budget, and a boot clock that only +/// the test and the injected sleeps move. +PoolConfig readinessPoolConfig(const String & pool_prefix, const std::shared_ptr> & fake_boot) +{ + PoolConfig config; + config.pool_prefix = pool_prefix; + config.server_id = kReadyUuid; + config.server_root_id = "s"; + config.background_watermark = true; + config.boot_ms_fn = [fake_boot] + { + return fake_boot->load(); + }; + config.wait_sleep_fn = [fake_boot](uint64_t ms) + { + *fake_boot += ms; + }; + config.retry_sleep_fn = [fake_boot](uint64_t ms) + { + *fake_boot += ms; + }; + return config; +} + +/// What a readiness scenario records on the lease thread. Heap-owned and captured by value: a `Pool` can +/// outlive the test frame. +struct ReadinessSeen +{ + std::thread::id opening_thread; + std::atomic renewal_on_opening_thread{true}; + std::atomic renewal_boot_ms{0}; + std::atomic failed_renewals{0}; + std::atomic request_bound_hit{false}; + std::atomic pool{nullptr}; + std::atomic lifecycle_at_renewal{-1}; + std::atomic may_mutate_at_renewal{-1}; + std::atomic lifecycle_at_arm{-1}; + std::atomic may_mutate_at_arm{-1}; + std::atomic lifecycle_after_arm{-1}; + std::atomic may_mutate_after_arm{-1}; + /// Written on the lease thread before the sink's barrier, read by the test after it. + String remount_outcome; + String remount_step; +}; +} + +/// Spec test 14, cases 1, 6 and 7. An open whose claim is too old returns only after the lease thread's +/// renewal armed the fence; a reclaim whose quiescence aged its claim the same way reports success with +/// the pool not `Live` and the fence latched, and the next renewal arms it. No sampled point sees the +/// fence armed while the pool is not `Live`. +TEST(CASMountRuntime, OpenAndRemountReportReadyOnlyWhenWritable) +{ + /// Case 1. The claim starts at 10 s and the adopt write ends at 30 s. + { + auto fake_boot = std::make_shared>(kReadyClaimBootMs); + auto seen = std::make_shared(); + seen->opening_thread = std::this_thread::get_id(); + auto backend = std::make_shared(); + seedFencedPredecessor(*backend, "ready-open"); + backend->mount_key = Layout("ready-open").mountKey("s"); + backend->on_mount_write = [fake_boot, seen](uint32_t write_no) + { + if (write_no == 2) + { + *fake_boot += kReadyClaimAgeMs; + } + else if (write_no == 3) + { + seen->renewal_boot_ms = fake_boot->load(); + seen->renewal_on_opening_thread = std::this_thread::get_id() == seen->opening_thread; + } + return false; + }; + auto store = Pool::open(backend, readinessPoolConfig("ready-open", fake_boot)); + ASSERT_TRUE(store); + EXPECT_EQ(backend->mount_writes.load(), 3u) << "the reclaim, the adopt and one renewal"; + EXPECT_FALSE(seen->renewal_on_opening_thread.load()) << "the lease thread renews, not the opening thread"; + EXPECT_EQ(seen->renewal_boot_ms.load(), kReadyClaimBootMs + kReadyClaimAgeMs) << "the renewal is sent with no cadence wait"; + EXPECT_EQ(store->lifecycle(), PoolLifecycle::Live); + EXPECT_TRUE(store->mayMutate()); + /// The claim's deadline (40 s) would leave 10 s at 30 s; only the renewal's (60 s) admits the append. + EXPECT_NO_THROW(publishPart(store, "srv/ready-open", "x", "payload")) << "a ref append is admitted when the open returns"; + } + + /// Cases 6 and 7. A fresh open arms at once. Then a reclaim: claim at 10 s, quiescence until 30 s. + { + auto fake_boot = std::make_shared>(kReadyClaimBootMs); + auto seen = std::make_shared(); + auto remounted = std::make_shared(); + auto renewed_after_arm = std::make_shared(); + auto backend = std::make_shared(); + seedFencedPredecessor(*backend, "ready-remount"); + const String mount_key = Layout("ready-remount").mountKey("s"); + backend->mount_key = mount_key; + /// 1, 2: the open's reclaim and adopt. 3: the test's fence-out. 4, 5: the remount's claim and adopt. + /// 6: the renewal that arms. 7: the renewal after it. + backend->on_mount_write = [fake_boot, seen, renewed_after_arm](uint32_t write_no) + { + if (write_no == 6) + { + Pool * pool = seen->pool.load(); + seen->renewal_boot_ms = fake_boot->load(); + seen->lifecycle_at_renewal = static_cast(pool->lifecycle()); + seen->may_mutate_at_renewal = pool->mayMutate(); + } + else if (write_no == 7) + { + Pool * pool = seen->pool.load(); + seen->lifecycle_after_arm = static_cast(pool->lifecycle()); + seen->may_mutate_after_arm = pool->mayMutate(); + renewed_after_arm->arriveAndWait(); + } + return false; + }; + PoolConfig config = readinessPoolConfig("ready-remount", fake_boot); + config.event_sink = [seen, remounted](const CasEvent & event) + { + if (event.type != CasEventType::MountRemount) + return; + seen->remount_outcome = event.outcome; + seen->remount_step = event.detail.at("step"); + remounted->arriveAndWait(); + }; + config.remount_quiesce_hook_for_test = [fake_boot] + { + *fake_boot += kReadyClaimAgeMs; + }; + auto store = Pool::open(backend, config); + ASSERT_TRUE(store); + ASSERT_EQ(backend->mount_writes.load(), 2u) << "a fresh claim arms at once"; + seen->pool = store.get(); + /// Runs inside the arm, after the new generation and before `Live` and the open fence. It also makes + /// the next renewal due at once, so its request shows the state the arm left. + store->setArmMountFenceInterpositionHookForTest([seen, fake_boot] + { + Pool * pool = seen->pool.load(); + seen->lifecycle_at_arm = static_cast(pool->lifecycle()); + seen->may_mutate_at_arm = pool->mayMutate(); + *fake_boot += kReadyPeriodMs; + }); + const uint64_t succeeded_before = eventCount(ProfileEvents::CASRemountSucceeded); + fenceOutMount(*backend, mount_key); + ASSERT_TRUE(store->scheduleRemountForTest()); + + remounted->waitUntilArrived(); + EXPECT_EQ(seen->remount_outcome, "ok"); + EXPECT_EQ(seen->remount_step, "claimed_not_armed"); + EXPECT_EQ(eventCount(ProfileEvents::CASRemountSucceeded), succeeded_before + 1); + EXPECT_EQ(store->lifecycle(), PoolLifecycle::TransientNotLive) << "case 6: not Live after the reclaim"; + EXPECT_FALSE(store->mayMutate()) << "case 6: the fence stays latched"; + remounted->release(); + + renewed_after_arm->waitUntilArrived(); + ASSERT_EQ(backend->mount_writes.load(), 7u); + EXPECT_EQ(seen->renewal_boot_ms.load(), kReadyClaimBootMs + kReadyClaimAgeMs) << "the loop renewed at once"; + /// Case 7: at every sampled point, an armed fence comes with `Live`. + EXPECT_EQ(seen->lifecycle_at_renewal.load(), static_cast(PoolLifecycle::TransientNotLive)); + EXPECT_EQ(seen->may_mutate_at_renewal.load(), 0) << "the renewal is sent under the latched fence"; + EXPECT_EQ(seen->lifecycle_at_arm.load(), static_cast(PoolLifecycle::TransientNotLive)); + EXPECT_EQ(seen->may_mutate_at_arm.load(), 0) << "inside the arm, before `Live`, the fence still refuses"; + EXPECT_EQ(seen->lifecycle_after_arm.load(), static_cast(PoolLifecycle::Live)); + EXPECT_EQ(seen->may_mutate_after_arm.load(), 1) << "after the arm: `Live` and writable"; + renewed_after_arm->release(); + } +} + +/// Spec test 14, case 2. Every readiness renewal fails: the open waits one lease on the fence clock, then +/// stops and joins the lease thread and fails, naming the last failed request. +TEST(CASMountRuntime, AnOpenWhoseReadinessRenewalKeepsFailingFailsAfterOneTtl) +{ + /// Longer than the largest spacing draw, so a failed request is retried at once. + constexpr uint64_t failed_request_ms = 1'300; + /// Far above the 24 failures one lease takes; reached only if the open never gives up. + constexpr uint32_t max_failed_renewals = 10'000; + auto fake_boot = std::make_shared>(kReadyClaimBootMs); + auto seen = std::make_shared(); + auto exits = std::make_shared(); + auto backend = std::make_shared(); + seedFencedPredecessor(*backend, "ready-fail"); + backend->mount_key = Layout("ready-fail").mountKey("s"); + backend->on_mount_write = [fake_boot, seen](uint32_t write_no) + { + if (write_no == 2) + *fake_boot += kReadyClaimAgeMs; + if (write_no < 3) + return false; + if (seen->failed_renewals.load() >= max_failed_renewals) + { + seen->request_bound_hit = true; + return false; + } + ++seen->failed_renewals; + *fake_boot += failed_request_ms; + return true; + }; + PoolConfig config = readinessPoolConfig("ready-fail", fake_boot); + config.worker_factory = [exits](std::function worker_body) + { + return ThreadFromGlobalPool([exits, body = std::move(worker_body)] + { + body(); + exits->recordExit(); + }); + }; + const uint64_t lost_before = eventCount(ProfileEvents::CASMountLeaseLost); + const uint64_t expired_before = eventCount(ProfileEvents::CASMountLeaseExpired); + + int code = 0; + String message; + try + { + (void)Pool::open(backend, config); + ADD_FAILURE() << "an open whose readiness renewal never succeeds must fail"; + } + catch (const DB::Exception & e) + { + code = e.code(); + message = e.message(); + } + + EXPECT_EQ(code, DB::ErrorCodes::ABORTED) << message; + EXPECT_NE(message.find("injected readiness renewal timeout"), String::npos) << "the failure names the last request: " << message; + EXPECT_FALSE(seen->request_bound_hit.load()) << "the open never gave up"; + /// The wait starts at 30 s or later and ends once the fence clock passes one lease after its start: + /// 23 failures reach 59.9 s, the 24th 61.2 s. + EXPECT_GE(seen->failed_renewals.load(), 24u) << "the open waited one lease, not until a lease bound"; + EXPECT_EQ(exits->count(), 1u) << "the lease thread was joined before the open failed"; + EXPECT_EQ(eventCount(ProfileEvents::CASMountLeaseLost), lost_before) << "a failed open is not a lease loss"; + EXPECT_EQ(eventCount(ProfileEvents::CASMountLeaseExpired), expired_before); +} + +/// Ruling O1: a writable open with no lease thread whose claim does not admit a ref append fails at once +/// with a retryable error, and sends no renewal. +TEST(CASPool, AnOpenWithoutALeaseThreadFailsWhenItsClaimIsTooOld) +{ + auto fake_boot = std::make_shared>(kReadyClaimBootMs); + auto backend = std::make_shared(); + seedFencedPredecessor(*backend, "ready-no-thread"); + backend->mount_key = Layout("ready-no-thread").mountKey("s"); + backend->on_mount_write = [fake_boot](uint32_t write_no) + { + if (write_no == 2) + *fake_boot += kReadyClaimAgeMs; + return false; + }; + PoolConfig config = readinessPoolConfig("ready-no-thread", fake_boot); + config.background_watermark = false; + const uint64_t attempts_before = eventCount(ProfileEvents::CASMountRenewalAttempts); + + int code = 0; + String message; + try + { + (void)Pool::open(backend, config); + ADD_FAILURE() << "an open with no lease thread and a claim too old must fail"; + } + catch (const DB::Exception & e) + { + code = e.code(); + message = e.message(); + } + + EXPECT_EQ(code, DB::ErrorCodes::ABORTED) << message; + EXPECT_NE(message.find("no lease thread"), String::npos) << message; + EXPECT_EQ(eventCount(ProfileEvents::CASMountRenewalAttempts), attempts_before) << "nothing renewed the claim"; +} + +namespace +{ +/// What the open's wait and the reclaim saw when the readiness renewal met a definitive answer. +struct DefinitiveReadinessSeen +{ + bool armed = false; + uint32_t puts_at_first_reclaim = 0; + String admit_at_first_reclaim = "unobserved"; + bool may_mutate_at_first_reclaim = true; + uint32_t reclaims = 0; + uint64_t boot_before_wait = 0; + uint64_t boot_after_wait = 0; + ReadinessView after; + uint64_t live_writer_epoch = 0; + uint64_t requested_generation = 0; + uint64_t lease_lost = 0; +}; + +/// The open's shape at the runtime level, with a lease of 1000 ms: the claim starts at 100 ms and the open +/// decides at 1070 ms, with 30 ms left and 40 ms needed. The slot is GC-fenced before the loop starts, so the +/// readiness renewal gets a definitive answer. The reclaim either claims epoch 2 and arms, or fails each +/// attempt and moves the fence clock 600 ms. +void runDefinitiveAnswerDuringReadiness(bool reclaim_arms, DefinitiveReadinessSeen & seen) +{ + const Layout layout(reclaim_arms ? "runtime-readiness-definitive-reclaimed" : "runtime-readiness-definitive-failing"); + const UInt128 uuid{1}; + uint64_t wall_ms = 1000; + std::atomic boot_ms{100}; + std::atomic puts{0}; + std::atomic reclaims{0}; + CasMountRuntime * runtime_ptr = nullptr; + const uint64_t lost_before = eventCount(ProfileEvents::CASMountLeaseLost); + CasEventSink sink; + auto backend = std::make_shared(); + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, + MountClaimResult::Claimed); + + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{ + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .background_watermark = true, + .boot_ms_fn = [&] { return boot_ms.load(); }}, + "test", sink, runtimeRenewBudget(), [&] + { + CasMountRuntime & reclaiming = *runtime_ptr; + if (++reclaims == 1) + { + seen.puts_at_first_reclaim = puts.load(); + seen.admit_at_first_reclaim = expiryAdmitName(reclaiming.admit(reclaiming.fenceGeneration(), 0)); + seen.may_mutate_at_first_reclaim = reclaiming.mayMutate(); + } + reclaiming.beginReclaim(); + if (!reclaim_arms) + { + boot_ms.fetch_add(600); + return false; + } + if (claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 2, wall_ms, 1000).kind + != MountClaimResult::Claimed) + return false; + reclaiming.installRenewer(uuid, 2, [&] { return wall_ms; }); + const uint64_t fresh_anchor = reclaiming.startRenewer(); + reclaiming.setLiveWriterEpoch(2); + /// On a regression that does not arm, the clock passes the open's bound and the arm's + /// notification ends the wait, so the test fails instead of hanging. + if (!reclaiming.armIfAdmissible(fresh_anchor + 1000)) + boot_ms.fetch_add(2000); + return true; + }); + CasMountRuntime & runtime = *runtime_holder; + runtime_ptr = &runtime; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + boot_ms = 1070; + ASSERT_FALSE(runtime.armIfAdmissible(anchor + 1000)) << "the claim leaves 30 ms, a ref append needs 40"; + fenceOutMount(*backend, layout.mountKey("test")); + backend->on_put = [&](uint32_t put_no) + { + ++puts; + /// A second renewal before any reclaim is the regression this test names. Fail it and move the + /// clock past the open's bound: the failure notifies the wait, which then gives up. + if (put_no >= 2 && reclaims.load() == 0) + { + boot_ms.fetch_add(2000); + return true; + } + return false; + }; + runtime.startBackgroundWorkers(std::chrono::milliseconds(300)); + + seen.boot_before_wait = boot_ms.load(); + seen.armed = runtime.waitUntilArmed(1000); + seen.boot_after_wait = boot_ms.load(); + + /// The join orders every write of the lease thread before the reads below. + runtime.stopBackgroundWorkers(); + seen.after = readinessViewOf(runtime); + seen.reclaims = reclaims.load(); + seen.live_writer_epoch = runtime.liveWriterEpoch(); + seen.requested_generation = runtime.remountRequestedGenerationForTest(); + seen.lease_lost = eventCount(ProfileEvents::CASMountLeaseLost) - lost_before; + runtime.finishTeardown(false); +} + +/// Throws a `LOGICAL_ERROR` from the renewal's conditional write of the mount key once `failing` is set: the +/// renewal hands it to the loop, which leaves through its own error path. +class LeaseLoopFailureBackend final : public DB::Cas::tests::CountingBackend +{ +public: + std::atomic failing{false}; + std::atomic failures{0}; + + std::expected write(const String & key, const String & bytes, + const std::optional & expected_value, DB::Cas::TransportAccess & access) override + { + if (failing.load() && expected_value && key.ends_with("/mount")) + { + /// Bounds a retry of the failure, should the engine ever retry one: the renewal then commits, + /// the fence arms and the test fails on `armed` instead of hanging. + if (++failures >= kReadinessRequestBound) + failing = false; + throw DB::Exception(DB::ErrorCodes::LOGICAL_ERROR, "injected lease-loop failure"); + } + return DB::Cas::tests::CountingBackend::write(key, bytes, expected_value, access); + } +}; + +struct LeaseThreadEndSeen +{ + uint32_t exits = 0; + uint32_t failures = 0; + bool armed = true; + uint64_t boot_before_wait = 0; + uint64_t boot_after_wait = 0; + bool may_mutate = true; +}; + +/// The open's shape with a 1000 ms lease, as above; the readiness renewal's first request throws a +/// `LOGICAL_ERROR`. After the thread is gone, every read of the fence clock moves it 600 ms, so the wait's +/// bound is reached on that clock. +void runLeaseThreadEndsOnItsOwn(LeaseThreadEndSeen & seen) +{ + const Layout layout("runtime-readiness-thread-ended"); + const UInt128 uuid{1}; + uint64_t wall_ms = 1000; + std::atomic boot_ms{100}; + std::atomic thread_gone{false}; + WorkerExitLatch exits; + RuntimeWorkerFactory factory = [&](std::function worker_body) + { + return ThreadFromGlobalPool([&, body = std::move(worker_body)] + { + body(); + exits.recordExit(); + }); + }; + CasEventSink sink; + auto backend = std::make_shared(); + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, + MountClaimResult::Claimed); + + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{ + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .background_watermark = true, + .boot_ms_fn = [&] { return thread_gone.load() ? boot_ms.fetch_add(600) + 600 : boot_ms.load(); }, + .worker_factory = factory}, + "test", sink, runtimeRenewBudget(), [] { return false; }); + CasMountRuntime & runtime = *runtime_holder; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + boot_ms = 1070; + ASSERT_FALSE(runtime.armIfAdmissible(anchor + 1000)); + backend->failing = true; + runtime.startBackgroundWorkers(std::chrono::milliseconds(300)); + + ASSERT_TRUE(exits.waitForAtLeast(1)) << "the loop must leave through its own error path"; + thread_gone = true; + seen.boot_before_wait = boot_ms.load(); + seen.armed = runtime.waitUntilArmed(1000); + seen.boot_after_wait = boot_ms.load(); + seen.may_mutate = runtime.mayMutate(); + seen.exits = static_cast(exits.count()); + seen.failures = backend->failures.load(); + runtime.stopBackgroundWorkers(); + runtime.finishTeardown(false); +} +} + +/// The open's readiness renewal meets a definitive answer: the renewal is terminal and raises one remount +/// generation, the fence stays latched, and the loop reclaims before any further renewal. The open's wait +/// ends armed when the reclaim arms under the new epoch, and gives up after one lease on the fence clock when +/// every reclaim fails. +TEST(CASMountRuntime, ADefinitiveAnswerDuringReadinessIsServedByAReclaim) +{ + DefinitiveReadinessSeen reclaimed; + ASSERT_NO_FATAL_FAILURE(runDefinitiveAnswerDuringReadiness(/*reclaim_arms=*/true, reclaimed)); + EXPECT_TRUE(reclaimed.armed) << "the reclaim armed the fence"; + EXPECT_EQ(reclaimed.puts_at_first_reclaim, 1u) << "the loop reclaimed before any further renewal"; + EXPECT_EQ(reclaimed.admit_at_first_reclaim, "LostOrRearmed") << "no write is admitted before the arm"; + EXPECT_FALSE(reclaimed.may_mutate_at_first_reclaim); + EXPECT_EQ(reclaimed.reclaims, 1u); + EXPECT_STREQ(reclaimed.after.admit, "Ok"); + EXPECT_TRUE(reclaimed.after.ref_append_ok); + EXPECT_EQ(reclaimed.after.lifecycle, PoolLifecycle::Live); + EXPECT_EQ(reclaimed.live_writer_epoch, 2u) << "armed under the new epoch"; + EXPECT_EQ(reclaimed.requested_generation, 1u) << "one definitive answer, one remount generation"; + EXPECT_EQ(reclaimed.lease_lost, 1u) << "the definitive answer counts one lease loss; the open's latch none"; + + DefinitiveReadinessSeen failing; + ASSERT_NO_FATAL_FAILURE(runDefinitiveAnswerDuringReadiness(/*reclaim_arms=*/false, failing)); + EXPECT_FALSE(failing.armed); + EXPECT_GE(failing.boot_after_wait, failing.boot_before_wait + 1000) << "the wait gave up after one lease on the fence clock"; + EXPECT_EQ(failing.puts_at_first_reclaim, 1u); + EXPECT_EQ(failing.admit_at_first_reclaim, "LostOrRearmed"); + EXPECT_GE(failing.reclaims, 1u); + EXPECT_STREQ(failing.after.admit, "LostOrRearmed") << "nothing armed"; + EXPECT_FALSE(failing.after.may_mutate); + EXPECT_EQ(failing.live_writer_epoch, 0u); + EXPECT_EQ(failing.requested_generation, 1u); + EXPECT_EQ(failing.lease_lost, 1u); +} + +#ifndef DEBUG_OR_SANITIZER_BUILD +/// The lease thread leaves through its own error path while the open waits. The fence stays latched and the +/// wait returns false; nothing notifies it, so it ends at its bound of one lease on the fence clock. +TEST(CASMountRuntime, AnOpenFailsWhenTheLeaseThreadEndedOnItsOwn) +{ + LeaseThreadEndSeen seen; + ASSERT_NO_FATAL_FAILURE(runLeaseThreadEndsOnItsOwn(seen)); + EXPECT_EQ(seen.exits, 1u); + EXPECT_EQ(seen.failures, 1u) << "the first readiness request reached the loop's error path"; + EXPECT_FALSE(seen.armed); + EXPECT_GE(seen.boot_after_wait, seen.boot_before_wait + 1000) << "the wait ended at its bound on the fence clock"; + EXPECT_FALSE(seen.may_mutate) << "the fence stays latched"; +} +#else +/// A `LOGICAL_ERROR` aborts a debug or sanitizer build where it is constructed, so the loop's error path is +/// reached only in a release build. Here the scenario must end the process. +TEST(CASMountRuntimeDeathTest, AnOpenFailsWhenTheLeaseThreadEndedOnItsOwn) +{ + EXPECT_DEATH( + { + LeaseThreadEndSeen seen; + runLeaseThreadEndsOnItsOwn(seen); + }, + "injected lease-loop failure"); +} +#endif From 78ae6eb32a041483f33095ba0b6518d6fef35723 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 07:01:20 +0200 Subject: [PATCH 11/37] Describe CAS mount readiness through the lease thread An open or a reclaim whose claim leaves too little lease for a ref append keeps the fence latched until a renewal arms it. Document the remount step claimed_not_armed and what ends a renewal early. Co-Authored-By: Claude Opus 5.5 --- .../cas/architecture/mounts-and-leases.md | 22 ++++++++++++++----- docs/en/antalya/cas/operations/debugging.md | 6 +++-- docs/en/antalya/cas/operations/monitoring.md | 7 ++++-- 3 files changed, 25 insertions(+), 10 deletions(-) diff --git a/docs/en/antalya/cas/architecture/mounts-and-leases.md b/docs/en/antalya/cas/architecture/mounts-and-leases.md index 1125f258b8c2..03217f64aa1e 100644 --- a/docs/en/antalya/cas/architecture/mounts-and-leases.md +++ b/docs/en/antalya/cas/architecture/mounts-and-leases.md @@ -112,8 +112,9 @@ only on one of these: absent object, or a request the store refuses on a clear attempt; - a stop or a remount request; - a deterministic local failure; -- a lifecycle other than `Live`; -- a lost fence. +- a terminal lifecycle, which includes a published `FORGET` intent. + +A lost fence alone does not end it: the renewal that makes a mount ready runs under a latched fence. A terminal renewer cannot mint another body or publish a clean farewell. Owner cancellation before any request is the only `NotAttempted` result and leaves clean release possible. Cancellation after a @@ -239,9 +240,9 @@ The in-process `PoolLifecycle` runtime, by contrast, is a literal enum (`CasMoun ```mermaid stateDiagram-v2 [*] --> Live: Pool constructed, fence unarmed - Live --> Live: mountWritable arms the fence + Live --> Live: the open's claim or first renewal arms the fence Live --> TransientNotLive: terminal renewal result, tripMountLost, lost=true - TransientNotLive --> Live: self-remount succeeds with a fresh epoch + TransientNotLive --> Live: the reclaim or the next renewal arms the fence under a fresh epoch TransientNotLive --> TransientNotLive: probe inconclusive, retry with backoff TransientNotLive --> IdentityLost: pool meta and owner both authoritatively absent TransientNotLive --> VanishedReplaced: foreign pool_id observed @@ -260,11 +261,20 @@ under a live mount is an operator-level event. **Writable open** runs in a strict order: bootstrap-residual proof, capability probe under a random per-mount prefix, pool-meta create-or-validate, `validateServerRootId`, owner claim, `allocateWriterEpoch`, mount claim and synchronous renewer start, arm the fence, then start the -runtime-owned lease thread before the writable pool becomes externally visible. If the claim consumed -the TTL, one fresh synchronous renewal re-anchors the deadline before the fence is armed. +runtime-owned lease thread before the writable pool becomes externally visible. Failure to start the lease thread closes the fence and fails the writable open. No incident path constructs a thread. +**Readiness.** An open arms the fence from its claim only when the claim's deadline, its start plus the +TTL, still leaves room for a ref append: `2 × envelope + margin` (16 s with the defaults). Otherwise the +fence stays latched, the lease thread renews at once, and the first renewal whose own deadline leaves that +room arms the fence; then the open returns. A renewal that commits with less room arms nothing, and the +next one follows at once. If no renewal arms the fence within one TTL, the open stops and joins the lease +thread and fails with `ABORTED`, naming the last failed renewal request. A reclaim follows the same rule: a +claim with too little lease left keeps the pool `TransientNotLive` with the fence latched, and the next +renewal arms the fence and reports `Live`. The `mount_remount` row of such a reclaim has `outcome = 'ok'` +and `step = 'claimed_not_armed'`. + One lease thread per writable mount renews the lease and runs the self-remount, one after the other. `scheduleRemount` increments a requested-generation latch and wakes the thread; a pending request also ends a renewal in progress. While the thread runs, only it replaces, starts or resets the renewer. diff --git a/docs/en/antalya/cas/operations/debugging.md b/docs/en/antalya/cas/operations/debugging.md index 7fbdaa5f6680..1050fa4e9599 100644 --- a/docs/en/antalya/cas/operations/debugging.md +++ b/docs/en/antalya/cas/operations/debugging.md @@ -150,8 +150,10 @@ follows: differs by classification, and only `external_lease_deadline` and `request_deadline` are about a deadline at all. - A following `mount_remount` row names the whole-chain `attempt_no` and final `step`. An `ok` row - restored `Live` under the reported fresh `writer_epoch`; a `failed` row's `step` and optional - `error` identify where that whole-chain attempt stopped. + with `step = 'publish_live'` restored `Live` under the reported fresh `writer_epoch`; with + `step = 'claimed_not_armed'` the claim succeeded with too little lease to admit a write, and the next + renewal arms the fence and reports `Live`. A `failed` row's `step` and optional `error` identify where + that whole-chain attempt stopped. Use deltas of the mount counters from [monitoring](/antalya/cas/operations/monitoring#mount-renewal-remount-counters) to check completeness: diff --git a/docs/en/antalya/cas/operations/monitoring.md b/docs/en/antalya/cas/operations/monitoring.md index 3c6b4a74d8dc..a749b1f91f46 100644 --- a/docs/en/antalya/cas/operations/monitoring.md +++ b/docs/en/antalya/cas/operations/monitoring.md @@ -62,8 +62,8 @@ window and correlate them with the `server_root_id` in `system.cas_log`. | `CASMountRenewalDeadlineExceeded` | One per logical renewal stopped by the external lease-safety deadline | The last confirmed lease no longer left enough safe time; this is narrower than request-budget exhaustion. Only the bounded renewals (startup, remount and direct) can reach it; the background renewal does not | | `CASMountLeaseExpired` | One per renewal that restored a lease that had expired | Moves at the restore, not when the lease expires. While it is expired, `system.cas_mounts` shows `lifecycle_reason = 'lease_expired'` and writes are refused; the `watermark_renew` row of the restoring renewal carries `expired_ms` | | `CASRemountAttempts` | One per invocation of the existing whole-chain remount attempt | Includes both successful and failed attempts | -| `CASRemountSucceeded` | One per whole-chain attempt that restored `Live` under a fresh writer epoch | Must be a subset of `CASRemountAttempts` | -| `CASRemountFailed` | One per whole-chain attempt that returned without restoring `Live` | Includes a named step exception or a step that returned transiently | +| `CASRemountSucceeded` | One per whole-chain attempt that claimed the mount under a fresh writer epoch | Must be a subset of `CASRemountAttempts`. Includes a reclaim whose claim left too little lease to arm the fence (`step = 'claimed_not_armed'`); the next renewal arms it | +| `CASRemountFailed` | One per whole-chain attempt that returned without claiming the mount | Includes a named step exception or a step that returned transiently | `CASMountLeaseLost` complements those nine counters. It increments exactly once per operational `Live -> TransientNotLive` recovery generation: either the initiating external loss or the first @@ -90,6 +90,9 @@ SETTINGS system_events_show_zero_values = 1; Ordinary first-attempt success produces no row. Every `mount_remount` attempt produces one final row with outcome `ok` or `failed` and details `attempt_no`, `step`, `server_root_id`, optional `writer_epoch`, and optional `error`. +An `ok` row has `step = 'publish_live'` when the reclaim armed the fence and reported `Live`, or +`step = 'claimed_not_armed'` when its claim left too little lease to admit a write; the pool then stays +`not_live` until the next renewal arms the fence. Default-level text logging is bounded per logical operation. A renewal logs nothing until it ends, and nothing at all when it succeeds on its first request. A renewal that needed a retry, was settled From 30ff8c80d7927afab0b81bfe110b77b0b582d153 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 07:26:01 +0200 Subject: [PATCH 12/37] Run the CAS readiness death test in a fresh process and pin the arm order The death-test child starts a lease thread, which a forked child of a runner with idle pool workers may never run; re-execute instead. The arm interposition hook now runs between the `Live` report and the fence opening, so a test can see their order. The failed-readiness texts name what ended the wait. Co-Authored-By: Claude Opus 5.5 --- .../ContentAddressed/Pool/CasMountRuntime.cpp | 4 +- .../ContentAddressed/Pool/CasMountRuntime.h | 7 +++- .../ContentAddressed/Pool/CasPool.cpp | 9 ++++- src/Disks/tests/gtest_cas_pool.cpp | 40 ++++++++++--------- 4 files changed, 36 insertions(+), 24 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index b23e58fa1201..1a36d953ed29 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -277,10 +277,10 @@ void CasMountRuntime::armFence(uint64_t deadline_boot_ms, bool report_live) /// A fresh lease incarnation is a fresh generation too: a durable-effect caller admitted under the /// PRIOR incarnation must re-check and abort rather than ride this re-arm through. fence_generation.fetch_add(1, std::memory_order_acq_rel); - if (arm_mount_fence_interposition_hook_for_test) - arm_mount_fence_interposition_hook_for_test(); if (report_live) noteRemounted(); + if (arm_mount_fence_interposition_hook_for_test) + arm_mount_fence_interposition_hook_for_test(); /// Open the gate LAST. A caller that observes `lost == false` with acquire semantics must also see /// the fresh generation; publishing the latch first exposes one admission window in which the dead /// generation looks live again. diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index 93b30c1a9cf4..3dbbeeea37f5 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -179,10 +179,13 @@ class CasMountRuntime /// rule holds; otherwise latch the fence. Returns whether it armed. bool armIfAdmissible(uint64_t deadline_boot_ms); /// Blocks until the fence is armed. Gives up after `timeout_ms` on the fence clock, on a stop, on a - /// terminal lifecycle, or when no lease thread runs. Returns whether the fence is armed. + /// terminal lifecycle, or when the lease thread was never started or was stopped. A lease thread that + /// ended on its own does not end the wait early: it waits out `timeout_ms`. Returns whether the fence + /// is armed. bool waitUntilArmed(uint64_t timeout_ms) const; /// Test-only interposition at the publication boundary between the re-armed generation and the - /// live fence. A caller admitted from this hook must be refused: the old generation is already + /// live fence: after the new generation and, for a production arm, the `Live` report, before `lost` + /// is cleared. A caller admitted from this hook must be refused: the old generation is already /// dead, while the new generation is not live until `lost` is cleared. Through `armIfAdmissible` /// it runs with `driver_mutex` held. void setArmMountFenceInterpositionHookForTest(std::function hook) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp index 4e8d25a481eb..67f7bbe7e00c 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp @@ -890,9 +890,14 @@ void Pool::mountWritable(PoolPtr & store, UInt128 our_uuid, MountClaimPolicy pol /// Joined before the failure propagates, so no renewal of this open runs after it. store->mount_runtime.stopBackgroundWorkers(); const String last_failure = store->mount_runtime.lastRenewFailure(); + if (store->mount_runtime.remountTerminal()) + throw Exception(ErrorCodes::ABORTED, + "CAS mount '{}': the pool became terminal while the open waited for a renewal to arm the fence; " + "last failed renewal request: {}", + srid, last_failure.empty() ? String("none") : last_failure); throw Exception(ErrorCodes::ABORTED, - "CAS mount '{}': no renewal left enough of the {} ms lease to admit a write within one lease after " - "the mount claim; last failed renewal request: {}", + "CAS mount '{}': no renewal armed the fence while the open waited, up to {} ms on the fence clock; " + "last failed renewal request: {}", srid, ttl_ms_u, last_failure.empty() ? String("none") : last_failure); } diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp index 566431f4fd76..7b57f974b8e1 100644 --- a/src/Disks/tests/gtest_cas_pool.cpp +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -4224,7 +4224,7 @@ TEST(CASPool, DisabledBackgroundDoesNotReserveRenewalCadence) auto store = Pool::open(backend, config); const String key = store->layout().mountKey("test"); EXPECT_EQ(backend->putOverwriteCount(key), 1u) - << "a disabled worker cadence must not force a synchronous startup redo"; + << "an open with no lease thread whose claim admits a ref append writes the lease only once"; } TEST(CASPool, DeterministicWorkerFailureFencesWithoutWaitingForCadence) @@ -6122,7 +6122,7 @@ TEST(CASMountRuntime, AReadinessRenewalArmsOnlyWithRoomForARefAppend) const uint64_t anchor = runtime.startRenewer(); ASSERT_EQ(anchor, kExpiryClaimBootMs); - /// Case 5, before the thread starts: the claim leaves 30 ms of its lease, a ref append needs 40. + /// Before the thread starts: the claim leaves 30 ms of its lease, a ref append needs 40. boot_ms = kReadinessDecideBootMs; const uint64_t generation_before = runtime.fenceGeneration(); EXPECT_FALSE(runtime.armIfAdmissible(anchor + kExpiryTtlMs)) << "a claim without room for a ref append arms nothing"; @@ -6167,12 +6167,12 @@ TEST(CASMountRuntime, AReadinessRenewalArmsOnlyWithRoomForARefAppend) FAIL() << "no renewal armed the fence; request bound hit: " << request_bound_hit.load(); } - /// Case 5, while the readiness renewal itself is sent. + /// While the readiness renewal itself is sent. EXPECT_STREQ(during_first_put.admit, "LostOrRearmed") << "the latched fence refuses a write while the renewal is sent"; EXPECT_FALSE(during_first_put.may_mutate); EXPECT_FALSE(during_first_put.expired); - /// Case 3: the first renewal started at 129.97 s and committed at 161.17 s, past its own start + TTL. + /// A stale success: the first renewal started at 129.97 s and committed at 161.17 s, past its own start + TTL. EXPECT_STREQ(after_stale.admit, "LostOrRearmed") << "a stale success arms nothing"; EXPECT_EQ(after_stale.lifecycle, PoolLifecycle::Live); EXPECT_EQ(after_stale.generation, latched_generation); @@ -6183,7 +6183,7 @@ TEST(CASMountRuntime, AReadinessRenewalArmsOnlyWithRoomForARefAppend) EXPECT_EQ(admitted_at[0], kReadinessDecideBootMs) << "the first renewal is due at once"; EXPECT_EQ(admitted_at[1], kReadinessStaleCommitBootMs) << "a stale success is followed at once"; - /// Case 4: the second renewal started at 161.17 s and committed at 191.135 s with 35 ms left. + /// A short success: the second renewal started at 161.17 s and committed at 191.135 s with 35 ms left. EXPECT_STREQ(after_short.admit, "LostOrRearmed") << "a success with less lease than a ref append needs arms nothing"; EXPECT_EQ(after_short.generation, latched_generation); EXPECT_TRUE(after_short.failure.empty()) << "a deadline in the future ends the run of trouble: " << after_short.failure; @@ -6627,13 +6627,13 @@ struct ReadinessSeen }; } -/// Spec test 14, cases 1, 6 and 7. An open whose claim is too old returns only after the lease thread's +/// An open whose claim is too old returns only after the lease thread's /// renewal armed the fence; a reclaim whose quiescence aged its claim the same way reports success with /// the pool not `Live` and the fence latched, and the next renewal arms it. No sampled point sees the /// fence armed while the pool is not `Live`. TEST(CASMountRuntime, OpenAndRemountReportReadyOnlyWhenWritable) { - /// Case 1. The claim starts at 10 s and the adopt write ends at 30 s. + /// An open whose claim starts at 10 s and whose adopt write ends at 30 s. { auto fake_boot = std::make_shared>(kReadyClaimBootMs); auto seen = std::make_shared(); @@ -6665,7 +6665,7 @@ TEST(CASMountRuntime, OpenAndRemountReportReadyOnlyWhenWritable) EXPECT_NO_THROW(publishPart(store, "srv/ready-open", "x", "payload")) << "a ref append is admitted when the open returns"; } - /// Cases 6 and 7. A fresh open arms at once. Then a reclaim: claim at 10 s, quiescence until 30 s. + /// A fresh open arms at once. Then a reclaim: claim at 10 s, quiescence until 30 s. { auto fake_boot = std::make_shared>(kReadyClaimBootMs); auto seen = std::make_shared(); @@ -6712,8 +6712,8 @@ TEST(CASMountRuntime, OpenAndRemountReportReadyOnlyWhenWritable) ASSERT_TRUE(store); ASSERT_EQ(backend->mount_writes.load(), 2u) << "a fresh claim arms at once"; seen->pool = store.get(); - /// Runs inside the arm, after the new generation and before `Live` and the open fence. It also makes - /// the next renewal due at once, so its request shows the state the arm left. + /// Runs inside the arm, after `Live` is reported and before the fence opens, so it sees the order of + /// the two. It also makes the next renewal due at once, so its request shows the state the arm left. store->setArmMountFenceInterpositionHookForTest([seen, fake_boot] { Pool * pool = seen->pool.load(); @@ -6729,25 +6729,25 @@ TEST(CASMountRuntime, OpenAndRemountReportReadyOnlyWhenWritable) EXPECT_EQ(seen->remount_outcome, "ok"); EXPECT_EQ(seen->remount_step, "claimed_not_armed"); EXPECT_EQ(eventCount(ProfileEvents::CASRemountSucceeded), succeeded_before + 1); - EXPECT_EQ(store->lifecycle(), PoolLifecycle::TransientNotLive) << "case 6: not Live after the reclaim"; - EXPECT_FALSE(store->mayMutate()) << "case 6: the fence stays latched"; + EXPECT_EQ(store->lifecycle(), PoolLifecycle::TransientNotLive) << "not Live after the reclaim"; + EXPECT_FALSE(store->mayMutate()) << "the fence stays latched"; remounted->release(); renewed_after_arm->waitUntilArrived(); ASSERT_EQ(backend->mount_writes.load(), 7u); EXPECT_EQ(seen->renewal_boot_ms.load(), kReadyClaimBootMs + kReadyClaimAgeMs) << "the loop renewed at once"; - /// Case 7: at every sampled point, an armed fence comes with `Live`. + /// At every sampled point, an armed fence comes with `Live`. EXPECT_EQ(seen->lifecycle_at_renewal.load(), static_cast(PoolLifecycle::TransientNotLive)); EXPECT_EQ(seen->may_mutate_at_renewal.load(), 0) << "the renewal is sent under the latched fence"; - EXPECT_EQ(seen->lifecycle_at_arm.load(), static_cast(PoolLifecycle::TransientNotLive)); - EXPECT_EQ(seen->may_mutate_at_arm.load(), 0) << "inside the arm, before `Live`, the fence still refuses"; + EXPECT_EQ(seen->lifecycle_at_arm.load(), static_cast(PoolLifecycle::Live)) << "`Live` is reported first"; + EXPECT_EQ(seen->may_mutate_at_arm.load(), 0) << "inside the arm, after `Live`, the fence still refuses"; EXPECT_EQ(seen->lifecycle_after_arm.load(), static_cast(PoolLifecycle::Live)); EXPECT_EQ(seen->may_mutate_after_arm.load(), 1) << "after the arm: `Live` and writable"; renewed_after_arm->release(); } } -/// Spec test 14, case 2. Every readiness renewal fails: the open waits one lease on the fence clock, then +/// Every readiness renewal fails: the open waits one lease on the fence clock, then /// stops and joins the lease thread and fails, naming the last failed request. TEST(CASMountRuntime, AnOpenWhoseReadinessRenewalKeepsFailingFailsAfterOneTtl) { @@ -6812,7 +6812,7 @@ TEST(CASMountRuntime, AnOpenWhoseReadinessRenewalKeepsFailingFailsAfterOneTtl) EXPECT_EQ(eventCount(ProfileEvents::CASMountLeaseExpired), expired_before); } -/// Ruling O1: a writable open with no lease thread whose claim does not admit a ref append fails at once +/// A writable open with no lease thread whose claim does not admit a ref append fails at once /// with a retryable error, and sends no renewal. TEST(CASPool, AnOpenWithoutALeaseThreadFailsWhenItsClaimIsTooOld) { @@ -7088,13 +7088,17 @@ TEST(CASMountRuntime, AnOpenFailsWhenTheLeaseThreadEndedOnItsOwn) } #else /// A `LOGICAL_ERROR` aborts a debug or sanitizer build where it is constructed, so the loop's error path is -/// reached only in a release build. Here the scenario must end the process. +/// reached only in a release build. Here the scenario must end the process. The child is a re-executed +/// process, not a fork: a forked child inherits the global thread pool's count of idle workers but not +/// the workers, so the lease thread's job could stay queued and the join would hang. TEST(CASMountRuntimeDeathTest, AnOpenFailsWhenTheLeaseThreadEndedOnItsOwn) { + GTEST_FLAG_SET(death_test_style, "threadsafe"); EXPECT_DEATH( { LeaseThreadEndSeen seen; runLeaseThreadEndsOnItsOwn(seen); + std::_Exit(0); }, "injected lease-loop failure"); } From e5e1dcbf283e3354212964a0a167259a82f7ae2b Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 07:26:01 +0200 Subject: [PATCH 13/37] Describe when a CAS remount counts as succeeded or failed A reclaim can claim the mount and still fail at a later step; success means it completed every step through the arm step. Co-Authored-By: Claude Opus 5.5 --- docs/en/antalya/cas/operations/monitoring.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/en/antalya/cas/operations/monitoring.md b/docs/en/antalya/cas/operations/monitoring.md index a749b1f91f46..57dc51c3a0ef 100644 --- a/docs/en/antalya/cas/operations/monitoring.md +++ b/docs/en/antalya/cas/operations/monitoring.md @@ -62,8 +62,8 @@ window and correlate them with the `server_root_id` in `system.cas_log`. | `CASMountRenewalDeadlineExceeded` | One per logical renewal stopped by the external lease-safety deadline | The last confirmed lease no longer left enough safe time; this is narrower than request-budget exhaustion. Only the bounded renewals (startup, remount and direct) can reach it; the background renewal does not | | `CASMountLeaseExpired` | One per renewal that restored a lease that had expired | Moves at the restore, not when the lease expires. While it is expired, `system.cas_mounts` shows `lifecycle_reason = 'lease_expired'` and writes are refused; the `watermark_renew` row of the restoring renewal carries `expired_ms` | | `CASRemountAttempts` | One per invocation of the existing whole-chain remount attempt | Includes both successful and failed attempts | -| `CASRemountSucceeded` | One per whole-chain attempt that claimed the mount under a fresh writer epoch | Must be a subset of `CASRemountAttempts`. Includes a reclaim whose claim left too little lease to arm the fence (`step = 'claimed_not_armed'`); the next renewal arms it | -| `CASRemountFailed` | One per whole-chain attempt that returned without claiming the mount | Includes a named step exception or a step that returned transiently | +| `CASRemountSucceeded` | One per whole-chain attempt that completed every step through the arm step, under a fresh writer epoch | Must be a subset of `CASRemountAttempts`. Includes a reclaim whose claim left too little lease to arm the fence (`step = 'claimed_not_armed'`); the next renewal arms it | +| `CASRemountFailed` | One per whole-chain attempt that stopped before the arm step | Includes a named step exception or a step that returned transiently, also after the mount claim succeeded (for example at `renewer_start` or `quiesce_ref_tables`) | `CASMountLeaseLost` complements those nine counters. It increments exactly once per operational `Live -> TransientNotLive` recovery generation: either the initiating external loss or the first From a127ae04af4c410cd5796b9d365530f62a98a9ac Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 08:22:49 +0200 Subject: [PATCH 14/37] Renew the CAS mount lease through one step, also from the test seam The startup and remount re-anchors had no caller left. renewOnce is the lease thread's renewal step: it renews on the lease plane with no lease bound, consumes the result and reports it. renewWatermarkOnce runs that step on the calling thread for tests. Tests that relied on the seam stopping at the lease deadline now end on a definitive answer or on their liveness. Co-Authored-By: Claude Opus 5.5 --- .../ContentAddressed/Pool/CasMountRuntime.cpp | 114 ++++++------------ .../ContentAddressed/Pool/CasMountRuntime.h | 45 +++---- .../ContentAddressed/Pool/CasPool.cpp | 3 - src/Disks/tests/gtest_cas_event_log.cpp | 97 ++------------- src/Disks/tests/gtest_cas_mount.cpp | 17 +-- src/Disks/tests/gtest_cas_observability.cpp | 32 ----- src/Disks/tests/gtest_cas_pool.cpp | 98 ++++++--------- 7 files changed, 104 insertions(+), 302 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index f5bd34fa380d..64367c2be1f9 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -400,7 +400,16 @@ uint64_t CasMountRuntime::peekNextBuildSeq() void CasMountRuntime::renewWatermarkOnce() { - (void)renewRenewerOnce(RenewCaller::Direct); + { + std::lock_guard lock(driver_mutex); + if (config.background_watermark) + throw Exception( + ErrorCodes::LOGICAL_ERROR, + "CAS mount runtime: direct renewal is disabled when background ownership is configured"); + } + const MountRenewResult result = renewOnce(); + if (result.outcome == MountRenewOutcome::Terminal) + std::rethrow_exception(result.failure); } uint64_t CasMountRuntime::allocateBuildSeq() @@ -498,37 +507,29 @@ uint64_t CasMountRuntime::startRenewer() return renewer->start([this] { return !renewalCancelled(); }); } -MountRenewOperationEnvironment CasMountRuntime::renewalEnvironment(RenewCaller caller) +MountRenewOperationEnvironment CasMountRuntime::renewalEnvironment() { - const bool loop = caller == RenewCaller::Loop; return MountRenewOperationEnvironment{ .boot_ms = [this] { return bootMsNow(); }, - .live = [this, caller] + .live = [this] { - return renewalLive(caller) && (!config.renewal_live_for_test || config.renewal_live_for_test()); + return renewalLive() && (!config.renewal_live_for_test || config.renewal_live_for_test()); }, .cancelled = [this] { return renewalCancelled(); }, - /// Only the loop keeps renewing past the lease; startup, remount and direct renewals stay - /// bounded by it. - .policy = loop ? MountRenewPolicy::UntilDefinitive : MountRenewPolicy::LeaseBound, - /// The loop counts its requests as they are sent, so an outage shows while it lasts. - .on_request = loop - ? std::function( - [this](const MountRenewRequestEvent & event) { noteRenewRequest(event); }) - : nullptr, + .policy = MountRenewPolicy::UntilDefinitive, + /// Counts the requests as they are sent, so an outage shows while it lasts. + .on_request = [this](const MountRenewRequestEvent & event) { noteRenewRequest(event); }, }; } -bool CasMountRuntime::renewalLive(RenewCaller caller) const +bool CasMountRuntime::renewalLive() const { std::lock_guard lock(driver_mutex); - if (workers_stop_requested) - return false; - if (caller != RenewCaller::Loop) - return true; /// A lost fence alone does not end it: the renewal that makes an open or a reclaim ready runs under /// one, and every trip that must end it comes with a request, a terminal lifecycle or a stop. - return remount_requested_generation <= remount_handled_generation && !remountTerminal(); + return !workers_stop_requested + && remount_requested_generation <= remount_handled_generation + && !remountTerminal(); } bool CasMountRuntime::renewalCancelled() const @@ -543,15 +544,8 @@ void CasMountRuntime::sleepInterruptibly(uint64_t ms) driver_cv.wait_for(lock, std::chrono::milliseconds(ms), [this] { return workers_stop_requested; }); } -void CasMountRuntime::consumeRenewResult(const MountRenewResult & result, RenewCaller caller) +void CasMountRuntime::consumeRenewResult(const MountRenewResult & result) { - /// The loop counts its requests as they are sent (`noteRenewRequest`). Every other renewal counts - /// them here, from its result. - if (caller != RenewCaller::Loop && result.attempts_sent > 0) - { - ProfileEvents::incrementNoTrace(ProfileEvents::CASMountRenewalAttempts, result.attempts_sent); - ProfileEvents::incrementNoTrace(ProfileEvents::CASMountRenewalRetries, result.attempts_sent - 1); - } if (result.resolved_by_read) ProfileEvents::incrementNoTrace(ProfileEvents::CASMountRenewalResolved); @@ -579,31 +573,16 @@ void CasMountRuntime::consumeRenewResult(const MountRenewResult & result, RenewC } else if (result.outcome == MountRenewOutcome::Terminal) { - switch (caller) + if (workers_stop_requested) + { + tripFenceWithoutOperationalLoss(); + } + else { - case RenewCaller::Loop: - if (workers_stop_requested) - { - tripFenceWithoutOperationalLoss(); - break; - } - tripMountLost(); - schedule_remount_calls_for_test.fetch_add(1, std::memory_order_relaxed); - if (!remountTerminal() && lossNeedsNewRequest()) - ++remount_requested_generation; - break; - case RenewCaller::Direct: - tripMountLost(); - schedule_remount_calls_for_test.fetch_add(1, std::memory_order_relaxed); - break; - case RenewCaller::Startup: - tripFenceWithoutOperationalLoss(); - break; - case RenewCaller::Remount: - /// The reclaim latched the fence at its start and reports this failure itself. - if (workers_stop_requested) - tripFenceWithoutOperationalLoss(); - break; + tripMountLost(); + schedule_remount_calls_for_test.fetch_add(1, std::memory_order_relaxed); + if (!remountTerminal() && lossNeedsNewRequest()) + ++remount_requested_generation; } } driver_cv.notify_all(); @@ -645,20 +624,13 @@ void CasMountRuntime::consumeRenewResult(const MountRenewResult & result, RenewC } reportMountRenewCompletion(result, std::nullopt); - - if (caller != RenewCaller::Loop) - std::rethrow_exception(result.failure); } -MountRenewResult CasMountRuntime::renewRenewerOnce(RenewCaller caller) +MountRenewResult CasMountRuntime::renewOnce() { MountLeaseRenewer * renewer = nullptr; { std::lock_guard lock(driver_mutex); - if (caller == RenewCaller::Direct && config.background_watermark) - throw Exception( - ErrorCodes::LOGICAL_ERROR, - "CAS mount runtime: direct renewal is disabled when background ownership is configured"); checkRenewerOwner(); if (!mount_renewer) throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: renewal without a renewer"); @@ -666,28 +638,12 @@ MountRenewResult CasMountRuntime::renewRenewerOnce(RenewCaller caller) throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: renewal requires an Active renewer"); renewer = mount_renewer.get(); } - /// Configuration is pointer/POD-only. A remount re-anchor retains its completed observation for the - /// whole-chain finalizer to deliver after `remount_mutex` is released. - configureMountRenewObservability(&server_root_id, &event_sink, caller == RenewCaller::Remount); - /// The remount re-anchor runs before `armIfAdmissible`, with the fence still latched, so it renews on - /// the renewer's open plane: admitted under the mount fence it could only ever give up. - const MountRenewResult result = caller == RenewCaller::Remount - ? renewer->renewForRemount(renewalEnvironment(caller)) - : renewer->renew(renewalEnvironment(caller)); - consumeRenewResult(result, caller); + configureMountRenewObservability(&server_root_id, &event_sink, /*deferred=*/false); + const MountRenewResult result = renewer->renew(renewalEnvironment()); + consumeRenewResult(result); return result; } -uint64_t CasMountRuntime::renewRenewerForStartupOnce() -{ - return renewRenewerOnce(RenewCaller::Startup).attempt_start_boot_ms; -} - -uint64_t CasMountRuntime::renewRenewerForRemountOnce() -{ - return renewRenewerOnce(RenewCaller::Remount).attempt_start_boot_ms; -} - void CasMountRuntime::renewerReset() { std::unique_ptr released; @@ -831,7 +787,7 @@ void CasMountRuntime::renewalLoop() config.renewal_admitted_hook_for_test(); try { - (void)renewRenewerOnce(RenewCaller::Loop); + (void)renewOnce(); } catch (...) { diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index 476c288bce1a..49ef1d1dbe2c 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -153,9 +153,9 @@ class CasMountRuntime uint64_t minActive(); /// Test/assertion accessor for the next-to-allocate build_seq under the lock. uint64_t peekNextBuildSeq(); - /// Renew the merged mount heartbeat once, including its build-watermark floor. A read-only runtime - /// has no renewer and fails with a logical exception rather than fabricating a heartbeat. - /// A test seam: it is not protected against a concurrent replacement of the renewer. + /// Runs the lease thread's renewal step once on the calling thread and rethrows a terminal failure: + /// the test seam. Refused when the runtime is configured for a lease thread, and on a read-only + /// runtime, which has no renewer. void renewWatermarkOnce(); /// ---- local write fence ---- @@ -164,7 +164,7 @@ class CasMountRuntime /// Latch the fence lost and count one lease loss. A trip that comes alone is re-armed by the next /// renewal that commits with room for a ref append, so every production caller pairs it with a /// remount request, a terminal lifecycle (a published FORGET intent included) or a stop, or trips - /// where no renewal follows: the lease loop's error exit and a direct renewal. + /// where no renewal follows: the lease loop's error exit. void tripMountLost(); /// Publish the BOOTTIME deadline from a successful lease renewal. void setMountDeadline(uint64_t deadline_boot_ms); @@ -317,15 +317,15 @@ class CasMountRuntime /// renewal that commits with a start more than a TTL ago does not end the expiry, so the first expired /// deadline is kept until a renewal restores the lease. Empty otherwise. std::optional leaseExpiredSinceBootMs() const; - /// Text of the last failed request of the worker's renewal; empty when no request failed since the + /// Text of the last failed renewal request; empty when no request failed since the /// last renewal that left the lease unexpired. String lastRenewFailure() const; /// The condition text for a request admitted under `admitted_generation` that is refused only /// because the lease expired; empty when the refusal has any other cause or there is none. std::optional leaseExpiredRefusal(uint64_t admitted_generation) const; - /// Counts each `PUT` of the worker's renewal as it is sent and keeps the text of every failed - /// `PUT` or resolve read, and wakes `waitUntilArmed` on each failure. Runs on the renewing thread, - /// never under `driver_mutex`. + /// Counts each `PUT` of a renewal as it is sent and keeps the text of every failed `PUT` or resolve + /// read, and wakes `waitUntilArmed` on each failure. Runs on the renewing thread, never under + /// `driver_mutex`. void noteRenewRequest(const MountRenewRequestEvent & event) noexcept; /// Writes one `WARNING` per lease expiry, at the first request event after it. Lease thread only. @@ -387,8 +387,6 @@ class CasMountRuntime /// ---- mount-lease renewer and the lease thread ---- void installRenewer(UInt128 our_uuid, uint64_t writer_epoch, const std::function & now_ms); uint64_t startRenewer(); - uint64_t renewRenewerForStartupOnce(); - uint64_t renewRenewerForRemountOnce(); void renewerReset(); void startBackgroundWorkers(std::chrono::milliseconds period); void stopBackgroundWorkers(); @@ -401,7 +399,7 @@ class CasMountRuntime bool scheduleRemountForTest(); void beginShutdownForTest(); /// Return how many remount requests were attempted, refused ones included: `scheduleRemount`, - /// `tripAndRequestRemount` and the loop's terminal renewals. This is useful for testing the renewer's loss callback without starting a real recovery. + /// `tripAndRequestRemount` and terminal renewals. This is useful for testing the renewer's loss callback without starting a real recovery. uint64_t scheduleRemountCallCountForTest() const { return schedule_remount_calls_for_test.load(std::memory_order_relaxed); @@ -429,16 +427,6 @@ class CasMountRuntime void emitEvent(CasEvent && e) const { if (event_sink) event_sink(std::move(e)); } private: - /// Who drives a renewal. Only the loop's renewal is unbounded, counts its requests as they are - /// sent, and raises a remount request when it ends terminal. - enum class RenewCaller : uint8_t - { - Loop, - Startup, - Remount, - Direct, - }; - /// A committed renewal that ended an expiry: how long the lease was expired, and the text of the /// last failed request. struct RestoredLease @@ -447,21 +435,22 @@ class CasMountRuntime String last_failure; }; - MountRenewResult renewRenewerOnce(RenewCaller caller); - MountRenewOperationEnvironment renewalEnvironment(RenewCaller caller); + /// The renewal step of the lease thread: renew, consume the result under `driver_mutex`, then report + /// it with no lock held. + MountRenewResult renewOnce(); + MountRenewOperationEnvironment renewalEnvironment(); std::optional leaseExpiredAt(uint64_t now_boot_ms) const; /// Publishes a committed renewal's deadline. Requires `driver_mutex`. A deadline in the future ends /// the current run of trouble: it clears the failure text and, when the lease was expired, counts /// the restore and returns it for the caller to log after the unlock. Empty when the lease was not /// expired or is still expired. std::optional publishRenewedDeadline(uint64_t deadline_boot_ms); - void consumeRenewResult(const MountRenewResult & result, RenewCaller caller); + void consumeRenewResult(const MountRenewResult & result); void renewalLoop(); ThreadFromGlobalPool makeWorker(std::function body); - /// The renewal's liveness. Every caller ends on a stop. A loop renewal also ends on a pending remount - /// request and on a terminal lifecycle, which includes a published FORGET intent; a lost fence alone - /// does not end it. FALSE ends the renewal. - bool renewalLive(RenewCaller caller) const; + /// The renewal's liveness: no stop, no pending remount request, no terminal lifecycle (a published + /// FORGET intent included); a lost fence alone does not end it. FALSE ends the renewal. + bool renewalLive() const; /// Whether this node has already been asked to stop. Sampled ONCE, before the write, so a refusal /// caused by the stop cannot be mistaken for one that preceded it. bool renewalCancelled() const; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp index 67f7bbe7e00c..295d3fc618f0 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp @@ -1240,9 +1240,6 @@ bool Pool::tryRemountOnce() String error; SCOPE_EXIT( { - /// A remount re-anchor completed while the whole-chain serializer was held. Drain its POD snapshot - /// first, after lock destruction, so renewal recovery/failure precedes and correlates with the - /// containing remount result without any callback or allocation under `remount_mutex`. deliverDeferredMountRenewObservability(attempt_no); if (succeeded) diff --git a/src/Disks/tests/gtest_cas_event_log.cpp b/src/Disks/tests/gtest_cas_event_log.cpp index 49fd515323d3..012a1906fcaf 100644 --- a/src/Disks/tests/gtest_cas_event_log.cpp +++ b/src/Disks/tests/gtest_cas_event_log.cpp @@ -41,24 +41,6 @@ class RenewalEventBackend final : public InMemoryBackend bool throw_before_next_write = false; bool throw_nonretryable_next_write = false; bool vanish_on_next_write = false; - /// Runs just before an armed fault throws. The engine draws its inter-attempt backoff randomly and - /// admits the reissue against that drawn duration, so a test that needs the ambiguity refused - /// rather than reissued has to move the injected clock here -- from inside the attempt, the only - /// point between admission and the resolve read a test can reach. - std::function before_throw; - - void armResolveProbe() - { - std::lock_guard lock(resolve_mutex); - observe_next_read = true; - resolve_started = false; - } - - bool resolveStarted() - { - std::lock_guard lock(resolve_mutex); - return resolve_started; - } /// A plain read/replace through the primitive surface, for fixtures that need to observe or seed /// state without going through the pool under test. @@ -74,21 +56,6 @@ class RenewalEventBackend final : public InMemoryBackend return std::holds_alternative((*op).replace(key, bytes, expected, Retry::standard())); } - /// The engine settles an ambiguous write by reading the key back, so the observation belongs on the - /// READ PRIMITIVE -- the resolve read never reaches the legacy `get`. - std::optional read(const String & key, DB::Cas::TransportAccess & access) override - { - { - std::lock_guard lock(resolve_mutex); - if (observe_next_read) - { - resolve_started = true; - observe_next_read = false; - } - } - return InMemoryBackend::read(key, access); - } - /// The faults sit on the WRITE PRIMITIVE, and only on a CONDITIONAL write: a lease renewal is a /// replace, so the pool's own create-if-absent writes must not consume a one-shot fault. std::expected write( @@ -107,18 +74,9 @@ class RenewalEventBackend final : public InMemoryBackend if (std::exchange(throw_nonretryable_next_write, false)) throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, "injected deterministic renewal rejection"); if (std::exchange(throw_before_next_write, false)) - { - if (before_throw) - before_throw(); throw Poco::TimeoutException("injected renewal timeout before commit"); - } return InMemoryBackend::write(key, bytes, expected_value, access); } - -private: - std::mutex resolve_mutex; - bool observe_next_read = false; - bool resolve_started = false; }; CasRequestBudget renewalEventBudget() @@ -285,43 +243,6 @@ TEST(CASEvent, WatermarkRenewEventsAreBoundedAndComplete) EXPECT_TRUE(renewals[0].detail.contains(key)) << "missing detail key " << key; } -/// An attempt that spends the lease it was admitted under must not START the resolving read. That read -/// is the only thing that can prove the ambiguous attempt landed, and issuing it past the lease-safe -/// bound would be a request made without the authority it was admitted under -- so the renewal reports -/// the deadline that refused it instead of resolving anything. -TEST(CASEvent, AnAmbiguityPastTheLeaseBoundNeverStartsTheResolvingRead) -{ - auto backend = std::make_shared(); - auto boot_ms = std::make_shared>(100); - /// Heap-owned, not a plain local: the Pool can outlive this stack frame (a background publish holds - /// `shared_from_this()`), so a by-reference capture of a local would dangle. - auto events = std::make_shared(); - auto store = openRenewalEventPool( - backend, boot_ms, renewalEventBudget(), "renewal-inflight-ambiguity"); - store->setEventSink([events](CasEvent event) - { - events->push(std::move(event)); - }); - - /// The lease was anchored at 100 with a 1000 ms TTL, so the fence expires at 1100 and holds a 20 ms - /// safety margin. At 1081 only 19 ms remain, and admission refuses the resolve read. - backend->before_throw = [boot_ms] - { - boot_ms->store(1'081); - }; - backend->throw_before_next_write = true; - backend->armResolveProbe(); - EXPECT_THROW(store->renewWatermarkOnce(), DB::Exception); - - EXPECT_FALSE(backend->resolveStarted()) - << "an attempt that consumed the lease must not start the resolving read"; - const std::vector renewals = watermarkRenewEvents(events->snapshot()); - ASSERT_EQ(renewals.size(), 1u); - EXPECT_EQ(renewals[0].outcome, "failed"); - EXPECT_EQ(renewals[0].detail.at("attempts_sent"), "1"); - EXPECT_EQ(renewals[0].detail.at("classification"), "external_lease_deadline"); -} - /// Ten renewals nested through each other's conflict sinks, against an eight-slot observation stack. /// The two calls beyond the stack get no rich event -- and must still report their own physical attempt /// count, which rides the write result rather than the suppressed observation. @@ -432,10 +353,8 @@ TEST(CASEvent, WatermarkRenewSinkFailureCannotChangeOutcome) EXPECT_TRUE(store->mayMutate()); } -/// The two terminal endings a renewal reaches without ever settling its write: the store refusing it -/// outright, and the lease refusing to admit it. There is no attempt-count ending -- the engine bounds a -/// write by time, never by a number of tries -- and the deadline ending that DOES send an attempt first -/// is `AnAmbiguityPastTheLeaseBoundNeverStartsTheResolvingRead`. +/// Two terminal endings and what their rows carry: the store rejecting the renewal outright, and the +/// slot vanishing under it. TEST(CASEvent, TerminalRenewalDetailsPreservePhysicalTruthAndClassification) { const auto one_failed_event = [](const std::vector & events) -> std::optional @@ -477,20 +396,18 @@ TEST(CASEvent, TerminalRenewalDetailsPreservePhysicalTruthAndClassification) /// Heap-owned, not a plain local: the Pool can outlive this stack frame (a background publish /// holds `shared_from_this()`), so a by-reference capture of a local would dangle. auto events = std::make_shared(); - auto store = openRenewalEventPool(backend, boot_ms, renewalEventBudget(), "renewal-deadline-details"); + auto store = openRenewalEventPool(backend, boot_ms, renewalEventBudget(), "renewal-vanished-details"); store->setEventSink([events](CasEvent event) { events->push(std::move(event)); }); - /// The lease was anchored at 100 with a 1000 ms TTL and holds a 20 ms safety margin, so 1090 - /// leaves 10 ms of it and admission refuses the renewal before its first attempt. - boot_ms->store(1090); + backend->vanish_on_next_write = true; EXPECT_THROW(store->renewWatermarkOnce(), DB::Exception); const std::optional failed = one_failed_event(events->snapshot()); - ASSERT_TRUE(failed.has_value()) << "the refused admission must reach the event log"; - EXPECT_EQ(failed->detail.at("attempts_sent"), "0"); - EXPECT_EQ(failed->detail.at("classification"), "external_lease_deadline"); + ASSERT_TRUE(failed.has_value()) << "the vanished slot must reach the event log"; + EXPECT_EQ(failed->detail.at("attempts_sent"), "1"); + EXPECT_EQ(failed->detail.at("classification"), "vanished"); } } diff --git a/src/Disks/tests/gtest_cas_mount.cpp b/src/Disks/tests/gtest_cas_mount.cpp index d94868ed661b..386248c10a1d 100644 --- a/src/Disks/tests/gtest_cas_mount.cpp +++ b/src/Disks/tests/gtest_cas_mount.cpp @@ -238,8 +238,9 @@ class RenewalLogBackend final : public InMemoryBackend { public: bool throw_before_next_overwrite = false; + bool refuse_next_overwrite = false; - /// The fault lives on the primitive every write reaches the store through, keyed to the mount + /// The faults live on the primitive every write reaches the store through, keyed to the mount /// slot so the pool's other conditional writes pass untouched. std::expected write( const String & key, @@ -247,8 +248,13 @@ class RenewalLogBackend final : public InMemoryBackend const std::optional & expected_value, TransportAccess & access) override { - if (expected_value && key.ends_with("/mount") && std::exchange(throw_before_next_overwrite, false)) - throw Poco::TimeoutException("injected renewal timeout before commit"); + if (expected_value && key.ends_with("/mount")) + { + if (std::exchange(refuse_next_overwrite, false)) + throw DB::Exception(DB::ErrorCodes::BAD_ARGUMENTS, "injected deterministic renewal rejection"); + if (std::exchange(throw_before_next_overwrite, false)) + throw Poco::TimeoutException("injected renewal timeout before commit"); + } return InMemoryBackend::write(key, bytes, expected_value, access); } }; @@ -373,10 +379,7 @@ TEST(CASMountAudit, RenewalDefaultLogsAreBounded) auto boot_ms = std::make_shared>(100); auto store = open_store(backend, boot_ms, "renewal-log-fenced"); ScopedRenewalLogCapture capture("information"); - /// The lease was claimed at boot 100 with the 1000 ms TTL above, so it expires at 1100. The - /// fence admits only while the remaining time strictly clears the safety margin plus whatever - /// the attempt reserves, so exactly `margin` remaining (with the reservation on top) refuses. - boot_ms->store(1100 - renewalLogBudget().lease_safety_margin_ms); + backend->refuse_next_overwrite = true; EXPECT_THROW(store->renewWatermarkOnce(), DB::Exception); const String output = capture.captured(); EXPECT_EQ(countRenewalLogText(output, "CAS mount renewal"), 1u) << output; diff --git a/src/Disks/tests/gtest_cas_observability.cpp b/src/Disks/tests/gtest_cas_observability.cpp index 2521aae58294..4e12c1988e95 100644 --- a/src/Disks/tests/gtest_cas_observability.cpp +++ b/src/Disks/tests/gtest_cas_observability.cpp @@ -177,38 +177,6 @@ TEST(CASObservability, RenewalCountersHaveExactPhysicalAndLogicalDeltas) run(RenewalCounterBackend::Fault::LandThenThrow, /*attempts=*/1, /*retries=*/0, /*resolved=*/1, /*recovered=*/1); } -TEST(CASObservability, ExternalLeaseDeadlineCountsOnceWithoutReconstructingAttempts) -{ - auto backend = std::make_shared(); - backend->setAttemptTimeoutMs(renewalCounterBudget().attempt_timeout_ms); - /// Held in a shared atomic, not a plain local: this test mutates it below, and the Pool can - /// outlive this stack frame (a background publish holds `shared_from_this()`), so a by-reference - /// capture of a local would dangle. - auto boot_ms = std::make_shared>(100); - auto store = Pool::open(backend, PoolConfig{ - .pool_prefix = "renewal-deadline-counter", - .server_root_id = "test", - .mount_lease_ttl_ms = std::chrono::milliseconds(1000), - .cas_request_budget = renewalCounterBudget(), - .boot_ms_fn = [boot_ms] - { - return boot_ms->load(); - }, - }); - - /// The fence deadline is 1100 and the safety margin 20, so admission refuses once fewer than - /// twenty milliseconds of lease remain. At 1090 only ten milliseconds remain, short of the margin - /// however much a single attempt reserves, so nothing can be started and the logical renewal ends - /// without reconstructing a sent attempt. - boot_ms->store(1090); - const RenewalCounterSnapshot before = renewalCounters(); - EXPECT_THROW(store->renewWatermarkOnce(), DB::Exception); - const RenewalCounterSnapshot after = renewalCounters(); - expectRenewalCounterDelta( - before, after, /*attempts=*/0, /*retries=*/0, /*resolved=*/0, /*recovered=*/0, - /*deadline_exceeded=*/1); -} - } /// B170/Task 1 (Part A audit events): `PartWriteTxn::stageManifest` writes a part-manifest body but never diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp index bc7d80f44fd4..6f3f5a01ef59 100644 --- a/src/Disks/tests/gtest_cas_pool.cpp +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -2020,11 +2020,6 @@ class RuntimeRenewBackend final : public DB::Cas::tests::CountingBackend Fault fault = Fault::None; DB::Cas::tests::ManualBarrier * barrier = nullptr; std::function after_commit; - /// Runs just before an armed fault throws. The engine draws its inter-attempt backoff randomly and - /// admits the reissue against that drawn duration, so a test that needs the ambiguity to be refused - /// rather than reissued has to move the injected clock here -- from inside the attempt, which is the - /// only point between admission and the resolve read a test can reach. - std::function before_throw; /// While it answers TRUE, every conditional write throws a transport timeout, before `fault` is /// consulted. Read on the renewing thread; set it before the workers start. std::function outage; @@ -2056,11 +2051,7 @@ class RuntimeRenewBackend final : public DB::Cas::tests::CountingBackend if (current == Fault::ThrowMemoryLimitExceeded) throw DB::Exception(DB::ErrorCodes::MEMORY_LIMIT_EXCEEDED, "injected memory limit exceeded on the renewal request"); if (current == Fault::ThrowBefore || current == Fault::BlockThenThrow) - { - if (before_throw) - before_throw(); throw Poco::TimeoutException("injected runtime renewal ambiguity before result"); - } auto result = DB::Cas::tests::CountingBackend::write(key, bytes, expected_value, access); if (after_commit) @@ -4116,53 +4107,40 @@ TEST(CASPoolShutdown, PreSendCancellationAllowsFarewellButAmbiguityDoesNot) EXPECT_NE(run(true), std::numeric_limits::max()); } -TEST(CASPool, DirectAndStartupTerminalFailuresRethrowTypedExceptions) +/// The test seam rethrows a terminal renewal as the typed failure it ended with. +TEST(CASPool, DirectTerminalFailureRethrowsTypedException) { - enum class Refusal : uint8_t { PreAttemptDeadline, RefusedAfterSend }; - const auto run = [](bool startup, Refusal refusal) + std::atomic renewal_live{true}; + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + auto backend = std::make_shared(); + const Layout layout("typed-direct"); + const UInt128 uuid{1}; + ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); + CasEventSink sink; + RuntimeUnderTest runtime_holder( + backend, layout, + MountConfig{ + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .boot_ms_fn = [&] { return boot_ms; }, + .renewal_live_for_test = [&] { return renewal_live.load(std::memory_order_acquire); }}, + "test", sink, runtimeRenewBudget(), [] { return false; }); + CasMountRuntime & runtime = *runtime_holder; + runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); + const uint64_t anchor = runtime.startRenewer(); + runtime.armMountFence(uuid, 1, anchor + 1000); + /// The renewal lands, then its liveness ends before the commit is confirmed. + backend->after_commit = [&] { renewal_live.store(false, std::memory_order_release); }; + try { - auto backend = std::make_shared(); - const Layout layout(startup ? "typed-startup" : "typed-direct"); - uint64_t wall_ms = 1000; - uint64_t boot_ms = 100; - std::atomic renewal_live{true}; - const UInt128 uuid{1}; - ASSERT_EQ(claimMount(*DB::Cas::tests::OperationForTest(backend), layout, "test", uuid, 1, wall_ms, 1000).kind, MountClaimResult::Claimed); - CasEventSink sink; - RuntimeUnderTest runtime_holder( - backend, layout, - MountConfig{ - .mount_lease_ttl_ms = std::chrono::milliseconds(1000), - .boot_ms_fn = [&] { return boot_ms; }, - .renewal_live_for_test = [&] { return renewal_live.load(std::memory_order_acquire); }}, - "test", sink, runtimeRenewBudget(), [] { return false; }); - CasMountRuntime & runtime = *runtime_holder; - runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); - const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); - if (refusal == Refusal::PreAttemptDeadline) - /// Past the point where the lease has more room left than the safety margin (deadline - /// `anchor + 1000` == 1100, margin 20), so admission refuses before anything is sent. - boot_ms = 1090; - else - backend->after_commit = [&] { renewal_live.store(false, std::memory_order_release); }; - try - { - if (startup) - (void)runtime.renewRenewerForStartupOnce(); - else - runtime.renewWatermarkOnce(); - ADD_FAILURE() << "terminal renewal did not propagate"; - } - catch (const DB::Exception & e) - { - EXPECT_EQ(e.code(), DB::ErrorCodes::NETWORK_ERROR) << e.message(); - } - runtime.finishTeardown(false); - }; - for (bool startup : {true, false}) - for (Refusal refusal : {Refusal::PreAttemptDeadline, Refusal::RefusedAfterSend}) - run(startup, refusal); + runtime.renewWatermarkOnce(); + ADD_FAILURE() << "terminal renewal did not propagate"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::NETWORK_ERROR) << e.message(); + } + runtime.finishTeardown(false); } TEST(CASPool, BackgroundCadenceMustFitLeaseBeforeWritablePublication) @@ -5204,16 +5182,10 @@ TEST(CASPool, RenewWatermarkOnceRefreshesFenceAndDepositsOneFailure) fake_boot->store(1200); EXPECT_TRUE(store->mayMutate()) << "direct success must refresh the local fence from attempt start"; - backend->fault = RuntimeRenewBackend::Fault::ThrowBefore; - /// The renewal that succeeded at 500 anchored the lease for its 1000 ms TTL, so it expires at 1500. - /// Expire it from inside the attempt: the fault alone no longer ends a renewal, because the engine - /// settles the ambiguity by reading and reissues, and the reissue commits. - backend->before_throw = [fake_boot] - { - fake_boot->store(1500); - }; + /// GC fences the slot out: a definitive answer, so the renewal ends on its first attempt. + fenceOutMount(*backend, store->layout().mountKey("test")); const uint64_t schedules_before = store->scheduleRemountCallCountForTest(); - expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] { store->renewWatermarkOnce(); }); + expectThrowsCode(DB::ErrorCodes::ABORTED, [&] { store->renewWatermarkOnce(); }); EXPECT_FALSE(store->mayMutate()); EXPECT_EQ(store->scheduleRemountCallCountForTest(), schedules_before + 1); } From 1facfbc249981b920b36c6a6b48ef17fa614bc52 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 08:29:19 +0200 Subject: [PATCH 15/37] Remove MountLeaseRenewer::renewForRemount No remount renews before it arms the fence, so the renewal on the claim-and-farewell plane has no caller. Co-Authored-By: Claude Opus 5.5 --- .../ContentAddressed/Pool/CasServerRoot.cpp | 5 --- .../ContentAddressed/Pool/CasServerRoot.h | 8 +---- src/Disks/tests/gtest_cas_heartbeat.cpp | 26 --------------- src/Disks/tests/gtest_cas_mount.cpp | 32 ------------------- 4 files changed, 1 insertion(+), 70 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp index cdf82fc68a47..fbd40d3309cb 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp @@ -1584,11 +1584,6 @@ MountRenewResult MountLeaseRenewer::renew(const MountRenewOperationEnvironment & environment.policy == MountRenewPolicy::UntilDefinitive ? worker_requests : mount_requests, environment); } -MountRenewResult MountLeaseRenewer::renewForRemount(const MountRenewOperationEnvironment & environment) -{ - return renewOn(open_requests, environment); -} - MountRenewResult MountLeaseRenewer::renewOn( CasRequests & plane, const MountRenewOperationEnvironment & environment) { diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h index 991465e07ee4..ac51c398a940 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h @@ -579,12 +579,6 @@ class MountLeaseRenewer /// `UntilDefinitive` runs on the worker plane with `Retry::untilDefinitive(kMountRenewRetrySpacingMs)` /// and ends on a definitive answer, a deterministic local failure, or when `environment.live` refuses. MountRenewResult renew(const MountRenewOperationEnvironment & environment); - /// The remount's re-anchor, which is bootstrap control rather than steady state: a remount renews - /// BEFORE it arms the fence for the new incarnation, so the fence is still latched lost and an - /// operation admitted under it would be refused before its first attempt. It admits on this - /// renewer's own open plane, the one the claim and the farewell use, so there is no plane for a - /// caller to get wrong. Same policy and same verdicts as `renew`. - MountRenewResult renewForRemount(const MountRenewOperationEnvironment & environment = {}); void release(); MountLeaseRenewerState state() const { return renewer_state; } @@ -596,7 +590,7 @@ class MountLeaseRenewer /// The incarnation every guarded write of this slot names. Engaged for exactly the states that /// admit such a write: `start` establishes it and each committed renewal replaces it. const Etag & precondition() const; - /// One renewal admitted on `plane`; `renew` and `renewForRemount` differ only in which they pass. + /// One renewal admitted on `plane`. MountRenewResult renewOn(CasRequests & plane, const MountRenewOperationEnvironment & environment); Etag claim(CasOperation & op, const String & body); [[noreturn]] void throwRenewConflict(const Observation & seen) const; diff --git a/src/Disks/tests/gtest_cas_heartbeat.cpp b/src/Disks/tests/gtest_cas_heartbeat.cpp index b4fd8d7c1570..67ca53cd6398 100644 --- a/src/Disks/tests/gtest_cas_heartbeat.cpp +++ b/src/Disks/tests/gtest_cas_heartbeat.cpp @@ -1969,29 +1969,3 @@ TEST(CASHeartbeat, AThrowingRequestReportChangesNoOutcome) EXPECT_EQ(f.renewer->state(), MountLeaseRenewerState::Active); } -/// The startup, remount and direct renewals keep their lease bound. -TEST(CASHeartbeat, BoundedPathsStillStopAtTheLeaseDeadline) -{ - for (const bool remount : {false, true}) - { - SCOPED_TRACE(remount ? "renewForRemount" : "renew"); - UnboundedRenewalFixture f; - std::vector sent_at; - f.backend->on_attempt = [&] { sent_at.push_back(f.boot_ms); }; - /// Longer than any lease, so only the bound can end the renewal. - f.backend->outage = [&] { return f.boot_ms < f.renewal_start + 60'000; }; - MountRenewOperationEnvironment environment = renewalEnvironment(f.boot_ms); - environment.policy = MountRenewPolicy::LeaseBound; - - const MountRenewResult result = remount ? f.renewer->renewForRemount(environment) : f.renewer->renew(environment); - - const DB::Exception failure = terminalException(result); - EXPECT_NE(failure.message().find("external_lease_deadline"), String::npos) << failure.message(); - ASSERT_TRUE(result.deadline_source.has_value()); - EXPECT_EQ(*result.deadline_source, GaveUp::Source::Lease); - ASSERT_FALSE(sent_at.empty()); - const uint64_t lease_safe = f.anchor + UnboundedRenewalFixture::ttl_ms - UnboundedRenewalFixture::margin_ms; - for (uint64_t at : sent_at) - EXPECT_LT(at, lease_safe) << "no request starts past the lease-safe bound"; - } -} diff --git a/src/Disks/tests/gtest_cas_mount.cpp b/src/Disks/tests/gtest_cas_mount.cpp index 386248c10a1d..5af96b8b23ee 100644 --- a/src/Disks/tests/gtest_cas_mount.cpp +++ b/src/Disks/tests/gtest_cas_mount.cpp @@ -2443,35 +2443,3 @@ TEST(CASServerRootClaim, OwnerLostToARacerIsDecidedFromTheConflictObservation) } } -/// A remount re-anchors its lease BEFORE it arms the fence for the new incarnation, so the fence is -/// still latched lost at that moment. The steady-state renewal is refused there — the sibling test -/// above pins that — and the remount's own renewal has to be admitted off the fence, or the pool could -/// never re-anchor and the remount attempt would fail on exactly the throttled store that caused it. -TEST(CASMountLease, RemountRenewalIsAdmittedOffTheMountFence) -{ - auto backend = std::make_shared(); - Layout l("p"); - uint64_t now = 1000; - uint64_t boot = 0; - - CasRequests mount_requests(backend, Fence{ - [] { return uint64_t{0}; }, - [](uint64_t, uint64_t) { return Fence::Admit::LostOrRearmed; }, - [](uint64_t) { throw DB::Exception(DB::ErrorCodes::LOGICAL_ERROR, "mount fence lost"); }}); - mount_requests.setNowFnForTest([&boot] { return boot; }); - mount_requests.setSleepFnForTest([&boot](uint64_t ms) { boot += ms; }); - CasRequests open_requests = openRequestsForTest(backend); - open_requests.setNowFnForTest([&boot] { return boot; }); - open_requests.setSleepFnForTest([&boot](uint64_t ms) { boot += ms; }); - - MountLeaseRenewer renewer(mount_requests, open_requests, open_requests, l, "r", UInt128(1), 7, - std::chrono::milliseconds(1000), [&] { return now; }, [] { return uint64_t{0}; }, - {}, std::chrono::milliseconds(0), [&] { return boot; }); - renewer.start(); - - const MountRenewResult redo = renewer.renewForRemount(); - EXPECT_EQ(redo.outcome, MountRenewOutcome::Committed); - - CasOperation reader = open_requests.admit(); - EXPECT_EQ(decodeMountLease(reader.read(l.mountKey("r"), Retry::standard())->bytes).seq, 2u); -} From 59c93ef74824ff529a99b87276f6bcc769ee3fb8 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 08:38:31 +0200 Subject: [PATCH 16/37] Run every CAS mount renewal on the lease plane without a lease bound The only renewal is the lease thread's, so MountRenewPolicy and the fence-gated renewal plane in MountLeaseRenewer and CasMountRuntime go. MountLeaseRenewer takes the claim-and-farewell plane and the lease plane, and renew retries with Retry::untilDefinitive. Renewer tests that stopped after one attempt at the lease bound now end through their liveness. Co-Authored-By: Claude Opus 5.5 --- .../ContentAddressed/Pool/CasMountRuntime.cpp | 5 +- .../ContentAddressed/Pool/CasMountRuntime.h | 9 +- .../ContentAddressed/Pool/CasPool.cpp | 7 +- .../ContentAddressed/Pool/CasRefLedger.h | 3 +- .../ContentAddressed/Pool/CasServerRoot.cpp | 25 +- .../ContentAddressed/Pool/CasServerRoot.h | 39 +-- src/Disks/tests/gtest_cas_event_log.cpp | 1 - src/Disks/tests/gtest_cas_gc_ack_floor.cpp | 4 +- src/Disks/tests/gtest_cas_heartbeat.cpp | 236 ++++++------------ src/Disks/tests/gtest_cas_mount.cpp | 37 +-- .../tests/gtest_cas_mount_claim_conflicts.cpp | 3 +- src/Disks/tests/gtest_cas_mount_runtime.cpp | 7 +- src/Disks/tests/gtest_cas_pool.cpp | 19 +- 13 files changed, 120 insertions(+), 275 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index 64367c2be1f9..890b51900d19 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -54,7 +54,6 @@ int64_t wallClockNowSeconds() CasMountRuntime::CasMountRuntime( BackendPtr backend_ptr_, - CasRequests & mount_requests_, CasRequests & farewell_requests_, CasRequests & lease_requests_, const Layout & layout_, @@ -64,7 +63,6 @@ CasMountRuntime::CasMountRuntime( CasRequestBudget cas_request_budget_, std::function remount_attempt_) : backend_ptr(std::move(backend_ptr_)) - , mount_requests(mount_requests_) , farewell_requests(farewell_requests_) , lease_requests(lease_requests_) , layout(layout_) @@ -479,7 +477,7 @@ void CasMountRuntime::installRenewer( const std::function & now_ms) { std::unique_ptr replaced = std::make_unique( - mount_requests, farewell_requests, lease_requests, layout, server_root_id, our_uuid, writer_epoch, + farewell_requests, lease_requests, layout, server_root_id, our_uuid, writer_epoch, config.mount_lease_ttl_ms, now_ms, [this] { return minActive(); }, [this](CasEvent e) { emitEvent(std::move(e)); }, @@ -516,7 +514,6 @@ MountRenewOperationEnvironment CasMountRuntime::renewalEnvironment() return renewalLive() && (!config.renewal_live_for_test || config.renewal_live_for_test()); }, .cancelled = [this] { return renewalCancelled(); }, - .policy = MountRenewPolicy::UntilDefinitive, /// Counts the requests as they are sent, so an outage shows while it lasts. .on_request = [this](const MountRenewRequestEvent & event) { noteRenewRequest(event); }, }; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index 49ef1d1dbe2c..6875f6e3c69c 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -127,11 +127,9 @@ class CasMountRuntime public: CasMountRuntime( BackendPtr backend_ptr_, - /// The planes the `MountLeaseRenewer` runs on: a bounded renewal under the mount fence, the - /// claim and the farewell on an open one, and the lease loop's renewal on `lease_requests_`, - /// which has no lease budget and whose sleep a stop wakes. Owned by `Pool` and outliving this - /// runtime. - CasRequests & mount_requests_, + /// The planes the `MountLeaseRenewer` runs on: the claim and the farewell on an open-fence one, + /// the renewal on `lease_requests_`, which has no lease budget and whose sleep a stop wakes. + /// Owned by `Pool` and outliving this runtime. CasRequests & farewell_requests_, CasRequests & lease_requests_, const Layout & layout_, @@ -476,7 +474,6 @@ class CasMountRuntime /// ---- injected environment (no `Pool` back-reference); initialized first, in this order ---- BackendPtr backend_ptr; - CasRequests & mount_requests; CasRequests & farewell_requests; CasRequests & lease_requests; const Layout & layout; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp index 295d3fc618f0..808f9d6bb6e9 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp @@ -231,8 +231,7 @@ Pool::Pool(BackendPtr backend_, PoolConfig config_, PoolMeta meta_) /// reader over that plane, never this one shared across two. The event sink is installed by the /// factory before writable mounting starts. , manifest_reader(mount_requests, pool_layout, meta, event_sink_, config.manifest_decode_cache_bytes) - /// Ref-log / ref-table subsystem, on the MOUNT plane: a ref-lane write and a mount-lease renewal - /// are then measured against the same fence and the same clock. Injected with the + /// Ref-log / ref-table subsystem, on the MOUNT plane. Injected with the /// RefLedgerConfig slice + the event-sink reference + the pool `cas_request_budget`, plus callbacks /// into the mount/watermark state that lives /// on `mount_runtime` (reached through Pool delegates). The callbacks capture `this`; they are @@ -255,13 +254,13 @@ Pool::Pool(BackendPtr backend_, PoolConfig config_, PoolMeta meta_) [this] (const RootNamespace & ns) { cancelInflightBuildsForNamespace(ns); }, config.recovery_pre_first_request_hook_for_test) /// Mount / write-fence / build-watermark / self-remount runtime. Injected with - /// backend/layout + the mount, farewell and lease planes + the `MountConfig` slice + `server_root_id` + the event-sink reference + the pool + /// backend/layout + the farewell and lease planes + the `MountConfig` slice + `server_root_id` + the event-sink reference + the pool /// `cas_request_budget` + the `remount_attempt` callback (== `Pool::tryRemountOnce`, whose claim/ /// recovery ORCHESTRATION stays on Pool). The callback captures `this`; it is invoked only at runtime /// (post-construction). Declared/constructed AFTER `ref_ledger`, preserving the original member order /// verbatim (mount destroyed first, ledger last; both orders proven safe -- see the header note). , mount_runtime( - pool_backend, mount_requests, farewell_requests, lease_requests, + pool_backend, farewell_requests, lease_requests, pool_layout, config.mountConfig(), config.server_root_id, event_sink_, config.cas_request_budget, [this] { return tryRemountOnce(); }) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefLedger.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefLedger.h index 73ce041c73da..0f53238dfdd3 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefLedger.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefLedger.h @@ -91,8 +91,7 @@ class CasRefLedger { public: CasRefLedger( - /// The mount plane. Every request this ledger makes is admitted on it, so a ref-lane write and - /// a mount-lease renewal are measured against the same fence and the same clock. + /// The mount plane. Every request this ledger makes is admitted on it. CasRequests & mount_requests_, const Layout & layout_, RefLedgerConfig config_, diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp index fbd40d3309cb..47c7419b1718 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp @@ -1324,7 +1324,7 @@ String describeRequestFailure(const std::exception & failure) } MountLeaseRenewer::MountLeaseRenewer( - CasRequests & mount_requests_, CasRequests & open_requests_, CasRequests & worker_requests_, + CasRequests & open_requests_, CasRequests & lease_requests_, const Layout & layout_, const String & srid_, UInt128 server_uuid_, uint64_t writer_epoch_, std::chrono::milliseconds ttl_, std::function now_ms_fn_, @@ -1332,9 +1332,8 @@ MountLeaseRenewer::MountLeaseRenewer( CasEventSink event_sink_, std::chrono::milliseconds lease_safety_margin_, std::function boot_ms_fn_) - : mount_requests(mount_requests_) - , open_requests(open_requests_) - , worker_requests(worker_requests_) + : open_requests(open_requests_) + , lease_requests(lease_requests_) , key(layout_.mountKey(srid_)) , srid(srid_) , server_uuid(server_uuid_) @@ -1579,13 +1578,6 @@ MountRenewResult MountLeaseRenewer::terminalResult(MountRenewResult result) } MountRenewResult MountLeaseRenewer::renew(const MountRenewOperationEnvironment & environment) -{ - return renewOn( - environment.policy == MountRenewPolicy::UntilDefinitive ? worker_requests : mount_requests, environment); -} - -MountRenewResult MountLeaseRenewer::renewOn( - CasRequests & plane, const MountRenewOperationEnvironment & environment) { const MountRenewObservabilityRegistration observability_registration = beginMountRenewObservabilityCall(); const MountRenewObservabilityCallGuard observability_guard(observability_registration); @@ -1624,7 +1616,7 @@ MountRenewResult MountLeaseRenewer::renewOn( MountRenewResult result; result.attempt_start_boot_ms = attempt_start_boot_ms; - CasOperation op = plane.admit(environment.live); + CasOperation op = lease_requests.admit(environment.live); if (environment.on_request) op.setRequestObserver([&on_request = environment.on_request](uint32_t attempt_no, const std::exception * failure) { @@ -1634,9 +1626,7 @@ MountRenewResult MountLeaseRenewer::renewOn( .failure_text = failure ? describeRequestFailure(*failure) : String{}, }); }); - const Retry policy = environment.policy == MountRenewPolicy::UntilDefinitive - ? Retry::untilDefinitive(kMountRenewRetrySpacingMs) - : Retry::untilLeaseSafe(confirmed_deadline_boot_ms, static_cast(lease_safety_margin.count())); + const Retry policy = Retry::untilDefinitive(kMountRenewRetrySpacingMs); std::optional written; try { @@ -1774,9 +1764,8 @@ void MountLeaseRenewer::terminate(CasOperation & op) ? std::numeric_limits::max() : doubled_reservation_ms + kFarewellSlackMs; const uint64_t farewell_window_ms = std::max(kFarewellBudgetMs, two_envelope_reservation_plus_slack_ms); - /// The derived window alone is not enough: mount-control activity must also never run past the - /// point this node's own fence may already be gone (the rule a bounded renewal enforces with - /// `Retry::untilLeaseSafe`). The precondition on this write already stops it from clobbering + /// The derived window alone is not enough: the farewell must also never run past the point this + /// node's own fence may already be gone. The precondition on this write already stops it from clobbering /// a successor if it DOES land late, but a shutdown holding the process open to retry a write past /// its own lease-safe deadline serves no one -- the successor's own reclaim does not wait for it. /// `confirmed_deadline_boot_ms` is set at `start()` and kept current by every successful `renew`, diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h index ac51c398a940..6e807ca17a80 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h @@ -61,14 +61,7 @@ struct MountRenewResult std::exception_ptr failure; }; -/// How far a renewal may retry. -enum class MountRenewPolicy : uint8_t -{ - LeaseBound, /// startup, remount and direct renewals: `Retry::untilLeaseSafe` - UntilDefinitive, /// the background worker: `Retry::untilDefinitive(kMountRenewRetrySpacingMs)` -}; - -/// The spacing of an `UntilDefinitive` renewal's retries. +/// The spacing of a renewal's retries. inline constexpr uint64_t kMountRenewRetrySpacingMs = 1000; /// One physical request of a renewal, reported as it happens: a `PUT` sent (`failed` false), a `PUT` @@ -90,7 +83,6 @@ struct MountRenewOperationEnvironment /// `NotAttempted` rather than terminal only when this node had already been asked to stop -- /// sampling it afterwards would read a flag that the refusal itself may have set. std::function cancelled; - MountRenewPolicy policy = MountRenewPolicy::LeaseBound; /// Called on the renewing thread for each `PUT` sent, each failed `PUT` and each failed resolve read. /// May be empty. What it throws is ignored. std::function on_request; @@ -549,19 +541,18 @@ bool isCreatorFenceTerminal(CasOperation & op, const Layout & layout, const Stri /// - foreign uuid → fail closed; /// - absent → `create`; expired-our-uuid (any epoch) → `replace` reclaim. /// -/// PLANES. A bounded renewal is admitted under the mount fence, because it writes under the authority -/// the fence tracks. The worker's renewal (`UntilDefinitive`) runs on a plane with no lease budget: it -/// keeps trying after the lease expired, and a stop, a remount request or a terminal lifecycle reaches -/// it through its liveness. The claim and the farewell are admitted off the fence: a self-remount claims with the -/// fence already latched lost, so a claim gated on the fence could never reclaim, and a farewell -/// refused because the fence has run down would leave the slot looking live until GC fences it out. -/// Neither is unguarded: a claim's safety is its own conditional write, and a caller that has shutdown -/// facts hands them over as a `Liveness`. +/// PLANES. A renewal runs on `lease_requests`, which has no lease budget: it keeps trying after the +/// lease expired, and a stop, a remount request or a terminal lifecycle reaches it through its +/// liveness. The claim and the farewell are admitted on `open_requests`, off the fence: a self-remount +/// claims with the fence already latched lost, so a claim gated on the fence could never reclaim, and a +/// farewell refused because the fence has run down would leave the slot looking live until GC fences it +/// out. Neither is unguarded: a claim's safety is its own conditional write, and a caller that has +/// shutdown facts hands them over as a `Liveness`. class MountLeaseRenewer { public: MountLeaseRenewer( - CasRequests & mount_requests_, CasRequests & open_requests_, CasRequests & worker_requests_, + CasRequests & open_requests_, CasRequests & lease_requests_, const Layout & layout_, const String & srid_, UInt128 server_uuid_, uint64_t writer_epoch_, std::chrono::milliseconds ttl_, std::function now_ms_fn_, @@ -575,9 +566,8 @@ class MountLeaseRenewer /// Adopt the already-claimed mount. Returns the exact pre-I/O BOOTTIME anchor. `liveness` carries /// the caller's shutdown terms; the mount fence is deliberately not consulted here. uint64_t start(Liveness liveness = {}); - /// The steady-state renewal. `LeaseBound` runs under the mount fence with `Retry::untilLeaseSafe`; - /// `UntilDefinitive` runs on the worker plane with `Retry::untilDefinitive(kMountRenewRetrySpacingMs)` - /// and ends on a definitive answer, a deterministic local failure, or when `environment.live` refuses. + /// One renewal on `lease_requests` under `Retry::untilDefinitive(kMountRenewRetrySpacingMs)`. Ends on + /// a definitive answer, a deterministic local failure, or when `environment.live` refuses. MountRenewResult renew(const MountRenewOperationEnvironment & environment); void release(); @@ -590,17 +580,14 @@ class MountLeaseRenewer /// The incarnation every guarded write of this slot names. Engaged for exactly the states that /// admit such a write: `start` establishes it and each committed renewal replaces it. const Etag & precondition() const; - /// One renewal admitted on `plane`. - MountRenewResult renewOn(CasRequests & plane, const MountRenewOperationEnvironment & environment); Etag claim(CasOperation & op, const String & body); [[noreturn]] void throwRenewConflict(const Observation & seen) const; MountRenewResult terminalResult(MountRenewResult result); void terminate(CasOperation & op); - CasRequests & mount_requests; CasRequests & open_requests; - /// The plane of an `UntilDefinitive` renewal: no lease budget, and a sleep a stop wakes. - CasRequests & worker_requests; + /// The plane of a renewal: no lease budget, and a sleep a stop wakes. + CasRequests & lease_requests; String key; String srid; diff --git a/src/Disks/tests/gtest_cas_event_log.cpp b/src/Disks/tests/gtest_cas_event_log.cpp index 012a1906fcaf..f1667869caaf 100644 --- a/src/Disks/tests/gtest_cas_event_log.cpp +++ b/src/Disks/tests/gtest_cas_event_log.cpp @@ -292,7 +292,6 @@ TEST(CASEvent, DeepReentrancyPreservesDeterministicPhysicalAttemptTruth) planes[index] = std::make_unique( backends[index], Fence::open(), [&] { return boot_ms; }, [&](uint64_t ms) { boot_ms += ms; }); renewers[index] = std::make_unique( - *planes[index], *planes[index], *planes[index], *layouts[index], diff --git a/src/Disks/tests/gtest_cas_gc_ack_floor.cpp b/src/Disks/tests/gtest_cas_gc_ack_floor.cpp index d0fff5bcff6b..ea63268a8fe3 100644 --- a/src/Disks/tests/gtest_cas_gc_ack_floor.cpp +++ b/src/Disks/tests/gtest_cas_gc_ack_floor.cpp @@ -933,7 +933,7 @@ void runExpiredMountFenceOutScenario(const PoolConfig & config) // never changes again. const String srid2 = "stale-server"; CasRequests renewer_requests = openRequestsForTest(backend); - MountLeaseRenewer srid2_renewer(renewer_requests, renewer_requests, renewer_requests, layout, srid2, DB::UInt128(0x2222), + MountLeaseRenewer srid2_renewer(renewer_requests, renewer_requests, layout, srid2, DB::UInt128(0x2222), /*writer_epoch=*/1, std::chrono::milliseconds(100), [] { return 1000u; }, [] { return 0u; }, {}, std::chrono::milliseconds(0), [] { return 0u; }); @@ -1061,7 +1061,7 @@ TEST(CASGCAckFloor, DefaultMonoClockTracksPoolsInjectedBootClockNotWallClock) // A stale mount, exactly as `ExpiredMountFencedOutAndExcluded`: one claim, never renewed again. const String srid2 = "stale-server"; CasRequests renewer_requests = openRequestsForTest(backend); - MountLeaseRenewer srid2_renewer(renewer_requests, renewer_requests, renewer_requests, layout, srid2, DB::UInt128(0x2222), + MountLeaseRenewer srid2_renewer(renewer_requests, renewer_requests, layout, srid2, DB::UInt128(0x2222), /*writer_epoch=*/1, std::chrono::milliseconds(100), [] { return 1000u; }, [fake_boot] diff --git a/src/Disks/tests/gtest_cas_heartbeat.cpp b/src/Disks/tests/gtest_cas_heartbeat.cpp index 67ca53cd6398..9f3f7e60f2a3 100644 --- a/src/Disks/tests/gtest_cas_heartbeat.cpp +++ b/src/Disks/tests/gtest_cas_heartbeat.cpp @@ -41,15 +41,12 @@ namespace /// The request planes this file's renewers run on. All are open-fence -- the exclusivity these tests /// exercise is the mount protocol's own, not a fence's -- on the same injected boot clock the renewer's /// lease deadline is expressed on, so they never disagree about how much budget is left. -/// `sleep_step_ms`, when set, makes one inter-attempt pause jump the clock past the lease bound: that -/// is how a test asks for exactly one physical attempt without a per-call attempt cap. It depends on -/// the engine checking the bound, sleeping, then checking again -- a reissue that slept first would -/// send a second attempt. `tests::OperationForTest` covers a fixture needing one operation, but -/// neither the planes a renewer takes nor this clock, which is why this stays local. +/// `tests::OperationForTest` covers a fixture needing one operation, but neither the planes a renewer +/// takes nor this clock, which is why this stays local. class Ops { public: - Ops(std::shared_ptr backend, uint64_t * boot_ms, uint64_t sleep_step_ms = 0) + Ops(std::shared_ptr backend, uint64_t * boot_ms) : mount(openRequestsForTest(backend)) , farewell(openRequestsForTest(backend)) , lease(openRequestsForTest(std::move(backend))) @@ -58,8 +55,7 @@ class Ops for (CasRequests * requests : {&mount, &farewell, &lease}) { requests->setNowFnForTest([boot_ms] { return *boot_ms; }); - requests->setSleepFnForTest( - [boot_ms, sleep_step_ms](uint64_t ms) { *boot_ms += sleep_step_ms ? sleep_step_ms : ms; }); + requests->setSleepFnForTest([boot_ms](uint64_t ms) { *boot_ms += ms; }); } } @@ -88,10 +84,7 @@ void seedOwnClaim(CasOperation & op, const Layout & l, const String & srid, UInt ASSERT_EQ(claimMount(op, l, srid, uuid, epoch, now_ms, ttl_ms).kind, MountClaimResult::Claimed); } -/// Not `final`: `EnvelopeEatingBackend` (the envelope-cutoff test below) derives from it to -/// reuse its `Attempt`/`attempts` bookkeeping while overriding `write`/`read` with its own always-fail -/// behavior instead of the scripted-action queue. -class RenewalScriptBackend : public InMemoryBackend +class RenewalScriptBackend final : public InMemoryBackend { public: enum class Action : uint8_t @@ -247,7 +240,6 @@ MountRenewOperationEnvironment renewalEnvironment( .boot_ms = [&boot_ms] { return boot_ms; }, .live = live, .cancelled = cancelled, - .policy = MountRenewPolicy::LeaseBound, .on_request = {}, }; } @@ -293,7 +285,7 @@ TEST(CASHeartbeat, AnchorCarriesFloor) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); - MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), [&] { return now_ms; }, [&] { return min_active_build_sequence_now; }, {}, std::chrono::milliseconds(0), [&] { return boot_ms; }); @@ -319,7 +311,7 @@ TEST(CASHeartbeat, RenewRereadsCallbackAndBumpsSeq) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); - MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), [&] { return now_ms; }, [&] { return min_active_build_sequence_now; }, {}, std::chrono::milliseconds(0), [&] { return boot_ms; }); @@ -347,7 +339,7 @@ TEST(CASHeartbeat, StopStampsExpiredAndFarewellSentinel) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); - MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(0), [&] { return boot_ms; }); @@ -403,7 +395,7 @@ TEST(CASHeartbeat, FarewellIsAdmittedUnderTheDefaultBudget) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/30000); - MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(30000), [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(2000), [&] { return boot_ms; }); @@ -436,7 +428,7 @@ TEST(CASHeartbeat, FarewellIsAdmittedUnderADifferentEnvelope) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/40000); - MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(40000), [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(2000), [&] { return boot_ms; }); @@ -470,7 +462,7 @@ TEST(CASHeartbeat, FarewellIsRefusedWhenTheLeaseExpiresBeforeItsDerivedWindow) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/5000); - MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(5000), [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(2000), [&] { return boot_ms; }); @@ -516,7 +508,7 @@ TEST(CASHeartbeat, ForeignIncarnationDuringFarewellLeavesTheSuccessorUntouchedAn Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); - MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(0), [&] { return boot_ms; }); @@ -573,7 +565,7 @@ TEST(CASHeartbeat, SameEpochUnfencedTouchIsUncertainNotFatal) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); - MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(0), [&] { return boot_ms; }); @@ -620,7 +612,7 @@ TEST(CASHeartbeat, SupersededTouchIsFailClosedNotFatal) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); - MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(0), [&] { return boot_ms; }); @@ -674,7 +666,7 @@ TEST(CASHeartbeat, ForeignUuidTouchFailsClosedWithoutAborting) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/100); - MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(0), [&] { return boot_ms; }); @@ -762,7 +754,7 @@ TEST(CASMountAudit, RenewerAdoptEmitsClaimAndTerminateEmitsRelease) std::vector seen; CasEventSink sink = [&](const CasEvent & e) { seen.push_back(e); }; - MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), [&] { return now_ms; }, [] { return uint64_t{5}; }, sink, std::chrono::milliseconds(0), [&] { return boot_ms; }); @@ -800,7 +792,7 @@ TEST(CASMountAudit, RenewerForeignConflictRefusesAndNamesHolder) ASSERT_EQ(claimMount(ops.op, layout, srid, uuid_x, /*our_epoch=*/1, now_ms, /*ttl_ms=*/100).kind, MountClaimResult::Claimed); - MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid_y, /*writer_epoch=*/1, + MountLeaseRenewer renewer(ops.farewell, ops.lease, layout, srid, uuid_y, /*writer_epoch=*/1, std::chrono::milliseconds(100), [&] { return now_ms; }, [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(2000), [&] { return boot_ms; }); @@ -842,7 +834,7 @@ TEST(CASMountAudit, RenewerAdoptRefusesFencedSelfWithTypedError) std::vector seen; CasEventSink sink = [&](const CasEvent & e) { seen.push_back(e); }; /// A renewer for the SAME (uuid, epoch) tries to adopt the now-fenced slot. - MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), [&] { return now_ms; }, [] { return uint64_t{5}; }, sink, std::chrono::milliseconds(2000), [&] { return boot_ms; }); @@ -882,7 +874,7 @@ TEST(CASHeartbeat, RenewOverFencedOwnSlotIsClassifiedNotForeign) std::vector seen; CasEventSink sink = [&](const CasEvent & e) { seen.push_back(e); }; - MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, + MountLeaseRenewer renewer(ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(100), [&] { return now_ms; }, [] { return uint64_t{5}; }, sink, std::chrono::milliseconds(0), [&] { return boot_ms; }); @@ -937,7 +929,7 @@ TEST(CASHeartbeat, RenewerStateAllowsOnlyActiveReleaseOrTerminal) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, "released", uuid, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, ops.lease, layout, "released", uuid, 9, std::chrono::milliseconds(1000), + ops.farewell, ops.lease, layout, "released", uuid, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{7}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); EXPECT_EQ(renewer.state(), MountLeaseRenewerState::New); @@ -957,17 +949,19 @@ TEST(CASHeartbeat, RenewerStateAllowsOnlyActiveReleaseOrTerminal) auto backend = std::make_shared(); uint64_t wall_ms = 1000; uint64_t boot_ms = 100; - /// One pause jumps the clock past the lease bound, so the ambiguous first attempt is the only - /// one this renewal ever sends and its verdict is the terminal one under test. - Ops ops(backend, &boot_ms, /*sleep_step_ms=*/10'000); + Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, "terminal", uuid, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, ops.lease, layout, "terminal", uuid, 9, std::chrono::milliseconds(1000), + ops.farewell, ops.lease, layout, "terminal", uuid, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{7}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); renewer.start(); backend->actions = {RenewalScriptBackend::Action::ThrowBefore}; - const MountRenewResult result = renewer.renew(renewalEnvironment(boot_ms)); + backend->read_calls = 0; + /// Live until the ambiguous attempt's resolving read has run, so that attempt is the only one + /// sent and the renewal ends terminal with it unsettled. + const MountRenewResult result = renewer.renew( + renewalEnvironment(boot_ms, /*live=*/[&] { return backend->read_calls == 0; })); EXPECT_EQ(result.outcome, MountRenewOutcome::Terminal); EXPECT_NE(result.failure, nullptr); EXPECT_EQ(renewer.state(), MountLeaseRenewerState::RenewalTerminal); @@ -990,7 +984,7 @@ TEST(CASHeartbeat, RenewalRetriesOneImmutableBodyAndAdoptsLostResponse) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, ops.lease, layout, srid, uuid, 9, std::chrono::milliseconds(1000), + ops.farewell, ops.lease, layout, srid, uuid, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{7}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); renewer.start(); @@ -1028,14 +1022,14 @@ TEST(CASHeartbeat, RenewalOverConnectFailuresRecoversWithoutASettleRead) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, srid, uuid, 9, wall_ms, 30000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, ops.lease, layout, srid, uuid, 9, std::chrono::milliseconds(30000), + ops.farewell, ops.lease, layout, srid, uuid, 9, std::chrono::milliseconds(30000), [&] { return wall_ms; }, [] { return uint64_t{7}; }, {}, std::chrono::milliseconds(2000), [&] { return boot_ms; }); renewer.start(); backend->attempts.clear(); backend->read_calls = 0; - /// Three seconds of "no free port" at 50 ms per hint, then the store answers. + /// Sixty connect failures, then the store answers. for (int i = 0; i < 60; ++i) backend->actions.push_back(RenewalScriptBackend::Action::ThrowConnectHint); backend->actions.push_back(RenewalScriptBackend::Action::Delegate); @@ -1050,34 +1044,6 @@ TEST(CASHeartbeat, RenewalOverConnectFailuresRecoversWithoutASettleRead) } #endif -TEST(CASHeartbeat, DeadlineBeforeSendTerminalizesWithTypedFailure) -{ - auto backend = std::make_shared(); - Layout layout("pool"); - uint64_t wall_ms = 1000; - uint64_t boot_ms = 100; - Ops ops(backend, &boot_ms); - seedOwnClaim(ops.op, layout, "test", UInt128{1}, 9, wall_ms, 100); - MountLeaseRenewer renewer( - ops.mount, ops.farewell, ops.lease, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(100), - [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), - [&] { return boot_ms; }); - renewer.start(); - backend->attempts.clear(); - backend->read_calls = 0; - boot_ms = 180; - const MountRenewResult result = renewer.renew(renewalEnvironment(boot_ms)); - const DB::Exception failure = terminalException(result); - EXPECT_EQ(failure.code(), DB::ErrorCodes::NETWORK_ERROR); - EXPECT_NE(failure.message().find("no attempt sent"), String::npos) << failure.message(); - EXPECT_NE(failure.message().find("external_lease_deadline"), String::npos) << failure.message(); - EXPECT_FALSE(result.sent_any); - ASSERT_TRUE(result.deadline_source.has_value()); - EXPECT_EQ(*result.deadline_source, GaveUp::Source::Lease); - EXPECT_TRUE(backend->attempts.empty()); - EXPECT_EQ(backend->read_calls, 0u) << "a pre-send terminal deadline must perform no diagnostic read"; -} - TEST(CASHeartbeat, CancellationBeforeSendIsNotAttemptedAndAllowsRelease) { auto backend = std::make_shared(); @@ -1087,7 +1053,7 @@ TEST(CASHeartbeat, CancellationBeforeSendIsNotAttemptedAndAllowsRelease) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, "test", UInt128{1}, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, ops.lease, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + ops.farewell, ops.lease, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); renewer.start(); @@ -1113,7 +1079,7 @@ TEST(CASHeartbeat, CancellationAfterSendIsTerminalAndForbidsRelease) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, "test", UInt128{1}, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, ops.lease, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + ops.farewell, ops.lease, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); renewer.start(); @@ -1142,7 +1108,7 @@ TEST(CASHeartbeat, SlowResolvedSuccessKeepsAttemptStartAnchor) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, "test", UInt128{1}, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, ops.lease, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + ops.farewell, ops.lease, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); renewer.start(); @@ -1167,7 +1133,7 @@ TEST(CASHeartbeat, SamePairTwinAndForeignOrSuccessorStayTerminal) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, "test", uuid, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, ops.lease, layout, "test", uuid, 9, std::chrono::milliseconds(1000), + ops.farewell, ops.lease, layout, "test", uuid, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); renewer.start(); @@ -1201,7 +1167,7 @@ TEST(CASHeartbeat, ExpectedPredecessorThenLateLandingIsAdoptedExactly) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, "test", UInt128{1}, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, ops.lease, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + ops.farewell, ops.lease, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); renewer.start(); @@ -1230,7 +1196,7 @@ TEST(CASHeartbeat, GcFenceAndVanishedMountStayTerminal) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, "test", UInt128{1}, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, ops.lease, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + ops.farewell, ops.lease, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); renewer.start(); @@ -1261,17 +1227,19 @@ TEST(CASHeartbeat, LateDeliveryAfterTerminalCannotRearmOrOverwriteSuccessor) auto backend = std::make_shared(); uint64_t wall_ms = 1000; uint64_t boot_ms = 100; - /// One pause jumps the clock past the lease bound, so the ambiguous first attempt is the only - /// one this renewal sends and the renewal ends terminal with that attempt still in flight. - Ops ops(backend, &boot_ms, /*sleep_step_ms=*/10'000); + Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, "before-reclaim", UInt128{1}, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, ops.lease, layout, "before-reclaim", UInt128{1}, 9, std::chrono::milliseconds(1000), + ops.farewell, ops.lease, layout, "before-reclaim", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, CasEventSink{}, std::chrono::milliseconds(20), [&] { return boot_ms; }); renewer.start(); backend->actions = {RenewalScriptBackend::Action::ThrowBeforeThenLandAfterResolve}; - const MountRenewResult result = renewer.renew(renewalEnvironment(boot_ms)); + backend->read_calls = 0; + /// Live until the resolving read has run: the delayed write lands during that read, and the + /// renewal ends terminal without a second attempt. + const MountRenewResult result = renewer.renew( + renewalEnvironment(boot_ms, /*live=*/[&] { return backend->read_calls == 0; })); EXPECT_EQ(result.outcome, MountRenewOutcome::Terminal); /// The delayed write landed during the resolving read. It carries this renewer's own epoch, and @@ -1285,10 +1253,10 @@ TEST(CASHeartbeat, LateDeliveryAfterTerminalCannotRearmOrOverwriteSuccessor) auto backend = std::make_shared(); uint64_t wall_ms = 1000; uint64_t boot_ms = 100; - Ops ops(backend, &boot_ms, /*sleep_step_ms=*/10'000); + Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, "after-successor", UInt128{1}, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, ops.lease, layout, "after-successor", UInt128{1}, 9, std::chrono::milliseconds(1000), + ops.farewell, ops.lease, layout, "after-successor", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); renewer.start(); @@ -1299,7 +1267,9 @@ TEST(CASHeartbeat, LateDeliveryAfterTerminalCannotRearmOrOverwriteSuccessor) = ops.op.read(layout.mountKey("after-successor"), Retry::standard())->etag; backend->actions = {RenewalScriptBackend::Action::ThrowBefore}; - const MountRenewResult result = renewer.renew(renewalEnvironment(boot_ms)); + backend->read_calls = 0; + const MountRenewResult result = renewer.renew( + renewalEnvironment(boot_ms, /*live=*/[&] { return backend->read_calls == 0; })); ASSERT_EQ(result.outcome, MountRenewOutcome::Terminal); ASSERT_FALSE(backend->attempts.empty()); const auto delayed = backend->attempts.back(); @@ -1314,7 +1284,7 @@ TEST(CASHeartbeat, LateDeliveryAfterTerminalCannotRearmOrOverwriteSuccessor) ASSERT_EQ(claimMount(ops.op, layout, "after-successor", UInt128{1}, 10, wall_ms, 1000).kind, MountClaimResult::Claimed); MountLeaseRenewer successor( - ops.mount, ops.farewell, ops.lease, layout, "after-successor", UInt128{1}, 10, std::chrono::milliseconds(1000), + ops.farewell, ops.lease, layout, "after-successor", UInt128{1}, 10, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); successor.start(); @@ -1336,7 +1306,7 @@ TEST(CASHeartbeat, WallClockStepsAndBootSuspendCannotExtendAuthority) Ops ops(backend, &boot_ms); seedOwnClaim(ops.op, layout, "test", UInt128{1}, 9, wall_ms, 1000); MountLeaseRenewer renewer( - ops.mount, ops.farewell, ops.lease, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), + ops.farewell, ops.lease, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(1000), [&] { return wall_ms; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); renewer.start(); @@ -1349,76 +1319,17 @@ TEST(CASHeartbeat, WallClockStepsAndBootSuspendCannotExtendAuthority) backend->attempts.clear(); boot_ms += 10'000; const MountRenewResult suspended = renewer.renew(renewalEnvironment(boot_ms)); - const DB::Exception failure = terminalException(suspended); - EXPECT_EQ(failure.code(), DB::ErrorCodes::NETWORK_ERROR); - EXPECT_TRUE(backend->attempts.empty()) << "suspend-sized BOOTTIME overshoot must close admission"; -} - -/// Every attempt costs the whole envelope (attempt 100 + 2 * cap 50 = 200 ms) and fails ambiguously. -/// Under a 1000 ms lease with a 100 ms margin the renewal must stop issuing before the cutoff rather -/// than start an attempt that cannot finish inside it. -namespace -{ -/// Bypasses `RenewalScriptBackend`'s scripted-action queue for a guarded mount write and instead -/// always fails it (and every read) once armed, each failure costing the whole envelope on the -/// injected boot clock. Left unarmed during `seedOwnClaim` (an unconditional read then an unguarded -/// create -- neither is a guarded mount write, but the read would still hit the always-throwing -/// override below) and during `renewer.start()`'s adopt read, so the fixture itself can land. -struct EnvelopeEatingBackend : RenewalScriptBackend -{ - uint64_t * boot_ms = nullptr; - bool armed = false; - uint64_t attemptTimeoutMs() const override { return 100; } - uint64_t attemptEnvelopeMs() const override { return 200; } - std::expected write(const String & key, const String & bytes, - const std::optional & expected_value, TransportAccess & access) override - { - if (armed && expected_value && key.ends_with("/mount")) - { - attempts.push_back({key, bytes, expected_value}); - *boot_ms += 200; - throw Poco::TimeoutException("the whole envelope, gone"); - } - return InMemoryBackend::write(key, bytes, expected_value, access); - } - std::optional read(const String & key, TransportAccess & access) override - { - if (armed) - { - *boot_ms += 200; - throw Poco::TimeoutException("the read too"); - } - return InMemoryBackend::read(key, access); - } -}; -} - -TEST(CASHeartbeat, RenewalStopsBeforeTheCutoffWhenEveryAttemptConsumesTheEnvelope) -{ - auto backend = std::make_shared(); - uint64_t wall_ms = 1000; - uint64_t boot_ms = 100; - backend->boot_ms = &boot_ms; - Layout layout("pool"); - Ops ops(backend, &boot_ms); - seedOwnClaim(ops.op, layout, "test", UInt128{0x1234}, 9, wall_ms, 1000); - MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, "test", UInt128{0x1234}, 9, std::chrono::milliseconds(1000), - [&] { return wall_ms; }, [] { return uint64_t{7}; }, {}, std::chrono::milliseconds(100), - [&] { return boot_ms; }); - renewer.start(); - const uint64_t cutoff = renewer.lastCommittedAttemptStartBootMs() + 1000 - 100; - backend->attempts.clear(); - backend->armed = true; - const MountRenewResult result = renewer.renew(renewalEnvironment(boot_ms)); - EXPECT_EQ(result.outcome, MountRenewOutcome::Terminal); - EXPECT_LE(boot_ms, cutoff) << "the last attempt started inside the cutoff and the engine did not start one that could not finish"; + ASSERT_EQ(suspended.outcome, MountRenewOutcome::Committed); + EXPECT_EQ(suspended.attempt_start_boot_ms, boot_ms) + << "a renewal after a suspend anchors at its own start, never at the deadline it last confirmed"; + EXPECT_EQ(renewer.lastCommittedAttemptStartBootMs(), boot_ms); + EXPECT_EQ(backend->attempts.size(), 1u); } namespace { /// The production shape at test scale: a 30 s lease, renewed one 10 s period after its anchor, with a -/// 2 s safety margin and the 7 s attempt envelope a bounded renewal reserves twice before each -/// request. That leaves a bounded renewal about 4 s of retries; an `UntilDefinitive` one has no bound. +/// 2 s safety margin and a 7 s attempt envelope. class UnboundedRenewalFixture { public: @@ -1434,7 +1345,7 @@ class UnboundedRenewalFixture ops = std::make_unique(backend, &boot_ms); seedOwnClaim(ops->op, layout, "test", UInt128{1}, 9, wall_ms, ttl_ms); renewer = std::make_unique( - ops->mount, ops->farewell, ops->lease, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(ttl_ms), + ops->farewell, ops->lease, layout, "test", UInt128{1}, 9, std::chrono::milliseconds(ttl_ms), [this] { return wall_ms; }, [] { return uint64_t{0}; }, [this](CasEvent event) { events.push_back(std::move(event)); }, std::chrono::milliseconds(margin_ms), [this] { return boot_ms; }); @@ -1450,7 +1361,6 @@ class UnboundedRenewalFixture UnboundedRenewalFixture & operator=(const UnboundedRenewalFixture &) = delete; MountRenewResult renew( - MountRenewPolicy policy, const std::function & live = {}, const std::function & cancelled = {}, std::function on_request = {}) @@ -1467,7 +1377,6 @@ class UnboundedRenewalFixture return !live || live(); }; MountRenewOperationEnvironment environment = renewalEnvironment(boot_ms, bounded_live, cancelled); - environment.policy = policy; environment.on_request = std::move(on_request); MountRenewResult result = renewer->renew(environment); EXPECT_FALSE(request_bound_hit) << "the renewal sent " << max_requests @@ -1508,13 +1417,12 @@ bool inSpacing(uint64_t gap_ms) } } -/// An outage longer than the cutoff a bounded renewal reserves for is retried to its end. TEST(CASHeartbeat, RenewalOutlivesTheReservationCutoff) { UnboundedRenewalFixture f; f.backend->outage = [&] { return f.boot_ms < f.renewal_start + 8'000; }; - const MountRenewResult result = f.renew(MountRenewPolicy::UntilDefinitive); + const MountRenewResult result = f.renew(); ASSERT_EQ(result.outcome, MountRenewOutcome::Committed); EXPECT_EQ(f.renewer->state(), MountLeaseRenewerState::Active); @@ -1529,7 +1437,7 @@ TEST(CASHeartbeat, RenewalSendsOneTupleOnEveryAttempt) UnboundedRenewalFixture f; f.backend->outage = [&] { return f.boot_ms < f.renewal_start + 8'000; }; - const MountRenewResult result = f.renew(MountRenewPolicy::UntilDefinitive); + const MountRenewResult result = f.renew(); ASSERT_EQ(result.outcome, MountRenewOutcome::Committed); const auto & attempts = f.backend->attempts; @@ -1551,7 +1459,7 @@ TEST(CASHeartbeat, RenewalSucceedsPastTheDeadlineWithItsFirstStart) UnboundedRenewalFixture f; f.backend->outage = [&] { return f.boot_ms < f.anchor + UnboundedRenewalFixture::ttl_ms + 15'000; }; - const MountRenewResult result = f.renew(MountRenewPolicy::UntilDefinitive); + const MountRenewResult result = f.renew(); ASSERT_EQ(result.outcome, MountRenewOutcome::Committed); EXPECT_EQ(result.attempt_start_boot_ms, f.renewal_start); @@ -1568,7 +1476,7 @@ TEST(CASHeartbeat, LandedAttemptIsAdoptedAfterALongOutage) std::vector reported; const MountRenewResult result = f.renew( - MountRenewPolicy::UntilDefinitive, {}, {}, [&](const MountRenewRequestEvent & event) { reported.push_back(event); }); + {}, {}, [&](const MountRenewRequestEvent & event) { reported.push_back(event); }); const uint64_t reads = f.backend->read_calls; ASSERT_EQ(result.outcome, MountRenewOutcome::Committed); @@ -1652,7 +1560,7 @@ TEST(CASHeartbeat, DefinitiveAnswersStayTerminalPastTheDeadline) /// A definitive answer that was retried would never end this renewal; the bound makes it fail instead. const uint64_t live_until = f.boot_ms + 600'000; - const MountRenewResult result = f.renew(MountRenewPolicy::UntilDefinitive, [&] { return f.boot_ms < live_until; }); + const MountRenewResult result = f.renew([&] { return f.boot_ms < live_until; }); const DB::Exception failure = terminalException(result); EXPECT_EQ(f.renewer->state(), MountLeaseRenewerState::RenewalTerminal); @@ -1723,7 +1631,7 @@ TEST(CASHeartbeat, AFenceSeenAfterALongOutageEndsTheRenewal) /// A fence that was retried would never end this renewal; the bound makes it fail instead. const MountRenewResult result = f.renew( - MountRenewPolicy::UntilDefinitive, [&] { return f.boot_ms < f.renewal_start + 600'000; }); + [&] { return f.boot_ms < f.renewal_start + 600'000; }); const DB::Exception failure = terminalException(result); EXPECT_NE(failure.message().find("fenced by GC"), String::npos) << failure.message(); @@ -1752,7 +1660,7 @@ TEST(CASHeartbeat, ARefusalAfterAnUnclearAttemptIsRetriedUntilStopped) std::vector reported; const MountRenewResult result = f.renew( - MountRenewPolicy::UntilDefinitive, /*live=*/[&] { return !stopped; }, /*cancelled=*/[&] { return stopped; }, + /*live=*/[&] { return !stopped; }, /*cancelled=*/[&] { return stopped; }, [&](const MountRenewRequestEvent & event) { reported.push_back(event); }); const DB::Exception failure = terminalException(result); @@ -1789,7 +1697,7 @@ TEST(CASHeartbeat, RenewalSpacesRetries) f.backend->outage = [&] { return f.boot_ms < f.renewal_start + 30'500; }; f.backend->outage_action = RenewalScriptBackend::Action::ThrowConnectHint; - ASSERT_EQ(f.renew(MountRenewPolicy::UntilDefinitive).outcome, MountRenewOutcome::Committed); + ASSERT_EQ(f.renew().outcome, MountRenewOutcome::Committed); ASSERT_GE(log.size(), 4u); /// The fuse: its settling read and its reissue follow at once. @@ -1814,7 +1722,7 @@ TEST(CASHeartbeat, RenewalSpacesRetries) f.backend->read_actions = {RenewalScriptBackend::Action::ThrowBefore, RenewalScriptBackend::Action::ThrowBefore}; f.backend->outage = [&] { return f.boot_ms < f.renewal_start + 30'500; }; - ASSERT_EQ(f.renew(MountRenewPolicy::UntilDefinitive).outcome, MountRenewOutcome::Committed); + ASSERT_EQ(f.renew().outcome, MountRenewOutcome::Committed); ASSERT_GE(log.size(), 6u); const uint64_t t0 = f.renewal_start; @@ -1849,7 +1757,7 @@ TEST(CASHeartbeat, RenewalSpacesRetries) f.backend->actions = {RenewalScriptBackend::Action::ThrowBefore}; f.backend->read_actions = {RenewalScriptBackend::Action::ThrowFirstAttemptFuse}; - ASSERT_EQ(f.renew(MountRenewPolicy::UntilDefinitive).outcome, MountRenewOutcome::Committed); + ASSERT_EQ(f.renew().outcome, MountRenewOutcome::Committed); ASSERT_EQ(log.size(), 4u); EXPECT_EQ(log[1], std::make_pair('R', f.renewal_start)); @@ -1873,7 +1781,7 @@ TEST(CASHeartbeat, RenewalSpacesRetries) f.backend->on_read = [&] { log.emplace_back('R', f.boot_ms); }; f.backend->outage = [&] { return failing; }; - ASSERT_EQ(f.renew(MountRenewPolicy::UntilDefinitive).outcome, MountRenewOutcome::Committed); + ASSERT_EQ(f.renew().outcome, MountRenewOutcome::Committed); ASSERT_GE(log.size(), 3u); std::optional previous_put; @@ -1908,7 +1816,7 @@ TEST(CASHeartbeat, StopDuringARetryWaitEndsTheRenewal) }); const MountRenewResult result = f.renew( - MountRenewPolicy::UntilDefinitive, /*live=*/[&] { return !stopped; }, /*cancelled=*/[&] { return stopped; }); + /*live=*/[&] { return !stopped; }, /*cancelled=*/[&] { return stopped; }); const DB::Exception failure = terminalException(result); EXPECT_EQ(failure.code(), DB::ErrorCodes::NETWORK_ERROR) << failure.message(); @@ -1923,7 +1831,7 @@ TEST(CASHeartbeat, StopDuringARetryWaitEndsTheRenewal) UnboundedRenewalFixture f; const MountRenewResult result = f.renew( - MountRenewPolicy::UntilDefinitive, /*live=*/[] { return false; }, /*cancelled=*/[] { return false; }); + /*live=*/[] { return false; }, /*cancelled=*/[] { return false; }); (void)terminalException(result); EXPECT_TRUE(f.backend->attempts.empty()); @@ -1938,7 +1846,7 @@ TEST(CASHeartbeat, RenewalReportsEachRequestAsItHappens) std::vector reported; const MountRenewResult result = f.renew( - MountRenewPolicy::UntilDefinitive, {}, {}, [&](const MountRenewRequestEvent & event) { reported.push_back(event); }); + {}, {}, [&](const MountRenewRequestEvent & event) { reported.push_back(event); }); ASSERT_EQ(result.outcome, MountRenewOutcome::Committed); const std::vector> expected{{1, false}, {1, true}, {2, false}, {2, true}, {3, false}}; @@ -1961,7 +1869,7 @@ TEST(CASHeartbeat, AThrowingRequestReportChangesNoOutcome) f.backend->actions = {RenewalScriptBackend::Action::ThrowBefore, RenewalScriptBackend::Action::ThrowBefore}; const MountRenewResult result = f.renew( - MountRenewPolicy::UntilDefinitive, {}, {}, + {}, {}, [](const MountRenewRequestEvent &) { throw std::runtime_error("injected report failure"); }); ASSERT_EQ(result.outcome, MountRenewOutcome::Committed); diff --git a/src/Disks/tests/gtest_cas_mount.cpp b/src/Disks/tests/gtest_cas_mount.cpp index 5af96b8b23ee..c3037c9716be 100644 --- a/src/Disks/tests/gtest_cas_mount.cpp +++ b/src/Disks/tests/gtest_cas_mount.cpp @@ -731,7 +731,7 @@ TEST(CASMountLease, AbsentClaimThenRenewBumpsSeq) Ops ops(b, &boot); auto r = claimMount(ops.op, l, "r", UInt128(1), /*epoch*/ 7, now, /*ttl*/ 100); EXPECT_EQ(r.kind, MountClaimResult::Claimed); - MountLeaseRenewer k(ops.mount, ops.farewell, ops.lease, l, "r", UInt128(1), 7, std::chrono::milliseconds(100), + MountLeaseRenewer k(ops.farewell, ops.lease, l, "r", UInt128(1), 7, std::chrono::milliseconds(100), [&] { return now; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(0), [&] { return boot; }); k.start(); @@ -751,7 +751,7 @@ TEST(CASMountLease, HolderBodiesMintFreshAttemptIdsAndFenceCopiesIt) const String key = layout.mountKey("r"); const MountLease claimed = decodeMountLease(ops.op.read(key, Retry::standard())->bytes); - MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, layout, "r", UInt128{1}, 7, std::chrono::milliseconds(100), + MountLeaseRenewer renewer(ops.farewell, ops.lease, layout, "r", UInt128{1}, 7, std::chrono::milliseconds(100), [&] { return now; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(0), [&] { return boot; }); renewer.start(); @@ -807,7 +807,7 @@ TEST(CASMountLease, VanishedBackingStoreStopsRenewalWithoutLogicalError) uint64_t boot = 0; Ops ops(b, &boot); ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), /*epoch*/ 7, now, /*ttl*/ 100).kind, MountClaimResult::Claimed); - MountLeaseRenewer k(ops.mount, ops.farewell, ops.lease, l, "r", UInt128(1), 7, std::chrono::milliseconds(100), + MountLeaseRenewer k(ops.farewell, ops.lease, l, "r", UInt128(1), 7, std::chrono::milliseconds(100), [&] { return now; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(0), [&] { return boot; }); k.start(); @@ -850,7 +850,7 @@ TEST(CASMountLease, TerminateAfterVanishedBackingStoreIsNoOpRelease) uint64_t now = 1000; Ops ops(b); ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), /*epoch*/ 7, now, /*ttl*/ 100).kind, MountClaimResult::Claimed); - MountLeaseRenewer k(ops.mount, ops.farewell, ops.lease, l, "r", UInt128(1), 7, std::chrono::milliseconds(100), + MountLeaseRenewer k(ops.farewell, ops.lease, l, "r", UInt128(1), 7, std::chrono::milliseconds(100), [&] { return now; }, [] { return uint64_t{0}; }); k.start(); @@ -1162,7 +1162,7 @@ TEST(CASMountLease, RenewerStartAdoptsOurOwnClaimNotDoubleStart) Ops ops(b); // The normal flow: claimMount writes the live mount under (uuid=1, epoch=7), THEN renewer.start(). ASSERT_EQ(claimMount(ops.op, l, "r", UInt128(1), /*epoch*/ 7, now, /*ttl*/ 100).kind, MountClaimResult::Claimed); - MountLeaseRenewer k(ops.mount, ops.farewell, ops.lease, l, "r", UInt128(1), /*epoch*/ 7, std::chrono::milliseconds(100), + MountLeaseRenewer k(ops.farewell, ops.lease, l, "r", UInt128(1), /*epoch*/ 7, std::chrono::milliseconds(100), [&] { return now; }, [] { return uint64_t{0}; }); EXPECT_NO_THROW(k.start()); // adopts our own live (uuid=1,epoch=7) mount — NOT a double-start EXPECT_EQ(decodeMountLease(ops.op.read(l.mountKey("r"), Retry::standard())->bytes).writer_epoch, 7u); @@ -2124,7 +2124,7 @@ TEST(CASMountObservation, RenewalDuringObservationRestartsIt) /// wrote (no seq bump, per the ADOPT RULE), then a synchronous renewal mints a new incarnation /// mid-observation. uint64_t renewer_wall = 500; - MountLeaseRenewer renewer(ops.mount, ops.farewell, ops.lease, l, "r", UInt128(1), 7, std::chrono::milliseconds(500), + MountLeaseRenewer renewer(ops.farewell, ops.lease, l, "r", UInt128(1), 7, std::chrono::milliseconds(500), [&] { return renewer_wall; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(0), [&] { return renewer_boot; }); renewer.start(); @@ -2313,7 +2313,7 @@ TEST(CASMountLease, ClaimAdoptIsTwoRequests) /// The absent-slot mint. backend->reads = backend->heads = backend->writes = 0; - MountLeaseRenewer minting(ops.mount, ops.farewell, ops.lease, l, "fresh", UInt128(1), 7, + MountLeaseRenewer minting(ops.farewell, ops.lease, l, "fresh", UInt128(1), 7, std::chrono::milliseconds(100), [&] { return now; }, [] { return uint64_t{0}; }); minting.start(); EXPECT_EQ(backend->reads, 1u); @@ -2324,7 +2324,7 @@ TEST(CASMountLease, ClaimAdoptIsTwoRequests) ASSERT_EQ(claimMount(ops.op, l, "adopted", UInt128(1), /*epoch*/ 7, now, /*ttl*/ 100).kind, MountClaimResult::Claimed); backend->reads = backend->heads = backend->writes = 0; - MountLeaseRenewer adopting(ops.mount, ops.farewell, ops.lease, l, "adopted", UInt128(1), 7, + MountLeaseRenewer adopting(ops.farewell, ops.lease, l, "adopted", UInt128(1), 7, std::chrono::milliseconds(100), [&] { return now; }, [] { return uint64_t{0}; }); adopting.start(); EXPECT_EQ(backend->reads, 1u); @@ -2332,11 +2332,8 @@ TEST(CASMountLease, ClaimAdoptIsTwoRequests) EXPECT_EQ(backend->heads, 0u); } -/// A mount whose fence has dropped must still hand its slot back: the renewal is refused (it would be -/// writing under authority this node no longer holds), while the farewell runs on the open plane and -/// lands. Deliberately two renewers: `release` is admitted only from `Active`, so a renewer whose -/// renewal already went terminal never reaches its own farewell -- the ordering the two halves below -/// pin separately. +/// The farewell is admitted on the claim-and-farewell plane, never on the renewal plane, so a renewal +/// plane that refuses everything cannot stop it. TEST(CASMountLease, FarewellRunsOnAnOpenFenceAfterTheMountFenceIsLost) { auto backend = std::make_shared(); @@ -2360,25 +2357,15 @@ TEST(CASMountLease, FarewellRunsOnAnOpenFenceAfterTheMountFenceIsLost) open_requests.setSleepFnForTest([&boot](uint64_t ms) { boot += ms; }); CasOperation seed = open_requests.admit(); - ASSERT_EQ(claimMount(seed, l, "renewing", UInt128(1), 7, now, /*ttl*/ 1000).kind, MountClaimResult::Claimed); ASSERT_EQ(claimMount(seed, l, "departing", UInt128(1), 7, now, /*ttl*/ 1000).kind, MountClaimResult::Claimed); - MountLeaseRenewer renewing(mount_requests, open_requests, open_requests, l, "renewing", UInt128(1), 7, - std::chrono::milliseconds(1000), [&] { return now; }, [] { return uint64_t{0}; }, - {}, std::chrono::milliseconds(0), [&] { return boot; }); - MountLeaseRenewer departing(mount_requests, open_requests, open_requests, l, "departing", UInt128(1), 7, + MountLeaseRenewer departing(open_requests, mount_requests, l, "departing", UInt128(1), 7, std::chrono::milliseconds(1000), [&] { return now; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(0), [&] { return boot; }); - renewing.start(); departing.start(); fence_lost = true; - const MountRenewResult refused = renewing.renew(MountRenewOperationEnvironment{}); - EXPECT_EQ(refused.outcome, MountRenewOutcome::Terminal); - EXPECT_FALSE(refused.sent_any); - EXPECT_FALSE(renewing.canRelease()) << "a terminal renewal leaves no farewell to run"; - EXPECT_NO_THROW(departing.release()); const MountLease farewell = decodeMountLease(seed.read(l.mountKey("departing"), Retry::standard())->bytes); EXPECT_EQ(farewell.min_active_build_sequence, std::numeric_limits::max()); @@ -2404,7 +2391,7 @@ TEST(CASMountLease, ClaimIsNotAdmittedUnderTheMountFence) open_requests.setNowFnForTest([&boot] { return boot; }); open_requests.setSleepFnForTest([&boot](uint64_t ms) { boot += ms; }); - MountLeaseRenewer renewer(mount_requests, open_requests, open_requests, l, "r", UInt128(1), 7, + MountLeaseRenewer renewer(open_requests, mount_requests, l, "r", UInt128(1), 7, std::chrono::milliseconds(1000), [&] { return now; }, [] { return uint64_t{0}; }, {}, std::chrono::milliseconds(0), [&] { return boot; }); EXPECT_NO_THROW(renewer.start()); diff --git a/src/Disks/tests/gtest_cas_mount_claim_conflicts.cpp b/src/Disks/tests/gtest_cas_mount_claim_conflicts.cpp index f7c974073a49..c4a3f2e216f0 100644 --- a/src/Disks/tests/gtest_cas_mount_claim_conflicts.cpp +++ b/src/Disks/tests/gtest_cas_mount_claim_conflicts.cpp @@ -17,7 +17,7 @@ namespace /// One renewer for the mount slot of server-root "r", under (uuid=1, epoch=7) unless overridden. All /// of its planes are the same open-fence one: what these tests exercise is the mount protocol's own -/// exclusivity, not a fence's, and no test here renews, which is the only caller of the mount plane. +/// exclusivity, not a fence's. MountLeaseRenewer makeRenewer( CasRequests & requests, uint64_t & now, @@ -25,7 +25,6 @@ MountLeaseRenewer makeRenewer( uint64_t epoch = 7) { return MountLeaseRenewer( - requests, requests, requests, Layout("p"), diff --git a/src/Disks/tests/gtest_cas_mount_runtime.cpp b/src/Disks/tests/gtest_cas_mount_runtime.cpp index 50f1f30ee31d..f0804bb39701 100644 --- a/src/Disks/tests/gtest_cas_mount_runtime.cpp +++ b/src/Disks/tests/gtest_cas_mount_runtime.cpp @@ -28,14 +28,10 @@ class RuntimeFixture explicit RuntimeFixture(uint64_t lease_safety_margin_ms, uint64_t attempt_timeout_ms = 10, std::optional connect_timeout_cap_ms = std::nullopt) : backend(std::make_shared()) - , mount(backend, Fence{ - [this] { return runtime.fenceGeneration(); }, - [this](uint64_t g, uint64_t needed) { return runtime.admit(g, needed); }, - [this](uint64_t g) { runtime.checkFenceOrThrow(g); }}) , farewell(backend, Fence::open()) , lease(backend, Fence::open()) , runtime( - backend, mount, farewell, lease, layout, + backend, farewell, lease, layout, MountConfig{.boot_ms_fn = [this] { return boot_ms; }}, "test", sink, CasRequestBudget{.attempt_timeout_ms = attempt_timeout_ms, @@ -53,7 +49,6 @@ class RuntimeFixture std::shared_ptr backend; Layout layout{"mount-runtime-admit"}; CasEventSink sink; - CasRequests mount; CasRequests farewell; CasRequests lease; CasMountRuntime runtime; diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp index 6f3f5a01ef59..e9bdbaf4b30e 100644 --- a/src/Disks/tests/gtest_cas_pool.cpp +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -2064,21 +2064,15 @@ class RuntimeRenewBackend final : public DB::Cas::tests::CountingBackend CasRequestBudget runtimeRenewBudget(); -/// A directly-constructed `CasMountRuntime` plus the request planes it needs. `Pool` builds those -/// from its own members; a test has no `Pool`, so the mount plane's fence reaches the runtime through -/// this holder -- the closures run only once the runtime is issuing requests, well after construction. +/// A directly-constructed `CasMountRuntime` plus the request planes `Pool` would give it. class RuntimeUnderTest { public: template RuntimeUnderTest(const std::shared_ptr & backend, Args &&... args) - : mount(backend, DB::Cas::Fence{ - [this] { return runtime.fenceGeneration(); }, - [this](uint64_t g, uint64_t needed) { return runtime.admit(g, needed); }, - [this](uint64_t g) { runtime.checkFenceOrThrow(g); }}) - , farewell(backend, DB::Cas::Fence::open()) + : farewell(backend, DB::Cas::Fence::open()) , lease(backend, DB::Cas::Fence::open()) - , runtime(backend, mount, farewell, lease, std::forward(args)...) + , runtime(backend, farewell, lease, std::forward(args)...) { /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the /// budget field alone; every construction of this holder pairs the two via `runtimeRenewBudget`, @@ -2088,7 +2082,6 @@ class RuntimeUnderTest /// against the clock it reads. Production runs both on `CLOCK_BOOTTIME`, so they agree; a test /// that injects one MUST inject the other, or `Retry::untilLeaseSafe` compares a synthetic /// deadline against real boottime, finds it long past, and refuses every request unsent. - mount.setNowFnForTest([this] { return runtime.bootMsNow(); }); farewell.setNowFnForTest([this] { return runtime.bootMsNow(); }); lease.setNowFnForTest([this] { return runtime.bootMsNow(); }); /// As `Pool` wires it: a stop wakes the worker renewal's wait. @@ -2112,17 +2105,13 @@ class RuntimeUnderTest CasMountRuntime & operator*() { return runtime; } - /// The retry wait of the mount and lease planes: the worker's renewal runs on the lease plane, the - /// other bounded ones on the mount plane. A remount renewal runs on the farewell plane, which this - /// does not reach. Call before the workers start. + /// The retry wait of the lease plane, where the renewal runs. Call before the lease thread starts. void setRetrySleepForTest(const std::function & sleep_fn) { - mount.setSleepFnForTest(sleep_fn); lease.setSleepFnForTest(sleep_fn); } private: - DB::Cas::CasRequests mount; DB::Cas::CasRequests farewell; DB::Cas::CasRequests lease; CasMountRuntime runtime; From 8ba79c80237816d587064d4fe2f374bebe119ee9 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 08:45:59 +0200 Subject: [PATCH 17/37] Remove the CAS mount renewal deadline counter and classifications No renewal is bounded by the lease or by a policy window, so CASMountRenewalDeadlineExceeded, the external_lease_deadline and request_deadline classifications and MountRenewResult::deadline_source cannot occur. A counter that cannot move reads as healthy, so they go, with their rows in the operations docs. Co-Authored-By: Claude Opus 5.5 --- .../cas/architecture/mounts-and-leases.md | 12 ++++------ docs/en/antalya/cas/operations/debugging.md | 14 ++++------- docs/en/antalya/cas/operations/monitoring.md | 9 ++++--- .../antalya/cas/operations/troubleshooting.md | 24 +++++++------------ src/Common/ProfileEvents.cpp | 1 - .../ContentAddressed/Pool/CasMountRuntime.cpp | 4 ---- .../ContentAddressed/Pool/CasServerRoot.cpp | 15 +++--------- .../ContentAddressed/Pool/CasServerRoot.h | 3 --- src/Disks/tests/gtest_cas_observability.cpp | 9 ++----- .../test_cas_mount_renewal_retry/test.py | 3 --- 10 files changed, 27 insertions(+), 67 deletions(-) diff --git a/docs/en/antalya/cas/architecture/mounts-and-leases.md b/docs/en/antalya/cas/architecture/mounts-and-leases.md index 7a98fab88682..e6fc591864e4 100644 --- a/docs/en/antalya/cas/architecture/mounts-and-leases.md +++ b/docs/en/antalya/cas/architecture/mounts-and-leases.md @@ -88,9 +88,7 @@ watermark — there is no separate watermark object. `MountLease` fields: `serve suspend correctly observes itself expired. The deadline decides about writes: none is admitted past it, and none is admitted once the remaining lease cannot cover the requests it may send plus the safety margin. The background renewal does not stop at the deadline. It retries timeouts, `5xx` - answers and connection errors about a second apart until the store answers. The renewals at startup, - after a remount and the direct renewal stay bounded: they stop at the last confirmed deadline minus - the safety margin. A retry, `GET`, response timestamp, or wall-clock step never extends authority. + answers and connection errors about a second apart until the store answers. A retry, `GET`, response timestamp, or wall-clock step never extends authority. - **Cadence.** The runtime normally starts a logical renewal every `cas_mount_renew_period_ms` (default 10 s), with TTL `cas_mount_lease_ttl_ms` (default 30 s, TTL/3 renewal ratio). The next beat is anchored at the committed body's pre-I/O BOOTTIME start. A slow recovery therefore causes an immediate @@ -142,10 +140,10 @@ local fence (latches `lost`, bumps the fence generation, moves the in-process ru `TransientNotLive`) and latches one self-remount generation. A confirmed foreign/successor or same-pair conflict remains a typed fail-closed error; it is never adopted. A real fence still costs only an epoch: recovery reclaims with a fresh one, bounded at three whole-chain attempts. This is the -general CAS posture: doubt about the source fails closed. Fenced mutations and the bounded renewals -(startup, remount and direct) retry transport ambiguity only inside authority already proved by the last -confirmed lease. The background renewal may keep retrying after that lease has expired; write authority -stays bounded by the start of the renewal that last succeeded plus the TTL. +general CAS posture: doubt about the source fails closed. Fenced mutations retry transport ambiguity +only inside authority already proved by the last confirmed lease. The renewal may keep retrying after +that lease has expired; write authority stays bounded by the start of the renewal that last succeeded +plus the TTL. GC's own view of a dead server is symmetric and clock-skew-immune: a slot becomes fence-eligible only after the leader observes the *same* renewal token hold stable, on its own monotonic clock, for `TTL + diff --git a/docs/en/antalya/cas/operations/debugging.md b/docs/en/antalya/cas/operations/debugging.md index 1050fa4e9599..b1184ddae67b 100644 --- a/docs/en/antalya/cas/operations/debugging.md +++ b/docs/en/antalya/cas/operations/debugging.md @@ -133,22 +133,18 @@ follows: request; `committed_after_retry` means a later identical physical `PUT` completed and the response itself proved it; `committed_after_expiry` means the renewal restored a lease that had expired (the row also has `expired_ms`). When more than one applies, the first of those three in that order wins. -- `outcome = 'failed'` carries the decisive `classification`: `external_lease_deadline` (the - confirmed lease's own safety margin, not the request policy, ran out first — check object-store - latency or `BOOTTIME` advancement before anything else), `request_deadline` (the ninety-second - request policy exhausted first), `unresolved` (every attempt was ambiguous and never settled by +- `outcome = 'failed'` carries the decisive `classification`: `unresolved` (every attempt was ambiguous and never settled by the time the operation gave up), `conflict` (an exact resolve read found another body — a same-pair twin, a GC-fenced body, a successor epoch, or a foreign holder), `cancelled` (a - renewal in flight was cancelled by shutdown or a remount park request; expected during graceful - shutdown), `fence_or_lifecycle_lost` (another local fence loss or a terminal lifecycle transition - closed admission while the operation was active), `deterministic_failure` (the store's own + renewal in flight was cancelled by shutdown; expected during graceful shutdown), + `fence_or_lifecycle_lost` (a pending remount request or a terminal lifecycle, including a published + FORGET intent, ended the renewal), `deterministic_failure` (the store's own answer proved the write never applied), and `vanished` (an exact resolve read proved the mount slot absent — the pool directory was removed or renamed out of band, or a decommission raced the renewal). `terminal_unclassified` means the renewal terminated through a path that assigned no classification; that is a defect to report together with the surrounding rows, not an operator condition. Do not collapse these into a generic timeout — the action - differs by classification, and only `external_lease_deadline` and `request_deadline` are about a - deadline at all. + differs by classification. - A following `mount_remount` row names the whole-chain `attempt_no` and final `step`. An `ok` row with `step = 'publish_live'` restored `Live` under the reported fresh `writer_epoch`; with `step = 'claimed_not_armed'` the claim succeeded with too little lease to admit a write, and the next diff --git a/docs/en/antalya/cas/operations/monitoring.md b/docs/en/antalya/cas/operations/monitoring.md index c8f1b4819d2f..6acf51a5b0cf 100644 --- a/docs/en/antalya/cas/operations/monitoring.md +++ b/docs/en/antalya/cas/operations/monitoring.md @@ -59,16 +59,15 @@ window and correlate them with the `server_root_id` in `system.cas_log`. | `CASMountRenewalRetries` | One per physical renewal `PUT` after the first in the same logical renewal | Positive growth shows in-period retry, not a later cadence beat | | `CASMountRenewalResolved` | One per logical renewal proved committed by an exact resolving `GET` | A response was ambiguous, but exact bytes and `write_attempt_id` proved the write | | `CASMountRenewalRecovered` | One per logical renewal committed after a retry or exact resolving `GET` | Recovered object-store blips that retained the existing mount incarnation | -| `CASMountRenewalDeadlineExceeded` | One per logical renewal stopped by the external lease-safety deadline | The last confirmed lease no longer left enough safe time; this is narrower than request-budget exhaustion. Only the bounded renewals (startup, remount and direct) can reach it; the background renewal does not | | `CASMountLeaseExpired` | One per renewal that restored a lease that had expired | Moves at the restore, not when the lease expires. While it is expired, `system.cas_mounts` shows `lifecycle_reason = 'lease_expired'` and writes are refused; the `watermark_renew` row of the restoring renewal carries `expired_ms` | | `CASRemountAttempts` | One per invocation of the existing whole-chain remount attempt | Includes both successful and failed attempts | | `CASRemountSucceeded` | One per whole-chain attempt that completed every step through the arm step, under a fresh writer epoch | Must be a subset of `CASRemountAttempts`. Includes a reclaim whose claim left too little lease to arm the fence (`step = 'claimed_not_armed'`); the next renewal arms it | | `CASRemountFailed` | One per whole-chain attempt that stopped before the arm step | Includes a named step exception or a step that returned transiently, also after the mount claim succeeded (for example at `renewer_start` or `quiesce_ref_tables`) | -`CASMountLeaseLost` complements those nine counters. It increments exactly once per operational +`CASMountLeaseLost` complements those counters. It increments exactly once per operational `Live -> TransientNotLive` recovery generation: either the initiating external loss or the first -ordinary terminal renewal consumer owns it. A parked terminal result and shutdown do not duplicate -the count. +ordinary terminal renewal consumer owns it. A renewal ended by a pending remount request does not +duplicate the count, and neither does shutdown. To inspect the current cumulative values, including counters that have never incremented: @@ -77,7 +76,7 @@ SELECT event, value FROM system.events WHERE event IN ( 'CASMountRenewalAttempts', 'CASMountRenewalRetries', 'CASMountRenewalResolved', - 'CASMountRenewalRecovered', 'CASMountRenewalDeadlineExceeded', 'CASMountLeaseLost', + 'CASMountRenewalRecovered', 'CASMountLeaseLost', 'CASMountLeaseExpired', 'CASRemountAttempts', 'CASRemountSucceeded', 'CASRemountFailed') SETTINGS system_events_show_zero_values = 1; ``` diff --git a/docs/en/antalya/cas/operations/troubleshooting.md b/docs/en/antalya/cas/operations/troubleshooting.md index 6d3631307f6f..670c9454f121 100644 --- a/docs/en/antalya/cas/operations/troubleshooting.md +++ b/docs/en/antalya/cas/operations/troubleshooting.md @@ -16,7 +16,7 @@ tools. | Symptom | Diagnosis | Action | |---|---|---| -| A server keeps losing its mount lease and self-remounting | Check `system.cas_mounts` for the server's own `state`/`expires_at`, then correlate `watermark_renew` and `mount_remount` in `system.cas_log`; losing the lease trips a local fence and latches a remount generation | Read the failed renewal's `classification` before changing anything — it alone now says why (see [the decision flow](#mount-renewal-remount-flow)). Only the bounded renewals (startup, remount and direct) can exceed a lease deadline; the background renewal does not. For those, look for object-store latency consuming the confirmed lease or BOOTTIME advancement; see [the mount lease](/antalya/cas/architecture/mounts-and-leases#mount-lease) | +| A server keeps losing its mount lease and self-remounting | Check `system.cas_mounts` for the server's own `state`/`expires_at`, then correlate `watermark_renew` and `mount_remount` in `system.cas_log`; losing the lease trips a local fence and latches a remount generation | Read the failed renewal's `classification` before changing anything — it says why (see [the decision flow](#mount-renewal-remount-flow)); see [the mount lease](/antalya/cas/architecture/mounts-and-leases#mount-lease) | | Writes fail with a transient error that names the lease and recover on their own | `system.cas_mounts` shows `lifecycle = 'not_live'`, `lifecycle_reason = 'lease_expired'` for the disk; `lifecycle_detail` is the text of the last failed renewal request. While refusals have started but the deadline has not passed, the row still shows `lifecycle = 'live'` (with the defaults, for the last 16 s of the lease); `lease_expired` appears once the deadline passes. `CASMountLeaseExpired` in `system.events` counts the expiries that a restore ended, not those that ended in a remount or a shutdown, and the `watermark_renew` row that ended one carries `expired_ms` in `detail` | The server could not renew its lease for longer than `cas_mount_lease_ttl_ms`, so it refuses writes until a renewal succeeds and leaves enough lease for a write's reservation. A renewal that succeeds after its own deadline (its start plus `cas_mount_lease_ttl_ms` already past) does not, and the next renewal follows at once. Failing renewals alone do not fence or remount it: it keeps the same `writer_epoch` unless a GC leader on another member fences the slot (token unchanged for `cas_mount_lease_ttl_ms + floor(cas_mount_lease_ttl_ms / 20) + cas_mount_renew_period_ms`) or the store answers definitively that the slot holds something else, and then it remounts under a new `writer_epoch`. Fix the object-store path named in `lifecycle_detail` (reachability, throttling, credentials). `CASMountRenewalAttempts` and `CASMountRenewalRetries` advance when a `PUT` is sent. When one `PUT` is unclear and only the reads that settle it fail, they stand still and the failure shows in `lifecycle_detail`. If `CASMountLeaseLost` rises as well, follow [the decision flow](#mount-renewal-remount-flow) | | Writes slow down or stall under load, with no exception reaching the client | S3 `SlowDown`/`ServiceUnavailable`/`RequestTimeout`/`InternalError` (5xx) responses are not on the request engine's `isDefinitelyRefusedWrite` definite-failure list (only malformed-request, entity-too-large, and access-denied that no credential refresh can fix are), so they classify as ambiguous and are retried automatically. Confirm with `sum(ProfileEvents['CASConditionalWriteUnresolved'])` rising alongside `sum(ProfileEvents['CASConditionalWriteAttempts'])` over `system.query_log` for the affected window (or `ProfileEvent_CASConditionalWriteUnresolved` in `system.metric_log` for a cumulative view across queries), and check `system.blob_storage_log` for `disk_name = ''` rows with a nonzero `error_code` around the same window | Nothing to configure per-request: the request engine retries the same `(key, bytes)` with capped-exponential backoff (200ms initial, capped at 5s, full jitter) until the 90-second operation deadline — there is no separate attempts ceiling, only the deadline — and the mount-lease renewer keeps extending the fence across the disruption — this is the "blips, throttling, partial outages" case the write path is built to survive. Confirm the mount lease itself is still renewing (`system.cas_mounts.expires_at` moving forward, `last_success_age_seconds` not climbing) — if it is, this is expected and self-resolving. If `SlowDown` responses are sustained rather than transient, check the bucket's request-rate limits against the pool's actual PUT/GET rate (see [bucket requirements](/antalya/cas/bucket-requirements)) and consider lowering `cas_blob_upload_pool_size` to reduce concurrent upload traffic; a write only surfaces a client-visible `NETWORK_ERROR` if the 90-second deadline is exhausted before the store recovers, and that error is retried by the ordinary merge/insert backoff, not silently dropped | | `GC` never seems to reclaim space after tables are dropped | `SELECT * FROM system.cas_gc_log WHERE event_type='Finish' ORDER BY event_time DESC LIMIT 5` — check `outcome`; also `SELECT is_leader FROM system.cas_mounts` on this node | If `outcome != 'Success'`/`'Deferred'`, see [reading GC health](/antalya/cas/operations/monitoring#gc-health); if this node is not the leader (`is_leader = 0`), it never reclaims for this disk — check the peer holding leadership. Reclamation also needs at least two full rounds past condemnation by design (the grace period is rounds, not acks) — a single manual `SYSTEM CAS GC RUN` will not finish it | @@ -48,28 +48,20 @@ Start with the `watermark_renew` timeline described in `WARNING` when the lease expires, written at the first request of the renewal after the expiry (a request that hangs delays it by up to one attempt timeout). The live state is the row in `system.cas_mounts` (`lifecycle_reason = 'lease_expired'`); the event moves only at the restore. -2. **External lease-safety exhaustion.** Only the renewals at startup, after a remount and the direct - renewal are bounded by the lease; the background renewal is not, so this case means one of those ran - out of lease. The failed row has `classification = 'external_lease_deadline'`; - `CASMountRenewalDeadlineExceeded` and `CASMountLeaseLost` rise. The runtime correctly refused to - manufacture authority beyond the last confirmed lease. Check object-store latency and - BOOTTIME/suspend history, then follow the ensuing remount. `classification = 'request_deadline'` is - the sibling case: the ninety-second request policy exhausted first rather than the lease's own - safety margin. -3. **Cancellation.** `classification = 'cancelled'` after a sent request is terminal and suppresses a +2. **Cancellation.** `classification = 'cancelled'` after a sent request is terminal and suppresses a clean farewell because the request may still land. Cancellation before any request remains `Active` and emits no failed aggregate row; during graceful shutdown that is the expected clean-release path. -4. **Confirmed conflict.** `classification = 'conflict'` means exact resolution found another body; +3. **Confirmed conflict.** `classification = 'conflict'` means exact resolution found another body; inspect `server_root_id`, `writer_epoch`, `seq`, and `write_attempt_id`. Same-pair twins, GC-fenced bodies, successor epochs, and foreign holders all remain fail closed. Do not delete or rewrite the mount key by hand. -5. **Fence or lifecycle loss.** `classification = 'fence_or_lifecycle_lost'` means another local loss, - remount park request, or terminal lifecycle closed admission while the operation was active. A - parked result reuses the already-requested recovery generation and must not double-count - `CASMountLeaseLost`. `classification = 'unresolved'` is a related but distinct case: every attempt +4. **Fence or lifecycle loss.** `classification = 'fence_or_lifecycle_lost'` means a pending remount + request or a terminal lifecycle, including a published FORGET intent, ended the renewal. A renewal + ended by a pending request raises no second recovery generation and does not count + `CASMountLeaseLost` again. `classification = 'unresolved'` is a related but distinct case: every attempt stayed ambiguous and the operation gave up without ever settling one way or the other. -6. **Whole-chain remount failure.** Read the following `mount_remount` row. Its `attempt_no`, `step`, +5. **Whole-chain remount failure.** Read the following `mount_remount` row. Its `attempt_no`, `step`, and optional `error` identify the failed owner/catalog/epoch/claim/install/quiescence/fence step. The current protocol retries the whole chain with bounded backoff; it does not preserve per-step progress. Repeated failure at the same step is the actionable signal. diff --git a/src/Common/ProfileEvents.cpp b/src/Common/ProfileEvents.cpp index 5cb260b8006c..a2a6b40d7c1a 100644 --- a/src/Common/ProfileEvents.cpp +++ b/src/Common/ProfileEvents.cpp @@ -956,7 +956,6 @@ The server successfully detected this situation and will download merged part fr M(CASMountRenewalRetries, "Number of physical conditional renewal PUTs sent after the first attempt of one logical CAS mount-lease renewal.", ValueType::Number) \ M(CASMountRenewalResolved, "Number of CAS mount-lease renewals whose committed outcome was proved by an exact resolving GET.", ValueType::Number) \ M(CASMountRenewalRecovered, "Number of logical CAS mount-lease renewals that committed after a physical retry or exact resolving GET.", ValueType::Number) \ - M(CASMountRenewalDeadlineExceeded, "Number of logical CAS mount-lease renewals stopped by the external lease-safety deadline. Growth means the last confirmed lease no longer had enough safe time for another physical attempt.", ValueType::Number) \ M(CASMountLeaseLost, "Counts exactly once per operational CAS mount-lease Live-to-TransientNotLive loss/recovery generation. The initiating external loss or the first ordinary terminal renewal consumer owns the increment, including external lease-safety deadline exhaustion; parked/classification/shutdown paths do not duplicate it.", ValueType::Number) \ M(CASMountLeaseExpired, "Number of times a renewal restored a CAS mount lease that had expired. While the lease is expired this server refuses writes and system.cas_mounts shows lifecycle_reason = 'lease_expired'; the watermark_renew event of the restoring renewal carries expired_ms.", ValueType::Number) \ M(CASRemountAttempts, "Number of invocations of the CAS whole-chain remount attempt.", ValueType::Number) \ diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index 890b51900d19..e611de55f6db 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -30,7 +30,6 @@ namespace ProfileEvents extern const Event CASMountRenewalRetries; extern const Event CASMountRenewalResolved; extern const Event CASMountRenewalRecovered; - extern const Event CASMountRenewalDeadlineExceeded; } namespace DB::Cas @@ -548,9 +547,6 @@ void CasMountRuntime::consumeRenewResult(const MountRenewResult & result) if (result.outcome == MountRenewOutcome::Committed && (result.attempts_sent > 1 || result.resolved_by_read)) ProfileEvents::incrementNoTrace(ProfileEvents::CASMountRenewalRecovered); - if (result.outcome == MountRenewOutcome::Terminal - && result.deadline_source == GaveUp::Source::Lease) - ProfileEvents::incrementNoTrace(ProfileEvents::CASMountRenewalDeadlineExceeded); /// One step under `driver_mutex`. The loop reads the remount generations only after it, so no /// reclaim can arm between a commit and the publication of its deadline. diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp index 47c7419b1718..466ed2458fbc 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp @@ -98,8 +98,8 @@ uint64_t defaultBootMs() } /// Why a renewal ended without a retained lease, in the vocabulary the audit event reports. Each -/// value is assigned from exactly one arm of the write's verdict, so the event never re-derives a -/// reason from state the request engine does not carry. +/// value is assigned from the write's verdict, so the event never re-derives a reason from state the +/// request engine does not carry. enum class MountRenewTerminalClassification : uint8_t { Unclassified, @@ -108,8 +108,6 @@ enum class MountRenewTerminalClassification : uint8_t Vanished, Cancelled, FenceOrLifecycleLost, - ExternalLeaseDeadline, - RequestDeadline, Unresolved, }; @@ -333,8 +331,6 @@ constexpr std::string_view terminalClassificationName(MountRenewTerminalClassifi case MountRenewTerminalClassification::Vanished: return "vanished"; case MountRenewTerminalClassification::Cancelled: return "cancelled"; case MountRenewTerminalClassification::FenceOrLifecycleLost: return "fence_or_lifecycle_lost"; - case MountRenewTerminalClassification::ExternalLeaseDeadline: return "external_lease_deadline"; - case MountRenewTerminalClassification::RequestDeadline: return "request_deadline"; case MountRenewTerminalClassification::Unresolved: return "unresolved"; case MountRenewTerminalClassification::Unclassified: return "terminal_unclassified"; } @@ -1686,8 +1682,6 @@ MountRenewResult MountLeaseRenewer::renew(const MountRenewOperationEnvironment & { result.sent_any = gave_up->sent_any; result.attempts_sent = gave_up->attempts_sent; - if (gave_up->why == GaveUp::Why::Deadline) - result.deadline_source = gave_up->deadline_source; /// Nothing was sent and the node was already stopping: the lease is exactly as it was, so this /// is a renewal that never ran, not one that lost its authority. @@ -1706,11 +1700,8 @@ MountRenewResult MountLeaseRenewer::renew(const MountRenewOperationEnvironment & ? MountRenewTerminalClassification::Cancelled : MountRenewTerminalClassification::FenceOrLifecycleLost; break; + /// Reachable only at the end of the clock's range: the policy has no lease bound and no window. case GaveUp::Why::Deadline: - classification = gave_up->deadline_source == GaveUp::Source::Lease - ? MountRenewTerminalClassification::ExternalLeaseDeadline - : MountRenewTerminalClassification::RequestDeadline; - break; case GaveUp::Why::Unresolved: classification = MountRenewTerminalClassification::Unresolved; break; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h index 6e807ca17a80..c06630c0ac08 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h @@ -55,9 +55,6 @@ struct MountRenewResult uint32_t attempts_sent = 0; bool resolved_by_read = false; bool sent_any = false; - /// Which bound ended a renewal that ran out of time; unset for every other ending, including a - /// committed one -- no deadline ended it, so naming one would invent a fact. - std::optional deadline_source; std::exception_ptr failure; }; diff --git a/src/Disks/tests/gtest_cas_observability.cpp b/src/Disks/tests/gtest_cas_observability.cpp index 4e12c1988e95..7fb88c6ba07c 100644 --- a/src/Disks/tests/gtest_cas_observability.cpp +++ b/src/Disks/tests/gtest_cas_observability.cpp @@ -24,7 +24,6 @@ extern const Event CASMountRenewalAttempts; extern const Event CASMountRenewalRetries; extern const Event CASMountRenewalResolved; extern const Event CASMountRenewalRecovered; -extern const Event CASMountRenewalDeadlineExceeded; } using namespace DB::Cas; @@ -91,7 +90,6 @@ struct RenewalCounterSnapshot uint64_t retries; uint64_t resolved; uint64_t recovered; - uint64_t deadline_exceeded; }; RenewalCounterSnapshot renewalCounters() @@ -102,7 +100,6 @@ RenewalCounterSnapshot renewalCounters() .retries = global_counters[ProfileEvents::CASMountRenewalRetries].load(), .resolved = global_counters[ProfileEvents::CASMountRenewalResolved].load(), .recovered = global_counters[ProfileEvents::CASMountRenewalRecovered].load(), - .deadline_exceeded = global_counters[ProfileEvents::CASMountRenewalDeadlineExceeded].load(), }; } @@ -112,14 +109,12 @@ void expectRenewalCounterDelta( uint64_t attempts, uint64_t retries, uint64_t resolved, - uint64_t recovered, - uint64_t deadline_exceeded) + uint64_t recovered) { EXPECT_EQ(after.attempts - before.attempts, attempts); EXPECT_EQ(after.retries - before.retries, retries); EXPECT_EQ(after.resolved - before.resolved, resolved); EXPECT_EQ(after.recovered - before.recovered, recovered); - EXPECT_EQ(after.deadline_exceeded - before.deadline_exceeded, deadline_exceeded); } /// Publish ONE ref naming a single-blob part through the real writer sequence (mirrors @@ -169,7 +164,7 @@ TEST(CASObservability, RenewalCountersHaveExactPhysicalAndLogicalDeltas) const RenewalCounterSnapshot before = renewalCounters(); EXPECT_NO_THROW(store->renewWatermarkOnce()); const RenewalCounterSnapshot after = renewalCounters(); - expectRenewalCounterDelta(before, after, attempts, retries, resolved, recovered, 0); + expectRenewalCounterDelta(before, after, attempts, retries, resolved, recovered); }; run(RenewalCounterBackend::Fault::None, /*attempts=*/1, /*retries=*/0, /*resolved=*/0, /*recovered=*/0); diff --git a/tests/integration/test_cas_mount_renewal_retry/test.py b/tests/integration/test_cas_mount_renewal_retry/test.py index 77588f3de45e..72b5b0063665 100644 --- a/tests/integration/test_cas_mount_renewal_retry/test.py +++ b/tests/integration/test_cas_mount_renewal_retry/test.py @@ -25,7 +25,6 @@ "CASMountRenewalRetries", "CASMountRenewalResolved", "CASMountRenewalRecovered", - "CASMountRenewalDeadlineExceeded", "CASMountLeaseLost", "CASMountLeaseExpired", "CASRemountAttempts", @@ -457,7 +456,6 @@ def recovered_snapshot(deadline): assert delta["CASMountRenewalAttempts"] > 1, delta assert delta["CASMountRenewalRetries"] > 0, delta assert delta["CASMountRenewalRecovered"] > 0, delta - assert delta["CASMountRenewalDeadlineExceeded"] == 0, delta assert delta["CASRemountAttempts"] == 0, delta assert delta["CASRemountSucceeded"] == 0, delta assert delta["CASRemountFailed"] == 0, delta @@ -584,7 +582,6 @@ def resolved_snapshot(deadline): assert delta["CASMountRenewalRetries"] == 0, delta assert delta["CASMountRenewalResolved"] == 1, delta assert delta["CASMountRenewalRecovered"] == 1, delta - assert delta["CASMountRenewalDeadlineExceeded"] == 0, delta assert delta["CASRemountAttempts"] == 0, delta assert delta["CASRemountSucceeded"] == 0, delta assert delta["CASRemountFailed"] == 0, delta From cbada5138f26a1e023d2e4c6e387aefb50e31769 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 09:05:50 +0200 Subject: [PATCH 18/37] Return the identity, ending and duration of a CAS mount renewal MountLeaseRenewer::renew fills writer_epoch, seq, write_attempt_id, classification and elapsed_ms in MountRenewResult, so the report of a renewal can be built from its result. throwRenewConflict hands its classification back through an out-parameter instead of a thread-local. A new event test pins the row of a terminal renewal that sent no request. Co-Authored-By: Claude Opus 5.5 --- .../ContentAddressed/Pool/CasServerRoot.cpp | 60 ++++------ .../ContentAddressed/Pool/CasServerRoot.h | 25 +++- src/Disks/tests/gtest_cas_event_log.cpp | 25 ++++ src/Disks/tests/gtest_cas_heartbeat.cpp | 110 ++++++++++++++++++ 4 files changed, 184 insertions(+), 36 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp index 466ed2458fbc..0b6cba167ad6 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp @@ -97,20 +97,6 @@ uint64_t defaultBootMs() return static_cast(ts.tv_sec) * 1000 + static_cast(ts.tv_nsec) / 1000000; } -/// Why a renewal ended without a retained lease, in the vocabulary the audit event reports. Each -/// value is assigned from the write's verdict, so the event never re-derives a reason from state the -/// request engine does not carry. -enum class MountRenewTerminalClassification : uint8_t -{ - Unclassified, - DeterministicFailure, - Conflict, - Vanished, - Cancelled, - FenceOrLifecycleLost, - Unresolved, -}; - /// One logical renewal's audit snapshot. Fixed-size and trivially copyable so a reentrant event sink /// gets a distinct stack slot instead of aliasing the call that is still running. struct MountRenewObservabilityContext @@ -167,12 +153,6 @@ MountRenewObservabilityContext * currentMountRenewObservability() noexcept return &mount_renew_observability.contexts[mount_renew_observability.depth - 1]; } -void markMountRenewTermination(MountRenewTerminalClassification classification) noexcept -{ - if (MountRenewObservabilityContext * context = currentMountRenewObservability()) - context->terminal_classification = classification; -} - enum class MountRenewObservabilityRegistration : uint8_t { Stack, @@ -463,6 +443,7 @@ void reportMountRenewCompletion(const MountRenewResult & result, std::optionalactive) return; context->completed = true; + context->terminal_classification = result.classification; context->outcome = result.outcome; context->attempts_sent = std::max(context->attempts_sent, result.attempts_sent); context->resolved_by_read = result.resolved_by_read; @@ -1478,11 +1459,12 @@ uint64_t MountLeaseRenewer::start(Liveness liveness) return attempt_start_boot_ms; } -[[noreturn]] void MountLeaseRenewer::throwRenewConflict(const Observation & seen) const +[[noreturn]] void MountLeaseRenewer::throwRenewConflict( + const Observation & seen, MountRenewTerminalClassification & classification) const { if (const Object * occupant = std::get_if(&seen)) { - markMountRenewTermination(MountRenewTerminalClassification::Conflict); + classification = MountRenewTerminalClassification::Conflict; const MountLease current = decodeMountLease(occupant->bytes); if (current.server_uuid == server_uuid && current.gc_fenced) { @@ -1530,7 +1512,7 @@ uint64_t MountLeaseRenewer::start(Liveness liveness) if (std::holds_alternative(seen)) { - markMountRenewTermination(MountRenewTerminalClassification::Vanished); + classification = MountRenewTerminalClassification::Vanished; emitMountEvent( event_sink, CasEventType::MountConflict, srid, "vanished", nullptr, "mount slot vanished while renewing -- failing closed"); @@ -1541,7 +1523,7 @@ uint64_t MountLeaseRenewer::start(Liveness liveness) /// The precondition was refused but nothing identifiable was read back: neither the successor nor /// an absence is established, so the only honest verdict is that this renewal settled nothing. - markMountRenewTermination(MountRenewTerminalClassification::Unresolved); + classification = MountRenewTerminalClassification::Unresolved; throwCasWriteRetryLater(fmt::format( "CAS mount-lease: key '{}' refused our precondition and the resolving read established neither " "an occupant nor an absence", key)); @@ -1611,6 +1593,14 @@ MountRenewResult MountLeaseRenewer::renew(const MountRenewOperationEnvironment & MountRenewResult result; result.attempt_start_boot_ms = attempt_start_boot_ms; + result.writer_epoch = writer_epoch; + result.seq = next_seq; + result.write_attempt_id = write_attempt_id; + const auto finished = [&boot_clock, attempt_start_boot_ms](MountRenewResult done) + { + done.elapsed_ms = elapsedSince(attempt_start_boot_ms, boot_clock()); + return done; + }; CasOperation op = lease_requests.admit(environment.live); if (environment.on_request) @@ -1631,9 +1621,9 @@ MountRenewResult MountLeaseRenewer::renew(const MountRenewOperationEnvironment & catch (...) { /// The engine surfaces a deterministic local failure unchanged rather than reissuing it. - markMountRenewTermination(MountRenewTerminalClassification::DeterministicFailure); + result.classification = MountRenewTerminalClassification::DeterministicFailure; result.failure = std::current_exception(); - return terminalResult(std::move(result)); + return finished(terminalResult(std::move(result))); } if (Committed * committed = std::get_if(&*written)) @@ -1649,7 +1639,7 @@ MountRenewResult MountLeaseRenewer::renew(const MountRenewOperationEnvironment & result.attempts_sent = committed->attempts_sent; result.resolved_by_read = committed->resolved_by_read; result.sent_any = committed->attempts_sent != 0; - return result; + return finished(std::move(result)); } if (const Conflict * conflict = std::get_if(&*written)) @@ -1658,24 +1648,24 @@ MountRenewResult MountLeaseRenewer::renew(const MountRenewOperationEnvironment & result.attempts_sent = conflict->attempts_sent; try { - throwRenewConflict(conflict->seen); + throwRenewConflict(conflict->seen, result.classification); } catch (...) { result.failure = std::current_exception(); } - return terminalResult(std::move(result)); + return finished(terminalResult(std::move(result))); } if (const Refused * refused = std::get_if(&*written)) { result.sent_any = true; result.attempts_sent = refused->attempts_sent; - markMountRenewTermination(MountRenewTerminalClassification::DeterministicFailure); + result.classification = MountRenewTerminalClassification::DeterministicFailure; result.failure = std::make_exception_ptr(Exception( refused->store_error, "CAS mount-lease: the store refused the renewal of key '{}': {}", key, refused->message)); - return terminalResult(std::move(result)); + return finished(terminalResult(std::move(result))); } if (const GaveUp * gave_up = std::get_if(&*written)) @@ -1687,9 +1677,9 @@ MountRenewResult MountLeaseRenewer::renew(const MountRenewOperationEnvironment & /// is a renewal that never ran, not one that lost its authority. if (gave_up->why == GaveUp::Why::FenceLost && !gave_up->sent_any && cancelled) { - markMountRenewTermination(MountRenewTerminalClassification::Cancelled); + result.classification = MountRenewTerminalClassification::Cancelled; result.outcome = MountRenewOutcome::NotAttempted; - return result; + return finished(std::move(result)); } MountRenewTerminalClassification classification = MountRenewTerminalClassification::Unresolved; @@ -1706,7 +1696,7 @@ MountRenewResult MountLeaseRenewer::renew(const MountRenewOperationEnvironment & classification = MountRenewTerminalClassification::Unresolved; break; } - markMountRenewTermination(classification); + result.classification = classification; result.failure = makeCasWriteRetryLaterExceptionPtr(fmt::format( "CAS mount-lease renewal for key '{}' did not retain the lease ({}, {} attempt sent, last " "observation: {})", @@ -1714,7 +1704,7 @@ MountRenewResult MountLeaseRenewer::renew(const MountRenewOperationEnvironment & terminalClassificationName(classification), gave_up->sent_any ? "at least one" : "no", detail::renderObservation(gave_up->last_seen))); - return terminalResult(std::move(result)); + return finished(terminalResult(std::move(result))); } /// The remaining alternative is `Declined`, which only a decide returning nothing produces; a diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h index c06630c0ac08..0977844be7c5 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h @@ -46,6 +46,19 @@ enum class MountRenewOutcome : uint8_t Terminal, }; +/// Why a renewal ended without retaining the lease, in the words of its audit event. `renew` sets it in +/// the arm of the write's verdict that ended the renewal. +enum class MountRenewTerminalClassification : uint8_t +{ + Unclassified, + DeterministicFailure, + Conflict, + Vanished, + Cancelled, + FenceOrLifecycleLost, + Unresolved, +}; + struct MountRenewResult { MountRenewOutcome outcome = MountRenewOutcome::Terminal; @@ -56,6 +69,14 @@ struct MountRenewResult bool resolved_by_read = false; bool sent_any = false; std::exception_ptr failure; + /// The body this renewal wrote or tried to write. + uint64_t writer_epoch = 0; + uint64_t seq = 0; + UInt128 write_attempt_id{}; + /// `Unclassified` for a committed renewal. + MountRenewTerminalClassification classification = MountRenewTerminalClassification::Unclassified; + /// From `attempt_start_boot_ms` to the return of `renew`, on the renewal's boot clock. + uint64_t elapsed_ms = 0; }; /// The spacing of a renewal's retries. @@ -578,7 +599,9 @@ class MountLeaseRenewer /// admit such a write: `start` establishes it and each committed renewal replaces it. const Etag & precondition() const; Etag claim(CasOperation & op, const String & body); - [[noreturn]] void throwRenewConflict(const Observation & seen) const; + /// Sets `classification` for the arm it takes, then throws what ended the renewal. The caller + /// holds the classification before the event sink runs, so the sink cannot change it. + [[noreturn]] void throwRenewConflict(const Observation & seen, MountRenewTerminalClassification & classification) const; MountRenewResult terminalResult(MountRenewResult result); void terminate(CasOperation & op); diff --git a/src/Disks/tests/gtest_cas_event_log.cpp b/src/Disks/tests/gtest_cas_event_log.cpp index f1667869caaf..712552a2f4fc 100644 --- a/src/Disks/tests/gtest_cas_event_log.cpp +++ b/src/Disks/tests/gtest_cas_event_log.cpp @@ -410,6 +410,31 @@ TEST(CASEvent, TerminalRenewalDetailsPreservePhysicalTruthAndClassification) } } +/// A pending remount request ends the renewal before its first request, and its row says so. +TEST(CASEvent, ATerminalRenewalThatSentNothingReportsZeroAttempts) +{ + auto backend = std::make_shared(); + auto boot_ms = std::make_shared>(100); + /// Heap-owned, not a plain local: the Pool can outlive this stack frame (a background publish holds + /// `shared_from_this()`), so a by-reference capture of a local would dangle. + auto events = std::make_shared(); + auto store = openRenewalEventPool(backend, boot_ms, renewalEventBudget(), "renewal-sent-nothing"); + store->setEventSink([events](CasEvent event) + { + events->push(std::move(event)); + }); + + (void)store->scheduleRemountForTest(); + EXPECT_THROW(store->renewWatermarkOnce(), DB::Exception); + + const std::vector renewals = watermarkRenewEvents(events->snapshot()); + ASSERT_EQ(renewals.size(), 1u); + EXPECT_EQ(renewals[0].outcome, "failed"); + EXPECT_EQ(renewals[0].detail.at("attempts_sent"), "0"); + EXPECT_EQ(renewals[0].detail.at("classification"), "fence_or_lifecycle_lost"); + EXPECT_EQ(renewals[0].detail.at("seq"), "2"); +} + TEST(CASEvent, ReentrantRenewalSinkPreservesOuterObservationIdentity) { auto backend = std::make_shared(); diff --git a/src/Disks/tests/gtest_cas_heartbeat.cpp b/src/Disks/tests/gtest_cas_heartbeat.cpp index 9f3f7e60f2a3..18d8ae5a83bf 100644 --- a/src/Disks/tests/gtest_cas_heartbeat.cpp +++ b/src/Disks/tests/gtest_cas_heartbeat.cpp @@ -1185,6 +1185,34 @@ TEST(CASHeartbeat, ExpectedPredecessorThenLateLandingIsAdoptedExactly) decodeMountLease(backend->attempts[0].bytes).write_attempt_id); } +namespace +{ +/// One renewer over a scripted store, started on its own seeded claim, for the cases of +/// `RenewReturnsWhatItsReportNeeds`. The clocks come first: hooks stored in `backend` capture them. +struct ReportFieldsCase +{ + explicit ReportFieldsCase(String srid_) + : srid(std::move(srid_)) + { + seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, wall_ms, /*ttl_ms=*/1000); + renewer.start(); + backend->attempts.clear(); + } + + uint64_t wall_ms = 1000; + uint64_t boot_ms = 100; + String srid; + UInt128 uuid{1}; + Layout layout{"pool"}; + std::shared_ptr backend = std::make_shared(); + Ops ops{backend, &boot_ms}; + MountLeaseRenewer renewer{ + ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, std::chrono::milliseconds(1000), + [this] { return wall_ms; }, [] { return uint64_t{0}; }, CasEventSink{}, std::chrono::milliseconds(20), + [this] { return boot_ms; }}; +}; +} + TEST(CASHeartbeat, GcFenceAndVanishedMountStayTerminal) { const auto run_case = [](bool vanish) @@ -1220,6 +1248,88 @@ TEST(CASHeartbeat, GcFenceAndVanishedMountStayTerminal) run_case(true); } +/// A renewal returns what its report names: the body it wrote or tried to write, why it ended, and +/// how long it took on its own boot clock. +TEST(CASHeartbeat, RenewReturnsWhatItsReportNeeds) +{ + { + ReportFieldsCase c("commit"); + c.backend->on_attempt = [&boot_ms = c.boot_ms] { boot_ms += 250; }; + const MountRenewResult result = c.renewer.renew(renewalEnvironment(c.boot_ms)); + ASSERT_EQ(result.outcome, MountRenewOutcome::Committed); + ASSERT_EQ(c.backend->attempts.size(), 1u); + const MountLease sent = decodeMountLease(c.backend->attempts.back().bytes); + EXPECT_EQ(result.writer_epoch, 9u); + EXPECT_EQ(result.seq, 2u); + EXPECT_EQ(sent.seq, 2u); + EXPECT_NE(result.write_attempt_id, UInt128{}); + EXPECT_EQ(result.write_attempt_id, sent.write_attempt_id); + EXPECT_EQ(result.classification, MountRenewTerminalClassification::Unclassified); + EXPECT_EQ(result.attempt_start_boot_ms, 100u); + EXPECT_EQ(result.elapsed_ms, 250u); + } + + { + ReportFieldsCase c("conflict"); + const String key = c.layout.mountKey(c.srid); + const auto got = c.ops.op.read(key, Retry::standard()); + ASSERT_TRUE(got.has_value()); + MountLease foreign = decodeMountLease(got->bytes); + foreign.server_uuid = UInt128{2}; + foreign.seq = 40; + foreign.write_attempt_id = UInt128{0xF0F0}; + mustCommit(c.ops.op.replace(key, encodeMountLease(foreign), got->etag, Retry::standard()), "foreign slot"); + c.backend->attempts.clear(); + c.backend->on_attempt = [&boot_ms = c.boot_ms] { boot_ms += 250; }; + const MountRenewResult result = c.renewer.renew(renewalEnvironment(c.boot_ms)); + ASSERT_EQ(result.outcome, MountRenewOutcome::Terminal); + ASSERT_EQ(c.backend->attempts.size(), 1u); + const MountLease sent = decodeMountLease(c.backend->attempts.back().bytes); + EXPECT_EQ(result.classification, MountRenewTerminalClassification::Conflict); + EXPECT_EQ(result.writer_epoch, 9u); + EXPECT_EQ(result.seq, 2u) << "the seq this renewal tried to write, not the occupant's"; + EXPECT_EQ(result.write_attempt_id, sent.write_attempt_id); + EXPECT_EQ(result.elapsed_ms, 250u); + } + + { + ReportFieldsCase c("vanished"); + const String key = c.layout.mountKey(c.srid); + const auto got = c.ops.op.read(key, Retry::standard()); + ASSERT_TRUE(got.has_value()); + ASSERT_EQ(c.ops.op.remove(key, got->etag, Retry::standard()), Removal::Removed); + const MountRenewResult result = c.renewer.renew(renewalEnvironment(c.boot_ms)); + ASSERT_EQ(result.outcome, MountRenewOutcome::Terminal); + EXPECT_EQ(result.classification, MountRenewTerminalClassification::Vanished); + EXPECT_EQ(result.seq, 2u); + EXPECT_NE(result.write_attempt_id, UInt128{}); + } + + { + /// Ended before its first request by a liveness that refuses with no stop requested. + ReportFieldsCase c("refused"); + const MountRenewResult result = c.renewer.renew(renewalEnvironment( + c.boot_ms, /*live=*/[] { return false; }, /*cancelled=*/[] { return false; })); + ASSERT_EQ(result.outcome, MountRenewOutcome::Terminal); + EXPECT_TRUE(c.backend->attempts.empty()); + EXPECT_EQ(result.classification, MountRenewTerminalClassification::FenceOrLifecycleLost); + EXPECT_EQ(result.writer_epoch, 9u); + EXPECT_EQ(result.seq, 2u); + EXPECT_NE(result.write_attempt_id, UInt128{}) << "the id is minted before the renewal is admitted"; + EXPECT_EQ(result.elapsed_ms, 0u); + } + + { + /// Ended before its first request by a stop. + ReportFieldsCase c("stopped"); + const MountRenewResult result = c.renewer.renew(renewalEnvironment( + c.boot_ms, /*live=*/[] { return false; }, /*cancelled=*/[] { return true; })); + ASSERT_EQ(result.outcome, MountRenewOutcome::NotAttempted); + EXPECT_EQ(result.classification, MountRenewTerminalClassification::Cancelled); + EXPECT_EQ(result.seq, 2u); + } +} + TEST(CASHeartbeat, LateDeliveryAfterTerminalCannotRearmOrOverwriteSuccessor) { Layout layout("pool"); From 0145e1c727a606a9b96b791193c13a09e9a53b29 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 09:15:08 +0200 Subject: [PATCH 19/37] Report CAS mount renewals from their result, without a thread-local stack renewOnce reads the fence deadline before the renewal and, after the consume step and with no lock held, passes it with the result to reportMountRenewCompletion; consumeRenewResult returns the restored lease for that report. The event fields and log texts are unchanged; the remount_attempt_no detail goes, since no renewal runs inside a remount. The per-thread observation stack, its deferred delivery from tryRemountOnce and its nesting are removed, together with the test of the eight-slot stack. Co-Authored-By: Claude Opus 5.5 --- .../ContentAddressed/Pool/CasMountRuntime.cpp | 44 +-- .../ContentAddressed/Pool/CasMountRuntime.h | 3 +- .../ContentAddressed/Pool/CasPool.cpp | 4 - .../ContentAddressed/Pool/CasServerRoot.cpp | 369 ++++-------------- .../ContentAddressed/Pool/CasServerRoot.h | 10 + src/Disks/tests/gtest_cas_event_log.cpp | 240 +++++++----- 6 files changed, 233 insertions(+), 437 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index e611de55f6db..1f65d92b3689 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -35,10 +35,6 @@ namespace ProfileEvents namespace DB::Cas { -void reportMountRenewCompletion(const MountRenewResult & result, std::optional expired_ms) noexcept; -void configureMountRenewObservability( - const String * server_root_id, const CasEventSink * event_sink, bool deferred) noexcept; - namespace { /// Wall-clock seconds since epoch — the `since` timestamp the lifecycle snapshot reports (spec §7). A @@ -540,7 +536,7 @@ void CasMountRuntime::sleepInterruptibly(uint64_t ms) driver_cv.wait_for(lock, std::chrono::milliseconds(ms), [this] { return workers_stop_requested; }); } -void CasMountRuntime::consumeRenewResult(const MountRenewResult & result) +std::optional CasMountRuntime::consumeRenewResult(const MountRenewResult & result) { if (result.resolved_by_read) ProfileEvents::incrementNoTrace(ProfileEvents::CASMountRenewalResolved); @@ -581,25 +577,8 @@ void CasMountRuntime::consumeRenewResult(const MountRenewResult & result) driver_cv.notify_all(); } - if (restored) - LOG_WARNING(getLogger("CasPool"), - "CAS mount lease of '{}' was expired for {} ms; a renewal restored it and writes resume. " - "Last failed renewal request: {}", - server_root_id, restored->expired_ms, restored->last_failure); - - if (result.outcome == MountRenewOutcome::Committed) - { - std::optional expired_ms; - if (restored) - expired_ms = restored->expired_ms; - reportMountRenewCompletion(result, expired_ms); - return; - } - if (result.outcome == MountRenewOutcome::NotAttempted) - { - reportMountRenewCompletion(result, std::nullopt); - return; - } + if (result.outcome != MountRenewOutcome::Terminal) + return restored; if (!result.failure) throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: terminal renewal has no failure"); @@ -615,8 +594,7 @@ void CasMountRuntime::consumeRenewResult(const MountRenewResult & result) catch (...) { } - - reportMountRenewCompletion(result, std::nullopt); + return restored; } MountRenewResult CasMountRuntime::renewOnce() @@ -631,9 +609,19 @@ MountRenewResult CasMountRuntime::renewOnce() throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: renewal requires an Active renewer"); renewer = mount_renewer.get(); } - configureMountRenewObservability(&server_root_id, &event_sink, /*deferred=*/false); + const uint64_t lease_deadline_before_boot_ms = mount_fence.deadline_boot_ms.load(std::memory_order_acquire); const MountRenewResult result = renewer->renew(renewalEnvironment()); - consumeRenewResult(result); + const std::optional restored = consumeRenewResult(result); + + if (restored) + LOG_WARNING(getLogger("CasPool"), + "CAS mount lease of '{}' was expired for {} ms; a renewal restored it and writes resume. " + "Last failed renewal request: {}", + server_root_id, restored->expired_ms, restored->last_failure); + std::optional expired_ms; + if (restored) + expired_ms = restored->expired_ms; + reportMountRenewCompletion(result, server_root_id, event_sink, lease_deadline_before_boot_ms, expired_ms); return result; } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index 6875f6e3c69c..2853ca062927 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -443,7 +443,8 @@ class CasMountRuntime /// the restore and returns it for the caller to log after the unlock. Empty when the lease was not /// expired or is still expired. std::optional publishRenewedDeadline(uint64_t deadline_boot_ms); - void consumeRenewResult(const MountRenewResult & result); + /// The consume step: takes `driver_mutex` itself and returns the restore for the caller to log. + std::optional consumeRenewResult(const MountRenewResult & result); void renewalLoop(); ThreadFromGlobalPool makeWorker(std::function body); /// The renewal's liveness: no stop, no pending remount request, no terminal lifecycle (a published diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp index 808f9d6bb6e9..3a5fee93ac65 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp @@ -70,8 +70,6 @@ namespace ProfileEvents namespace DB::Cas { -void deliverDeferredMountRenewObservability(uint64_t remount_attempt_no) noexcept; - namespace { @@ -1239,8 +1237,6 @@ bool Pool::tryRemountOnce() String error; SCOPE_EXIT( { - deliverDeferredMountRenewObservability(attempt_no); - if (succeeded) ProfileEvents::increment(ProfileEvents::CASRemountSucceeded); else diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp index 0b6cba167ad6..3b3bb2d1d5fd 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp @@ -46,11 +46,6 @@ namespace ErrorCodes namespace DB::Cas { -void reportMountRenewCompletion(const MountRenewResult & result, std::optional expired_ms) noexcept; -void configureMountRenewObservability( - const String * server_root_id, const CasEventSink * event_sink, bool deferred) noexcept; -void deliverDeferredMountRenewObservability(uint64_t remount_attempt_no) noexcept; - /// The owner, epoch, and mount-lease wire codecs are implemented in /// `Formats/CasServerRootFormats`; this file contains the mount-safety protocol logic that uses /// those codecs. @@ -97,177 +92,30 @@ uint64_t defaultBootMs() return static_cast(ts.tv_sec) * 1000 + static_cast(ts.tv_nsec) / 1000000; } -/// One logical renewal's audit snapshot. Fixed-size and trivially copyable so a reentrant event sink -/// gets a distinct stack slot instead of aliasing the call that is still running. -struct MountRenewObservabilityContext -{ - bool active = false; - bool completed = false; - bool deferred = false; - const String * server_root_id = nullptr; - const CasEventSink * event_sink = nullptr; - uint64_t writer_epoch = 0; - uint64_t seq = 0; - UInt128 write_attempt_id{}; - uint64_t observability_start_boot_ms = 0; - uint64_t confirmed_deadline_boot_ms = 0; - uint64_t initial_confirmed_budget_ms = 0; - MountRenewOutcome outcome = MountRenewOutcome::NotAttempted; - MountRenewTerminalClassification terminal_classification = MountRenewTerminalClassification::Unclassified; - uint32_t attempts_sent = 0; - bool resolved_by_read = false; - /// How long the lease had been expired when this renewal restored it; empty unless it did. - std::optional expired_ms = std::nullopt; -}; - -static_assert(std::is_trivially_copyable_v); - -struct MountRenewObservabilityConfiguration -{ - bool configured = false; - bool deferred = false; - const String * server_root_id = nullptr; - const CasEventSink * event_sink = nullptr; -}; - -/// Event sinks may synchronously renew another Pool on the same thread. A fixed stack keeps every -/// registered outer per-call snapshot stable without allocation, including while a remount re-anchor -/// holds `remount_mutex`. Overflow suppresses rich event/log delivery for the nested call rather than -/// aliasing an outer call or changing protocol behavior; physical attempt truth is independently -/// retained by the stack-local observer in `MountLeaseRenewer::renew`. -struct MountRenewObservabilityStack -{ - static constexpr size_t capacity = 8; - std::array contexts; - size_t depth = 0; - size_t suppressed_depth = 0; - MountRenewObservabilityConfiguration pending; -}; - -thread_local MountRenewObservabilityStack mount_renew_observability; - -MountRenewObservabilityContext * currentMountRenewObservability() noexcept -{ - if (mount_renew_observability.suppressed_depth != 0 || mount_renew_observability.depth == 0) - return nullptr; - return &mount_renew_observability.contexts[mount_renew_observability.depth - 1]; -} - -enum class MountRenewObservabilityRegistration : uint8_t -{ - Stack, - Suppressed, - Ignored, -}; - -MountRenewObservabilityRegistration beginMountRenewObservabilityCall() noexcept -{ - const MountRenewObservabilityConfiguration configured = std::exchange( - mount_renew_observability.pending, MountRenewObservabilityConfiguration{}); - if (!configured.configured) - return MountRenewObservabilityRegistration::Ignored; - if (mount_renew_observability.depth == MountRenewObservabilityStack::capacity) - { - ++mount_renew_observability.suppressed_depth; - return MountRenewObservabilityRegistration::Suppressed; - } - - mount_renew_observability.contexts[mount_renew_observability.depth++] = MountRenewObservabilityContext{ - .deferred = configured.deferred, - .server_root_id = configured.server_root_id, - .event_sink = configured.event_sink, - }; - return MountRenewObservabilityRegistration::Stack; -} - -void abandonMountRenewObservabilityCall() noexcept -{ - if (mount_renew_observability.suppressed_depth != 0) - { - --mount_renew_observability.suppressed_depth; - return; - } - if (mount_renew_observability.depth != 0) - --mount_renew_observability.depth; -} - -class MountRenewObservabilityCallGuard -{ -public: - explicit MountRenewObservabilityCallGuard(MountRenewObservabilityRegistration registration_) - : registration(registration_) - , uncaught_on_entry(std::uncaught_exceptions()) - { - } - - ~MountRenewObservabilityCallGuard() - { - if (registration == MountRenewObservabilityRegistration::Ignored) - return; - if (std::uncaught_exceptions() > uncaught_on_entry) - abandonMountRenewObservabilityCall(); - } - -private: - MountRenewObservabilityRegistration registration; - int uncaught_on_entry; -}; - -void initializeMountRenewObservability( - const String & server_root_id, - uint64_t writer_epoch, - uint64_t seq, - UInt128 write_attempt_id, - uint64_t attempt_start_boot_ms, - uint64_t confirmed_deadline_boot_ms, - const CasEventSink & event_sink) noexcept -{ - MountRenewObservabilityContext * context = currentMountRenewObservability(); - if (!context) - return; - const bool deferred = context->deferred; - const String * configured_server_root_id = context->server_root_id; - const CasEventSink * configured_event_sink = context->event_sink; - *context = MountRenewObservabilityContext{ - .active = true, - .completed = false, - .deferred = deferred, - .server_root_id = configured_server_root_id ? configured_server_root_id : &server_root_id, - .event_sink = configured_event_sink ? configured_event_sink : &event_sink, - .writer_epoch = writer_epoch, - .seq = seq, - .write_attempt_id = write_attempt_id, - .observability_start_boot_ms = defaultBootMs(), - .confirmed_deadline_boot_ms = confirmed_deadline_boot_ms, - .initial_confirmed_budget_ms = confirmed_deadline_boot_ms > attempt_start_boot_ms - ? confirmed_deadline_boot_ms - attempt_start_boot_ms - : 0, - }; -} - uint64_t elapsedSince(uint64_t start_boot_ms, uint64_t now_boot_ms) { return now_boot_ms >= start_boot_ms ? now_boot_ms - start_boot_ms : 0; } -uint64_t remainingConfirmedBudget(const MountRenewObservabilityContext & context, uint64_t now_boot_ms) +/// What was left of the lease the renewal started under when the renewal returned. +uint64_t remainingConfirmedBudget(const MountRenewResult & result, uint64_t lease_deadline_before_boot_ms) { - const uint64_t elapsed_ms = elapsedSince(context.observability_start_boot_ms, now_boot_ms); - return context.initial_confirmed_budget_ms > elapsed_ms - ? context.initial_confirmed_budget_ms - elapsed_ms + const uint64_t initial_budget_ms = lease_deadline_before_boot_ms > result.attempt_start_boot_ms + ? lease_deadline_before_boot_ms - result.attempt_start_boot_ms : 0; + return initial_budget_ms > result.elapsed_ms ? initial_budget_ms - result.elapsed_ms : 0; } void emitMountRenewEvent( - const MountRenewObservabilityContext & context, - const String & write_attempt_id, + const MountRenewResult & result, + const String & server_root_id, + const CasEventSink & event_sink, + uint64_t lease_deadline_before_boot_ms, + std::optional expired_ms, std::string_view outcome, - uint32_t attempts_sent, - uint64_t now_boot_ms, - std::string_view classification, - uint64_t remount_attempt_no) noexcept + std::string_view classification) noexcept { - if (!context.event_sink || !*context.event_sink || !context.server_root_id) + if (!event_sink) return; try { @@ -276,25 +124,24 @@ void emitMountRenewEvent( event.outcome = String{outcome}; if (outcome != "recovered") event.reason = "CAS mount renewal ended without retained authority and fenced the mount"; - else if (context.expired_ms) + else if (expired_ms) event.reason = "CAS mount renewal restored a lease that had expired"; else event.reason = "CAS mount renewal committed after a retry or a resolving read"; event.detail = { - {"server_root_id", *context.server_root_id}, - {"writer_epoch", std::to_string(context.writer_epoch)}, - {"seq", std::to_string(context.seq)}, - {"write_attempt_id", write_attempt_id}, - {"attempts_sent", std::to_string(attempts_sent)}, - {"elapsed_ms", std::to_string(elapsedSince(context.observability_start_boot_ms, now_boot_ms))}, - {"remaining_confirmed_budget_ms", std::to_string(remainingConfirmedBudget(context, now_boot_ms))}, + {"server_root_id", server_root_id}, + {"writer_epoch", std::to_string(result.writer_epoch)}, + {"seq", std::to_string(result.seq)}, + {"write_attempt_id", u128ToHex(result.write_attempt_id).substr(0, 12)}, + {"attempts_sent", std::to_string(result.attempts_sent)}, + {"elapsed_ms", std::to_string(result.elapsed_ms)}, + {"remaining_confirmed_budget_ms", + std::to_string(remainingConfirmedBudget(result, lease_deadline_before_boot_ms))}, {"classification", String{classification}}, }; - if (remount_attempt_no != 0) - event.detail["remount_attempt_no"] = std::to_string(remount_attempt_no); - if (context.expired_ms) - event.detail["expired_ms"] = std::to_string(*context.expired_ms); - (*context.event_sink)(std::move(event)); + if (expired_ms) + event.detail["expired_ms"] = std::to_string(*expired_ms); + event_sink(std::move(event)); } catch (...) { @@ -317,72 +164,84 @@ constexpr std::string_view terminalClassificationName(MountRenewTerminalClassifi return "terminal_unclassified"; } -void deliverMountRenewObservability( - const MountRenewObservabilityContext & context, uint64_t remount_attempt_no) noexcept +/// Forward declaration: defined below (same TU-unique anonymous namespace) — `allocateWriterEpoch` +/// names the current mount holder in its DecommissionRecovery live-refusal message. +String describeMountHolder(const MountLease & m); + +std::optional readOwnerObject(CasOperation & op, const Layout & l, const String & server_root_id) +{ + const auto got = op.read(l.ownerKey(server_root_id), Retry::standard()); + if (!got) + return std::nullopt; + return decodeOwner(got->bytes); +} + +void throwIfOwnerRetired(const OwnerObject & owner, const String & srid) { - if (!context.active || !context.completed || !context.server_root_id) + if (!owner.retired_at_ms) return; + throw Exception(ErrorCodes::CORRUPTED_DATA, + "CAS server-root '{}' was explicitly decommissioned by an operator (tombstoned at {} ms) " + "and is refusing to silently resume — if you genuinely intend to bring this server-root " + "back, manually clear the owner object's tombstone field and restart " + "(same manual-recovery pattern as an owner anchor lost over existing data)", + srid, *owner.retired_at_ms); +} +} + +void reportMountRenewCompletion( + const MountRenewResult & result, + const String & server_root_id, + const CasEventSink & event_sink, + uint64_t lease_deadline_before_boot_ms, + std::optional expired_ms) noexcept +{ try { - const uint64_t now_boot_ms = defaultBootMs(); - const String write_attempt_id = u128ToHex(context.write_attempt_id).substr(0, 12); - - const bool recovered = context.outcome == MountRenewOutcome::Committed - && (context.attempts_sent > 1 || context.resolved_by_read || context.expired_ms.has_value()); + const bool recovered = result.outcome == MountRenewOutcome::Committed + && (result.attempts_sent > 1 || result.resolved_by_read || expired_ms.has_value()); if (recovered) { std::string_view classification = "committed_after_expiry"; - if (context.resolved_by_read) + if (result.resolved_by_read) classification = "committed_by_read"; - else if (context.attempts_sent > 1) + else if (result.attempts_sent > 1) classification = "committed_after_retry"; emitMountRenewEvent( - context, - write_attempt_id, - "recovered", - context.attempts_sent, - now_boot_ms, - classification, - remount_attempt_no); + result, server_root_id, event_sink, lease_deadline_before_boot_ms, expired_ms, "recovered", classification); try { LOG_INFO( getLogger("CasMountLeaseRenewer"), "CAS mount renewal '{}' recovered after {} physical attempts in {} ms " "(classification={}, confirmed_deadline_boot_ms={})", - *context.server_root_id, - context.attempts_sent, - elapsedSince(context.observability_start_boot_ms, now_boot_ms), + server_root_id, + result.attempts_sent, + result.elapsed_ms, classification, - context.confirmed_deadline_boot_ms); + lease_deadline_before_boot_ms); } catch (...) { } } - else if (context.outcome == MountRenewOutcome::Terminal) + else if (result.outcome == MountRenewOutcome::Terminal) { - const std::string_view classification = terminalClassificationName(context.terminal_classification); + const std::string_view classification = terminalClassificationName(result.classification); emitMountRenewEvent( - context, - write_attempt_id, - "failed", - context.attempts_sent, - now_boot_ms, - classification, - remount_attempt_no); + result, server_root_id, event_sink, lease_deadline_before_boot_ms, expired_ms, "failed", classification); try { LOG_WARNING( getLogger("CasMountLeaseRenewer"), "CAS mount renewal '{}' fenced after {} physical attempts in {} ms " "(classification={}, confirmed_deadline_boot_ms={})", - *context.server_root_id, - context.attempts_sent, - elapsedSince(context.observability_start_boot_ms, now_boot_ms), + server_root_id, + result.attempts_sent, + result.elapsed_ms, classification, - context.confirmed_deadline_boot_ms); + lease_deadline_before_boot_ms); } catch (...) { @@ -395,79 +254,6 @@ void deliverMountRenewObservability( } } -/// Forward declaration: defined below (same TU-unique anonymous namespace) — `allocateWriterEpoch` -/// names the current mount holder in its DecommissionRecovery live-refusal message. -String describeMountHolder(const MountLease & m); - -std::optional readOwnerObject(CasOperation & op, const Layout & l, const String & server_root_id) -{ - const auto got = op.read(l.ownerKey(server_root_id), Retry::standard()); - if (!got) - return std::nullopt; - return decodeOwner(got->bytes); -} - -void throwIfOwnerRetired(const OwnerObject & owner, const String & srid) -{ - if (!owner.retired_at_ms) - return; - - throw Exception(ErrorCodes::CORRUPTED_DATA, - "CAS server-root '{}' was explicitly decommissioned by an operator (tombstoned at {} ms) " - "and is refusing to silently resume — if you genuinely intend to bring this server-root " - "back, manually clear the owner object's tombstone field and restart " - "(same manual-recovery pattern as an owner anchor lost over existing data)", - srid, *owner.retired_at_ms); -} -} - -void configureMountRenewObservability( - const String * server_root_id, const CasEventSink * event_sink, bool deferred) noexcept -{ - mount_renew_observability.pending = MountRenewObservabilityConfiguration{ - .configured = true, - .deferred = deferred, - .server_root_id = server_root_id, - .event_sink = event_sink, - }; -} - -void reportMountRenewCompletion(const MountRenewResult & result, std::optional expired_ms) noexcept -{ - if (mount_renew_observability.suppressed_depth != 0) - { - --mount_renew_observability.suppressed_depth; - return; - } - MountRenewObservabilityContext * context = currentMountRenewObservability(); - if (!context || !context->active) - return; - context->completed = true; - context->terminal_classification = result.classification; - context->outcome = result.outcome; - context->attempts_sent = std::max(context->attempts_sent, result.attempts_sent); - context->resolved_by_read = result.resolved_by_read; - context->expired_ms = expired_ms; - if (context->deferred) - return; - - /// Pop before invoking any callback. A reentrant sink gets a distinct stack slot and cannot alter - /// the completed outer snapshot. - const MountRenewObservabilityContext completed = *context; - --mount_renew_observability.depth; - deliverMountRenewObservability(completed, /*remount_attempt_no=*/0); -} - -void deliverDeferredMountRenewObservability(uint64_t remount_attempt_no) noexcept -{ - MountRenewObservabilityContext * context = currentMountRenewObservability(); - if (!context || !context->deferred) - return; - const MountRenewObservabilityContext completed = *context; - --mount_renew_observability.depth; - deliverMountRenewObservability(completed, remount_attempt_no); -} - bool serverRootSubtreeEmpty( CasOperation & op, const Layout & l, const String & srid, const RefCatalog & catalog_observation) { @@ -1557,9 +1343,6 @@ MountRenewResult MountLeaseRenewer::terminalResult(MountRenewResult result) MountRenewResult MountLeaseRenewer::renew(const MountRenewOperationEnvironment & environment) { - const MountRenewObservabilityRegistration observability_registration = beginMountRenewObservabilityCall(); - const MountRenewObservabilityCallGuard observability_guard(observability_registration); - if (renewer_state != MountLeaseRenewerState::Active) throw Exception( ErrorCodes::LOGICAL_ERROR, @@ -1579,18 +1362,6 @@ MountRenewResult MountLeaseRenewer::renew(const MountRenewOperationEnvironment & const UInt128 write_attempt_id = newMountWriteAttemptId(); const String body = encodeBody(next_seq, wall_ms, min_active_build_sequence_fn(), write_attempt_id); - if (observability_registration != MountRenewObservabilityRegistration::Ignored) - { - initializeMountRenewObservability( - srid, - writer_epoch, - next_seq, - write_attempt_id, - attempt_start_boot_ms, - confirmed_deadline_boot_ms, - event_sink); - } - MountRenewResult result; result.attempt_start_boot_ms = attempt_start_boot_ms; result.writer_epoch = writer_epoch; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h index 0977844be7c5..640c374ff70c 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h @@ -79,6 +79,16 @@ struct MountRenewResult uint64_t elapsed_ms = 0; }; +/// One event and one log line for a renewal that retried, was resolved by a read, restored an expired +/// lease, or ended terminal; silent otherwise. `lease_deadline_before_boot_ms` is the lease deadline +/// the renewal started under. Never throws. +void reportMountRenewCompletion( + const MountRenewResult & result, + const String & server_root_id, + const CasEventSink & event_sink, + uint64_t lease_deadline_before_boot_ms, + std::optional expired_ms) noexcept; + /// The spacing of a renewal's retries. inline constexpr uint64_t kMountRenewRetrySpacingMs = 1000; diff --git a/src/Disks/tests/gtest_cas_event_log.cpp b/src/Disks/tests/gtest_cas_event_log.cpp index 712552a2f4fc..59041135989a 100644 --- a/src/Disks/tests/gtest_cas_event_log.cpp +++ b/src/Disks/tests/gtest_cas_event_log.cpp @@ -9,6 +9,11 @@ #include #include #include +#include +#include +#include +#include +#include #include #include #include @@ -25,13 +30,6 @@ extern const int BAD_ARGUMENTS; extern const int NETWORK_ERROR; } -namespace DB::Cas -{ -void configureMountRenewObservability( - const String * server_root_id, const CasEventSink * event_sink, bool deferred) noexcept; -void reportMountRenewCompletion(const MountRenewResult & result, std::optional expired_ms) noexcept; -} - namespace { @@ -42,20 +40,14 @@ class RenewalEventBackend final : public InMemoryBackend bool throw_nonretryable_next_write = false; bool vanish_on_next_write = false; - /// A plain read/replace through the primitive surface, for fixtures that need to observe or seed - /// state without going through the pool under test. + /// A plain read through the primitive surface, for fixtures that need to observe state without + /// going through the pool under test. std::optional readForTest(const String & key) { DB::Cas::tests::OperationForTest op(*this); return (*op).read(key, Retry::standard()); } - bool replaceForTest(const String & key, const String & bytes, const Etag & expected) - { - DB::Cas::tests::OperationForTest op(*this); - return std::holds_alternative((*op).replace(key, bytes, expected, Retry::standard())); - } - /// The faults sit on the WRITE PRIMITIVE, and only on a CONDITIONAL write: a lease renewal is a /// replace, so the pool's own create-if-absent writes must not consume a one-shot fault. std::expected write( @@ -123,6 +115,37 @@ std::vector watermarkRenewEvents(const std::vector & events) return result; } +/// Captures what the renewer's logger writes at `information` and above while it lives. +class ScopedRenewerLogCapture +{ +public: + ScopedRenewerLogCapture() + : logger(getLogger("CasMountLeaseRenewer")) + , channel(new Poco::StreamChannel(stream)) + , old_channel(logger->getChannel(), /*shared=*/true) + , old_level(logger->getLevel()) + { + logger->setChannel(channel.get()); + logger->setLevel("information"); + } + + ~ScopedRenewerLogCapture() + { + logger->setChannel(old_channel); + logger->setLevel(old_level); + } + + String captured() const { return stream.str(); } + +private: + LoggerPtr logger; + std::ostringstream stream; // STYLE_CHECK_ALLOW_STD_STRING_STREAM + Poco::AutoPtr channel; + /// A real reference (shared=true), so the previous channel cannot die while ours is installed. + Poco::AutoPtr old_channel; + int old_level; +}; + } /// Round-B opt §6: `reason` is templated rationale (a handful of distinct strings repeated across @@ -243,96 +266,6 @@ TEST(CASEvent, WatermarkRenewEventsAreBoundedAndComplete) EXPECT_TRUE(renewals[0].detail.contains(key)) << "missing detail key " << key; } -/// Ten renewals nested through each other's conflict sinks, against an eight-slot observation stack. -/// The two calls beyond the stack get no rich event -- and must still report their own physical attempt -/// count, which rides the write result rather than the suppressed observation. -TEST(CASEvent, DeepReentrancyPreservesDeterministicPhysicalAttemptTruth) -{ - constexpr size_t depth = 10; - constexpr size_t observation_stack_capacity = 8; - std::array, depth> backends; - std::array, depth> layouts; - std::array, depth> planes; - std::array, depth> renewers; - std::array server_root_ids; - std::array sinks; - std::array renew_events{}; - uint64_t wall_ms = 100; - uint64_t boot_ms = 100; - std::optional deepest_result; - std::function renew_at; - - renew_at = [&](size_t index) - { - configureMountRenewObservability(&server_root_ids[index], &sinks[index], /*deferred=*/false); - MountRenewResult result = renewers[index]->renew(MountRenewOperationEnvironment{}); - reportMountRenewCompletion(result, std::nullopt); - return result; - }; - - for (size_t index = 0; index < depth; ++index) - { - backends[index] = std::make_shared(); - layouts[index] = std::make_unique(fmt::format("deep-renewal-{}", index)); - server_root_ids[index] = fmt::format("deep-{}", index); - sinks[index] = [&, index](CasEvent event) - { - if (event.type == CasEventType::WatermarkRenew) - ++renew_events[index]; - if (event.type == CasEventType::MountConflict && index + 1 < depth) - { - MountRenewResult child_result = renew_at(index + 1); - if (index + 2 == depth) - deepest_result = std::move(child_result); - } - }; - /// One open-fence plane per renewer, on the same injected clock the renewer anchors its lease - /// against, and with a sleep that advances it: the deepest renewal reissues, and no unit test - /// may serve the engine's jittered backoff for real. - planes[index] = std::make_unique( - backends[index], Fence::open(), [&] { return boot_ms; }, [&](uint64_t ms) { boot_ms += ms; }); - renewers[index] = std::make_unique( - *planes[index], - *planes[index], - *layouts[index], - server_root_ids[index], - UInt128(index + 1), - 7, - std::chrono::milliseconds(1000), - [&] { return wall_ms; }, - [] { return uint64_t{0}; }, - sinks[index], - std::chrono::milliseconds(0), - [&] { return boot_ms; }); - renewers[index]->start(); - - if (index + 1 < depth) - { - const String key = layouts[index]->mountKey(server_root_ids[index]); - auto observed = backends[index]->readForTest(key); - ASSERT_TRUE(observed.has_value()); - MountLease foreign = decodeMountLease(observed->bytes); - foreign.server_uuid = UInt128(100 + index); - ASSERT_TRUE(backends[index]->replaceForTest(key, encodeMountLease(foreign), observed->etag)); - } - } - /// The deepest slot is the only one nobody took, so its renewal can recover: the attempt is lost - /// before its answer, the resolve read finds the precondition intact, and the reissue commits. - backends.back()->throw_before_next_write = true; - - const MountRenewResult outer_result = renew_at(0); - EXPECT_EQ(outer_result.outcome, MountRenewOutcome::Terminal); - ASSERT_TRUE(deepest_result.has_value()); - EXPECT_EQ(deepest_result->outcome, MountRenewOutcome::Committed); - EXPECT_EQ(deepest_result->attempts_sent, 2u) - << "nesting beyond the rich-event stack must not erase physical attempt truth"; - - for (size_t index = 0; index < depth; ++index) - EXPECT_EQ(renew_events[index], index < observation_stack_capacity ? 1u : 0u) - << "renewal " << index << " is " << (index < observation_stack_capacity ? "on" : "beyond") - << " the observation stack"; -} - TEST(CASEvent, WatermarkRenewSinkFailureCannotChangeOutcome) { auto backend = std::make_shared(); @@ -514,6 +447,103 @@ TEST(CASEvent, PreCompletionConflictReentrancyPreservesOuterTerminalObservation) EXPECT_EQ(renewals[0].detail.at("classification"), "vanished"); } +/// A report needs nothing registered before the renewal: every field of the event and of the log line +/// comes from the result and from what the caller passes. +TEST(CASEvent, TheReportIsBuiltFromTheResult) +{ + /// Everything the plane, the renewer and the sink capture is declared before them. + uint64_t wall_ms = 100; + uint64_t boot_ms = 100; + const String server_root_id = "reporting"; + std::vector events; + const CasEventSink sink = [&events](CasEvent event) { events.push_back(std::move(event)); }; + auto backend = std::make_shared(); + const Layout layout("report-from-result"); + CasRequests plane(backend, Fence::open(), [&] { return boot_ms; }, [&](uint64_t ms) { boot_ms += ms; }); + MountLeaseRenewer renewer( + plane, plane, layout, server_root_id, UInt128(0x77), 7, std::chrono::milliseconds(1000), + [&] { return wall_ms; }, [] { return uint64_t{0}; }, CasEventSink{}, std::chrono::milliseconds(0), + [&] { return boot_ms; }); + renewer.start(); + const ScopedRenewerLogCapture log; + + /// The first attempt's answer is lost and the reissue commits. + backend->throw_before_next_write = true; + const MountRenewResult retried = renewer.renew(MountRenewOperationEnvironment{}); + ASSERT_EQ(retried.outcome, MountRenewOutcome::Committed); + ASSERT_EQ(retried.attempts_sent, 2u); + const uint64_t retried_deadline = retried.attempt_start_boot_ms + retried.elapsed_ms + 777; + reportMountRenewCompletion(retried, server_root_id, sink, retried_deadline, std::nullopt); + + ASSERT_EQ(events.size(), 1u); + EXPECT_EQ(events[0].type, CasEventType::WatermarkRenew); + EXPECT_EQ(events[0].outcome, "recovered"); + EXPECT_EQ(events[0].reason, "CAS mount renewal committed after a retry or a resolving read"); + EXPECT_EQ(events[0].detail.at("server_root_id"), server_root_id); + EXPECT_EQ(events[0].detail.at("writer_epoch"), "7"); + EXPECT_EQ(events[0].detail.at("seq"), "2"); + EXPECT_EQ(events[0].detail.at("write_attempt_id"), u128ToHex(retried.write_attempt_id).substr(0, 12)); + EXPECT_EQ(events[0].detail.at("attempts_sent"), "2"); + EXPECT_EQ(events[0].detail.at("elapsed_ms"), std::to_string(retried.elapsed_ms)); + EXPECT_EQ(events[0].detail.at("remaining_confirmed_budget_ms"), "777"); + EXPECT_EQ(events[0].detail.at("classification"), "committed_after_retry"); + EXPECT_EQ(events[0].detail.size(), 8u) << "the eight keys and no expired_ms"; + EXPECT_NE(log.captured().find(fmt::format( + "CAS mount renewal '{}' recovered after 2 physical attempts in {} ms " + "(classification=committed_after_retry, confirmed_deadline_boot_ms={})", + server_root_id, retried.elapsed_ms, retried_deadline)), + String::npos) + << log.captured(); + + /// A first-attempt commit that restored an expired lease. + events.clear(); + const MountRenewResult restoring = renewer.renew(MountRenewOperationEnvironment{}); + ASSERT_EQ(restoring.outcome, MountRenewOutcome::Committed); + ASSERT_EQ(restoring.attempts_sent, 1u); + reportMountRenewCompletion(restoring, server_root_id, sink, restoring.attempt_start_boot_ms, 5); + ASSERT_EQ(events.size(), 1u); + EXPECT_EQ(events[0].outcome, "recovered"); + EXPECT_EQ(events[0].reason, "CAS mount renewal restored a lease that had expired"); + EXPECT_EQ(events[0].detail.at("seq"), "3"); + EXPECT_EQ(events[0].detail.at("classification"), "committed_after_expiry"); + EXPECT_EQ(events[0].detail.at("expired_ms"), "5"); + EXPECT_EQ(events[0].detail.at("remaining_confirmed_budget_ms"), "0"); + + /// A first-attempt commit and a renewal that never ran report nothing. + events.clear(); + const MountRenewResult quiet = renewer.renew(MountRenewOperationEnvironment{}); + ASSERT_EQ(quiet.outcome, MountRenewOutcome::Committed); + ASSERT_EQ(quiet.attempts_sent, 1u); + reportMountRenewCompletion(quiet, server_root_id, sink, quiet.attempt_start_boot_ms + 1000, std::nullopt); + const MountRenewResult skipped = renewer.renew(MountRenewOperationEnvironment{ + .boot_ms = {}, + .live = [] { return false; }, + .cancelled = [] { return true; }, + .on_request = {}, + }); + ASSERT_EQ(skipped.outcome, MountRenewOutcome::NotAttempted); + reportMountRenewCompletion(skipped, server_root_id, sink, skipped.attempt_start_boot_ms + 1000, std::nullopt); + EXPECT_TRUE(events.empty()); + + /// A terminal renewal: the slot vanished under it. + backend->vanish_on_next_write = true; + const MountRenewResult vanished = renewer.renew(MountRenewOperationEnvironment{}); + ASSERT_EQ(vanished.outcome, MountRenewOutcome::Terminal); + const uint64_t vanished_deadline = vanished.attempt_start_boot_ms + 1000; + reportMountRenewCompletion(vanished, server_root_id, sink, vanished_deadline, std::nullopt); + ASSERT_EQ(events.size(), 1u); + EXPECT_EQ(events[0].outcome, "failed"); + EXPECT_EQ(events[0].reason, "CAS mount renewal ended without retained authority and fenced the mount"); + EXPECT_EQ(events[0].detail.at("classification"), "vanished"); + EXPECT_EQ(events[0].detail.at("attempts_sent"), std::to_string(vanished.attempts_sent)); + EXPECT_NE(log.captured().find(fmt::format( + "CAS mount renewal '{}' fenced after {} physical attempts in {} ms " + "(classification=vanished, confirmed_deadline_boot_ms={})", + server_root_id, vanished.attempts_sent, vanished.elapsed_ms, vanished_deadline)), + String::npos) + << log.captured(); +} + /// Round-B opt §6: `emitEvent` takes the event BY VALUE (moved-through, not `const &`), so a /// caller's local is genuinely moved-from -- not merely copied via a const reference -- by the time /// the sink runs. Mirrors `makeCasEventSink`'s own move-out-of-the-by-value-event idiom (a small test From 68462f8f8cb2194945f9480e52c9e55ff4b479aa Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 09:23:26 +0200 Subject: [PATCH 20/37] Pass the lease deadline to the CAS mount farewell MountLeaseRenewer::release takes the deadline that bounds the farewell, and the renewer no longer keeps its own copy of the lease deadline. finishTeardown passes the start of the last committed renewal plus the TTL, the value the copy held. Co-Authored-By: Claude Opus 5.5 --- .../ContentAddressed/Pool/CasMountRuntime.cpp | 6 +- .../ContentAddressed/Pool/CasServerRoot.cpp | 27 ++---- .../ContentAddressed/Pool/CasServerRoot.h | 7 +- src/Disks/tests/gtest_cas_heartbeat.cpp | 85 ++++++++++++++++--- src/Disks/tests/gtest_cas_mount.cpp | 4 +- 5 files changed, 93 insertions(+), 36 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index 1f65d92b3689..06190b7eb8d6 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -1077,9 +1077,13 @@ void CasMountRuntime::finishTeardown(bool drained) return; if (drained && mount_renewer->state() == MountLeaseRenewerState::Active) { + const uint64_t ttl_ms = static_cast(config.mount_lease_ttl_ms.count()); + const uint64_t start_boot_ms = mount_renewer->lastCommittedAttemptStartBootMs(); try { - mount_renewer->release(); + mount_renewer->release(start_boot_ms > std::numeric_limits::max() - ttl_ms + ? std::numeric_limits::max() + : start_boot_ms + ttl_ms); } catch (...) { diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp index 3b3bb2d1d5fd..edb057ffce9c 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp @@ -1237,10 +1237,6 @@ uint64_t MountLeaseRenewer::start(Liveness liveness) seq = 1; last_etag = etag; last_committed_attempt_start_boot_ms = attempt_start_boot_ms; - const uint64_t ttl_ms = static_cast(ttl.count()); - confirmed_deadline_boot_ms = attempt_start_boot_ms > std::numeric_limits::max() - ttl_ms - ? std::numeric_limits::max() - : attempt_start_boot_ms + ttl_ms; renewer_state = MountLeaseRenewerState::Active; return attempt_start_boot_ms; } @@ -1402,10 +1398,6 @@ MountRenewResult MountLeaseRenewer::renew(const MountRenewOperationEnvironment & seq = next_seq; last_etag = std::move(committed->etag); last_committed_attempt_start_boot_ms = attempt_start_boot_ms; - const uint64_t ttl_ms = static_cast(ttl.count()); - confirmed_deadline_boot_ms = attempt_start_boot_ms > std::numeric_limits::max() - ttl_ms - ? std::numeric_limits::max() - : attempt_start_boot_ms + ttl_ms; result.outcome = MountRenewOutcome::Committed; result.attempts_sent = committed->attempts_sent; result.resolved_by_read = committed->resolved_by_read; @@ -1485,7 +1477,7 @@ MountRenewResult MountLeaseRenewer::renew(const MountRenewOperationEnvironment & "CAS mount-lease: the renewal of key '{}' was declined, which a replace cannot report", key); } -void MountLeaseRenewer::terminate(CasOperation & op) +void MountLeaseRenewer::terminate(CasOperation & op, uint64_t lease_deadline_boot_ms) { const uint64_t wall_ms = now_ms_fn(); const String body = encodeMountLease(MountLease{ @@ -1504,10 +1496,10 @@ void MountLeaseRenewer::terminate(CasOperation & op) /// `CasOperation::writeLoop` -- is exactly `2 * open_requests.attemptReservationMs()`. A window /// below that value refuses the write before its first attempt, deterministically, on every call: /// `kFarewellBudgetMs` alone predates the attempt-envelope reservation and can no longer be trusted - /// to admit it. Saturating, like every other deadline computation on this path (see the - /// `expires_at_ms`/`confirmed_deadline_boot_ms` arithmetic above): an operator-configured envelope - /// is not bounds-checked against this doubling, and wrapping past `UINT64_MAX` would turn a too-long - /// window into a too-SHORT one -- the exact failure mode this fix exists to remove. + /// to admit it. Saturating, like every other deadline computation on this path: an + /// operator-configured envelope is not bounds-checked against this doubling, and wrapping past + /// `UINT64_MAX` would turn a too-long window into a too-SHORT one -- the exact failure mode this fix + /// exists to remove. const uint64_t reservation_ms = open_requests.attemptReservationMs(); const uint64_t doubled_reservation_ms = reservation_ms > std::numeric_limits::max() / 2 ? std::numeric_limits::max() @@ -1520,11 +1512,8 @@ void MountLeaseRenewer::terminate(CasOperation & op) /// node's own fence may already be gone. The precondition on this write already stops it from clobbering /// a successor if it DOES land late, but a shutdown holding the process open to retry a write past /// its own lease-safe deadline serves no one -- the successor's own reclaim does not wait for it. - /// `confirmed_deadline_boot_ms` is set at `start()` and kept current by every successful `renew`, - /// so it is valid here whenever `terminate` runs (only reachable from `release`, which requires - /// `Active`, which `start` alone establishes). WriteResult written = op.replace(key, body, precondition(), - Retry::untilLeaseSafe(confirmed_deadline_boot_ms, static_cast(lease_safety_margin.count()), farewell_window_ms)); + Retry::untilLeaseSafe(lease_deadline_boot_ms, static_cast(lease_safety_margin.count()), farewell_window_ms)); if (Committed * committed = std::get_if(&written)) { @@ -1557,7 +1546,7 @@ void MountLeaseRenewer::terminate(CasOperation & op) orThrow(std::move(written), fmt::format("CAS mount-lease release of key '{}'", key)); } -void MountLeaseRenewer::release() +void MountLeaseRenewer::release(uint64_t lease_deadline_boot_ms) { if (renewer_state != MountLeaseRenewerState::Active) throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount-lease: release is allowed only in Active state for key '{}'", key); @@ -1566,7 +1555,7 @@ void MountLeaseRenewer::release() /// run down still has to hand the slot back, and refusing the write there would leave the slot /// looking live until GC fences it out. CasOperation op = open_requests.admit(); - terminate(op); + terminate(op, lease_deadline_boot_ms); } void sweepOwnMountStaging(IObjectStorage & object_storage, const String & mount_staging_prefix) noexcept diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h index 640c374ff70c..23c0fde83ce6 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h @@ -597,7 +597,9 @@ class MountLeaseRenewer /// One renewal on `lease_requests` under `Retry::untilDefinitive(kMountRenewRetrySpacingMs)`. Ends on /// a definitive answer, a deterministic local failure, or when `environment.live` refuses. MountRenewResult renew(const MountRenewOperationEnvironment & environment); - void release(); + /// The farewell. It gives up rather than run past `lease_deadline_boot_ms` less the safety margin; + /// the deadline is on the boot clock this renewer anchors its renewals on. + void release(uint64_t lease_deadline_boot_ms); MountLeaseRenewerState state() const { return renewer_state; } bool canRelease() const { return renewer_state == MountLeaseRenewerState::Active; } @@ -613,7 +615,7 @@ class MountLeaseRenewer /// holds the classification before the event sink runs, so the sink cannot change it. [[noreturn]] void throwRenewConflict(const Observation & seen, MountRenewTerminalClassification & classification) const; MountRenewResult terminalResult(MountRenewResult result); - void terminate(CasOperation & op); + void terminate(CasOperation & op, uint64_t lease_deadline_boot_ms); CasRequests & open_requests; /// The plane of a renewal: no lease budget, and a sleep a stop wakes. @@ -636,7 +638,6 @@ class MountLeaseRenewer /// The incarnation our last landed write created; every renewal and the farewell name it as the /// precondition. Unset only before `start` has landed one. std::optional last_etag; - uint64_t confirmed_deadline_boot_ms = 0; uint64_t last_committed_attempt_start_boot_ms = 0; }; diff --git a/src/Disks/tests/gtest_cas_heartbeat.cpp b/src/Disks/tests/gtest_cas_heartbeat.cpp index 18d8ae5a83bf..5d02938b72dd 100644 --- a/src/Disks/tests/gtest_cas_heartbeat.cpp +++ b/src/Disks/tests/gtest_cas_heartbeat.cpp @@ -346,7 +346,7 @@ TEST(CASHeartbeat, StopStampsExpiredAndFarewellSentinel) renewer.start(); now_ms = 2000; - renewer.release(); + renewer.release(renewer.lastCommittedAttemptStartBootMs() + 100); auto m = decodeMountLease(ops.op.read(layout.mountKey(srid), Retry::standard())->bytes); /// Terminal body stamps the lease already-expired (so a same-server reopen reclaims immediately) @@ -402,7 +402,7 @@ TEST(CASHeartbeat, FarewellIsAdmittedUnderTheDefaultBudget) renewer.start(); now_ms = 2000; - EXPECT_NO_THROW(renewer.release()) + EXPECT_NO_THROW(renewer.release(renewer.lastCommittedAttemptStartBootMs() + 30000)) << "the farewell's policy window must admit the write's own two-envelope reservation " "(2 * 7000 ms with the shipped defaults) -- otherwise a clean shutdown never hands the " "mount slot back and every restart pays a full incarnation-stability observation"; @@ -435,7 +435,7 @@ TEST(CASHeartbeat, FarewellIsAdmittedUnderADifferentEnvelope) renewer.start(); now_ms = 2000; - EXPECT_NO_THROW(renewer.release()) + EXPECT_NO_THROW(renewer.release(renewer.lastCommittedAttemptStartBootMs() + 40000)) << "the farewell's policy window must be DERIVED from this backend's own envelope " "(2 * 9000 ms), not hardcoded to the shipped-default window -- a window fixed at " "16000 ms would refuse this write's 18000 ms reservation"; @@ -474,7 +474,7 @@ TEST(CASHeartbeat, FarewellIsRefusedWhenTheLeaseExpiresBeforeItsDerivedWindow) bool threw = false; try { - renewer.release(); + renewer.release(renewer.lastCommittedAttemptStartBootMs() + 5000); } catch (const DB::Exception & e) { @@ -493,6 +493,69 @@ TEST(CASHeartbeat, FarewellIsRefusedWhenTheLeaseExpiresBeforeItsDerivedWindow) << "the refused write must not have landed"; } +/// The farewell is bounded by the deadline its caller passes. A near deadline refuses it although the +/// renewer's own lease would admit it, and a far one admits it although that lease would refuse it. +TEST(CASHeartbeat, FarewellIsBoundByTheDeadlineItIsGiven) +{ + Layout layout("pool"); + const String srid = "test"; + const UInt128 uuid(0x1234); + + { + auto backend = std::make_shared(); + uint64_t now_ms = 1000; + uint64_t boot_ms = 100; + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/30000); + MountLeaseRenewer renewer(ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, + std::chrono::milliseconds(30000), [&] { return now_ms; }, + [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(2000), + [&] { return boot_ms; }); + renewer.start(); + + now_ms = 2000; + String message; + int code = 0; + try + { + /// A TTL of 30000 admits this farewell (`FarewellIsAdmittedUnderTheDefaultBudget`); the + /// deadline passed here leaves too little past the margin for the write's reservation. + renewer.release(boot_ms + 5000); + ADD_FAILURE() << "a farewell must be refused when the deadline it is given cannot admit it"; + } + catch (const DB::Exception & e) + { + message = e.message(); + code = e.code(); + } + EXPECT_EQ(code, DB::ErrorCodes::NETWORK_ERROR) << message; + EXPECT_NE(message.find("gave up at the lease deadline after zero attempt(s)"), String::npos) << message; + EXPECT_NE(decodeMountLease(ops.op.read(layout.mountKey(srid), Retry::standard())->bytes).min_active_build_sequence, + std::numeric_limits::max()) + << "the refused farewell must not have landed"; + } + + { + auto backend = std::make_shared(); + uint64_t now_ms = 1000; + uint64_t boot_ms = 100; + Ops ops(backend, &boot_ms); + seedOwnClaim(ops.op, layout, srid, uuid, /*epoch=*/9, now_ms, /*ttl_ms=*/5000); + MountLeaseRenewer renewer(ops.farewell, ops.lease, layout, srid, uuid, /*writer_epoch=*/9, + std::chrono::milliseconds(5000), [&] { return now_ms; }, + [] { return uint64_t{5}; }, {}, std::chrono::milliseconds(2000), + [&] { return boot_ms; }); + renewer.start(); + + now_ms = 2000; + /// A TTL of 5000 refuses this farewell (`FarewellIsRefusedWhenTheLeaseExpiresBeforeItsDerivedWindow`). + EXPECT_NO_THROW(renewer.release(boot_ms + 30000)); + const MountLease farewell = decodeMountLease(ops.op.read(layout.mountKey(srid), Retry::standard())->bytes); + EXPECT_LE(farewell.expires_at_ms, now_ms); + EXPECT_EQ(farewell.min_active_build_sequence, std::numeric_limits::max()); + } +} + /// The lease bound added above must not change what an ordinary Conflict outcome does: a successor /// that took the slot (a different, unfenced incarnation) before this node's own shutdown could /// publish its farewell must be left untouched, and the release must report the conflict rather than @@ -532,7 +595,7 @@ TEST(CASHeartbeat, ForeignIncarnationDuringFarewellLeavesTheSuccessorUntouchedAn int code = 0; try { - renewer.release(); + renewer.release(renewer.lastCommittedAttemptStartBootMs() + 100); FAIL() << "a farewell that finds a foreign, unfenced incarnation must report the conflict, " "not silently succeed or clobber the successor"; } @@ -766,7 +829,7 @@ TEST(CASMountAudit, RenewerAdoptEmitsClaimAndTerminateEmitsRelease) seen.clear(); now_ms = 2000; - renewer.release(); + renewer.release(renewer.lastCommittedAttemptStartBootMs() + 100); ASSERT_EQ(seen.size(), 1u); EXPECT_EQ(seen[0].type, CasEventType::MountRelease); @@ -934,15 +997,15 @@ TEST(CASHeartbeat, RenewerStateAllowsOnlyActiveReleaseOrTerminal) [&] { return boot_ms; }); EXPECT_EQ(renewer.state(), MountLeaseRenewerState::New); EXPECT_RENEWER_STATE_REJECTION(renewer.renew(renewalEnvironment(boot_ms))); - EXPECT_RENEWER_STATE_REJECTION(renewer.release()); + EXPECT_RENEWER_STATE_REJECTION(renewer.release(renewer.lastCommittedAttemptStartBootMs() + 1000)); EXPECT_EQ(renewer.start(), 100u); EXPECT_RENEWER_STATE_REJECTION(renewer.start()); EXPECT_EQ(renewer.state(), MountLeaseRenewerState::Active); - renewer.release(); + renewer.release(renewer.lastCommittedAttemptStartBootMs() + 1000); EXPECT_EQ(renewer.state(), MountLeaseRenewerState::Released); EXPECT_RENEWER_STATE_REJECTION(renewer.start()); EXPECT_RENEWER_STATE_REJECTION(renewer.renew(renewalEnvironment(boot_ms))); - EXPECT_RENEWER_STATE_REJECTION(renewer.release()); + EXPECT_RENEWER_STATE_REJECTION(renewer.release(renewer.lastCommittedAttemptStartBootMs() + 1000)); } { @@ -967,7 +1030,7 @@ TEST(CASHeartbeat, RenewerStateAllowsOnlyActiveReleaseOrTerminal) EXPECT_EQ(renewer.state(), MountLeaseRenewerState::RenewalTerminal); EXPECT_RENEWER_STATE_REJECTION(renewer.start()); EXPECT_RENEWER_STATE_REJECTION(renewer.renew(renewalEnvironment(boot_ms))); - EXPECT_RENEWER_STATE_REJECTION(renewer.release()); + EXPECT_RENEWER_STATE_REJECTION(renewer.release(renewer.lastCommittedAttemptStartBootMs() + 1000)); } #undef EXPECT_RENEWER_STATE_REJECTION @@ -1065,7 +1128,7 @@ TEST(CASHeartbeat, CancellationBeforeSendIsNotAttemptedAndAllowsRelease) EXPECT_EQ(result.failure, nullptr); EXPECT_EQ(renewer.state(), MountLeaseRenewerState::Active); EXPECT_TRUE(backend->attempts.empty()); - EXPECT_NO_THROW(renewer.release()); + EXPECT_NO_THROW(renewer.release(renewer.lastCommittedAttemptStartBootMs() + 1000)); EXPECT_EQ(renewer.state(), MountLeaseRenewerState::Released); } diff --git a/src/Disks/tests/gtest_cas_mount.cpp b/src/Disks/tests/gtest_cas_mount.cpp index c3037c9716be..f647fef8aea9 100644 --- a/src/Disks/tests/gtest_cas_mount.cpp +++ b/src/Disks/tests/gtest_cas_mount.cpp @@ -861,7 +861,7 @@ TEST(CASMountLease, TerminateAfterVanishedBackingStoreIsNoOpRelease) /// a renewal, so the farewell's guarded write is the first thing to observe it. ASSERT_EQ(ops.op.removeCurrent(mount_key, Retry::standard()), Removal::Removed); - EXPECT_NO_THROW(k.release()) + EXPECT_NO_THROW(k.release(k.lastCommittedAttemptStartBootMs() + 100)) << "clean release against a vanished store must be a no-op, not a LOGICAL_ERROR abort"; EXPECT_EQ(ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(), lost_before); } @@ -2366,7 +2366,7 @@ TEST(CASMountLease, FarewellRunsOnAnOpenFenceAfterTheMountFenceIsLost) fence_lost = true; - EXPECT_NO_THROW(departing.release()); + EXPECT_NO_THROW(departing.release(departing.lastCommittedAttemptStartBootMs() + 1000)); const MountLease farewell = decodeMountLease(seed.read(l.mountKey("departing"), Retry::standard())->bytes); EXPECT_EQ(farewell.min_active_build_sequence, std::numeric_limits::max()); } From 531f6cb0f2849d4bc228f1452417acc71957cf65 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 09:35:09 +0200 Subject: [PATCH 21/37] Define the two-envelope write cost once in CasRequestBudget `refAppendFenceOk` and `validateCasRequestBudget` each doubled the attempt envelope with their own saturation. Both call `writeAndSettlementReadMs` now, and the `refAppendReservationMs` helper that wrapped the doubling is gone. `MountLeaseRenewer::terminate` keeps its own reservation: it doubles the plane's reservation, not the budget's envelope. Co-Authored-By: Claude Sonnet 5.5 --- .../Backend/CasRequestBudget.cpp | 12 +++++++---- .../Backend/CasRequestBudget.h | 5 +++++ .../ContentAddressed/Pool/CasMountRuntime.cpp | 13 ++++-------- .../ContentAddressed/Pool/CasMountRuntime.h | 2 -- src/Disks/tests/gtest_cas_mount.cpp | 20 +++++++++++++++++++ 5 files changed, 37 insertions(+), 15 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.cpp index ca00928ef9e6..4aa95cf930d6 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.cpp @@ -27,6 +27,12 @@ uint64_t CasRequestBudget::attemptEnvelopeMs() const : attempt_timeout_ms + connects; } +uint64_t CasRequestBudget::writeAndSettlementReadMs() const +{ + const uint64_t envelope = attemptEnvelopeMs(); + return envelope > std::numeric_limits::max() / 2 ? std::numeric_limits::max() : 2 * envelope; +} + void validateCasRequestBudget(const CasRequestBudget & budget, uint64_t mount_lease_ttl_ms, uint64_t mount_renew_period_ms, bool background_renewal) { @@ -47,11 +53,9 @@ void validateCasRequestBudget(const CasRequestBudget & budget, uint64_t mount_le if (background_renewal) { /// A renewal is a write: two envelopes (the attempt and its settlement read) after one period. - /// Saturating doubling first (matching the production horizon checks' own arithmetic), then - /// subtraction-based comparisons against the TTL -- no truncating division, so this enforces + /// Subtraction-based comparisons against the TTL, no truncating division, so this enforces /// exactly the inequality the exception message states, not an off-by-one-tighter one. - const uint64_t two_envelope = envelope > std::numeric_limits::max() / 2 - ? std::numeric_limits::max() : 2 * envelope; + const uint64_t two_envelope = budget.writeAndSettlementReadMs(); const bool cadence_fits = mount_renew_period_ms < mount_lease_ttl_ms && two_envelope < mount_lease_ttl_ms - mount_renew_period_ms && budget.lease_safety_margin_ms < mount_lease_ttl_ms - mount_renew_period_ms - two_envelope; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.h index 1d96e41b199d..f255cfbbe211 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.h @@ -33,6 +33,11 @@ struct CasRequestBudget /// every attempt it starts. uint64_t attemptEnvelopeMs() const; + /// What a conditional write can cost end to end: its attempt and the read that settles it, two + /// envelopes, saturating. Ref-log appends are admitted against this, and so is the renewal cadence rule + /// of `validateCasRequestBudget`. + uint64_t writeAndSettlementReadMs() const; + /// Recovery-level retry (`CasRefLedger::ensureRefTableRecovered`): a whole ref-table recovery /// attempt (LIST + snapshot/log GETs + seal PUT) that fails with a transient NETWORK_ERROR is /// retried, with capped-exponential backoff, until this total wall-clock budget is spent — then the diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index 06190b7eb8d6..0db29a68fda1 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -156,16 +156,11 @@ Fence::Admit CasMountRuntime::budgetAdmits(uint64_t deadline_boot_ms, uint64_t n return Fence::Admit::Ok; } -uint64_t CasMountRuntime::refAppendReservationMs() const -{ - /// Two envelopes: a write and its settlement read, which is what `writeLoop` reserves. - const uint64_t envelope_ms = cas_request_budget.attemptEnvelopeMs(); - return envelope_ms > std::numeric_limits::max() / 2 ? std::numeric_limits::max() : 2 * envelope_ms; -} - bool CasMountRuntime::refAppendFenceOk() const { - return admit(fenceGeneration(), refAppendReservationMs()) == Fence::Admit::Ok; + /// A write and its settlement read, which is what `writeLoop` reserves: a ref-log attempt is not + /// started when it cannot plausibly finish, safety margin included, before the lease expires. + return admit(fenceGeneration(), cas_request_budget.writeAndSettlementReadMs()) == Fence::Admit::Ok; } std::optional CasMountRuntime::leaseExpiredAt(uint64_t now_boot_ms) const @@ -361,7 +356,7 @@ bool CasMountRuntime::canArm(uint64_t deadline_boot_ms) const return !workers_stop_requested && !remountTerminal() && remount_requested_generation <= remount_handled_generation - && budgetAdmits(deadline_boot_ms, bootMsNow(), refAppendReservationMs()) == Fence::Admit::Ok; + && budgetAdmits(deadline_boot_ms, bootMsNow(), cas_request_budget.writeAndSettlementReadMs()) == Fence::Admit::Ok; } bool CasMountRuntime::lossNeedsNewRequest() const diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index 2853ca062927..1a1b48ee2e3c 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -463,8 +463,6 @@ class CasMountRuntime bool canArm(uint64_t deadline_boot_ms) const; /// `admit`'s budget verdict for a lease that ends at `deadline_boot_ms`, at `now_boot_ms`. Fence::Admit budgetAdmits(uint64_t deadline_boot_ms, uint64_t now_boot_ms, uint64_t needed_ms) const; - /// What a ref append reserves: a write and the read that settles it, two attempt envelopes. - uint64_t refAppendReservationMs() const; /// Whether a new loss needs a new remount generation. Requires `driver_mutex`. False only while a /// request that no reclaim has snapshotted is pending: the reclaim that serves it latches after the /// loss. diff --git a/src/Disks/tests/gtest_cas_mount.cpp b/src/Disks/tests/gtest_cas_mount.cpp index f647fef8aea9..8182a61d8204 100644 --- a/src/Disks/tests/gtest_cas_mount.cpp +++ b/src/Disks/tests/gtest_cas_mount.cpp @@ -1340,6 +1340,26 @@ TEST(CASRequestBudget, ValidateAcceptsDefaultsAndRejectsAnOverflowingSumWithoutW }); } +TEST(CASRequestBudget, WriteAndSettlementReadIsTwoEnvelopesAndSaturates) +{ + const CasRequestBudget defaults{}; + EXPECT_EQ(defaults.attemptEnvelopeMs(), 7000u); + EXPECT_EQ(defaults.writeAndSettlementReadMs(), 14000u); + + /// No S3 client: the envelope is the attempt alone. + const CasRequestBudget no_connect_cap{.attempt_timeout_ms = 5000, .connect_timeout_cap_ms = std::nullopt}; + EXPECT_EQ(no_connect_cap.writeAndSettlementReadMs(), 10'000u); + + /// The largest envelope that doubles without wrapping, and the first one that does not. + constexpr uint64_t half_max = std::numeric_limits::max() / 2; + const CasRequestBudget fits{.attempt_timeout_ms = half_max - 2000}; + ASSERT_EQ(fits.attemptEnvelopeMs(), half_max); + EXPECT_EQ(fits.writeAndSettlementReadMs(), 2 * half_max); + const CasRequestBudget wraps{.attempt_timeout_ms = half_max - 1999}; + ASSERT_EQ(wraps.attemptEnvelopeMs(), half_max + 1); + EXPECT_EQ(wraps.writeAndSettlementReadMs(), std::numeric_limits::max()); +} + /// Pool::open must call validateCasRequestBudget itself (not just the free function in isolation, /// pinned directly above): an inconsistent cas_request_budget must refuse a writable mount end-to-end /// (RFC cas-s3-timeout-retry-control §required-timeout-model), never mount silently with a budget that From dabc19a0aa4603413f133dbb58df09696d66437d Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 09:37:27 +0200 Subject: [PATCH 22/37] Name the default safety margin and the default write window `2000` stood in three places and `90'000` in three. `kFarewellSlackMs` is a different quantity and keeps its own value. Co-Authored-By: Claude Sonnet 5.5 --- .../ContentAddressed/Backend/CasRequestBudget.h | 8 ++++++-- .../ContentAddressed/Backend/CasRetry.h | 13 ++++++++----- .../ContentAddressed/ContentAddressedSettings.cpp | 3 ++- .../ContentAddressed/Pool/CasServerRoot.h | 3 ++- 4 files changed, 18 insertions(+), 9 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.h index f255cfbbe211..59dc55366c32 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.h @@ -5,6 +5,9 @@ namespace DB::Cas { +/// The default margin kept between a request and the mount lease deadline. +inline constexpr uint64_t kDefaultLeaseSafetyMarginMs = 2000; + /// The limits a writable mount is configured with. `CasMountRuntime::admit` measures a request against /// `lease_safety_margin_ms` plus a caller-supplied need expressed in attempt envelopes (one for an /// ordinary attempt, TWO for a ref-log append's write-plus-settlement-read -- see @@ -20,7 +23,7 @@ struct CasRequestBudget /// Startup-only margin folded into `validateCasRequestBudget`'s inequality against the mount lease /// TTL. Not consulted at runtime by the engine itself -- the caller's fence (backed by the local /// write fence's own deadline) is what actually gates lease-relative timing per attempt. - uint64_t lease_safety_margin_ms = 2000; + uint64_t lease_safety_margin_ms = kDefaultLeaseSafetyMarginMs; /// The cap the single-attempt client puts on one TCP connect and again on one TLS handshake, /// frozen when the pool opens as `min(disk connect_timeout_ms, attempt_timeout_ms)` (a configured @@ -43,7 +46,8 @@ struct CasRequestBudget /// retried, with capped-exponential backoff, until this total wall-clock budget is spent — then the /// error propagates and the table's load fails for this touch (the `lazy_load_tables` database /// setting makes the NEXT touch retry). This sits ON TOP of each write's own `Retry` policy window - /// (`Retry::standard()`'s 90s): one recovery attempt may itself burn ~90s inside a single seal PUT. + /// (`Retry::standard()`, `kStandardWriteWindowMs`): one recovery attempt may itself spend a whole + /// window inside a single seal PUT. /// Independent of the mount-lease invariants validated in `validateCasRequestBudget` — not part of /// that inequality set. uint64_t recovery_retry_budget_ms = 120000; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRetry.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRetry.h index 7534b0810962..f06ade446d09 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRetry.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRetry.h @@ -6,6 +6,9 @@ namespace DB::Cas { +/// The window of the default write policy. +inline constexpr uint64_t kStandardWriteWindowMs = 90'000; + /// A retry policy for one logical CAS write, expressed WITHOUT touching a clock: `window_ms` is the /// policy's own budget measured from the call's start, and `lease_deadline_ms` (already reduced by /// the caller's safety margin) is an absolute bound on whatever clock the caller's mount lease is @@ -42,22 +45,22 @@ struct Retry /// A policy with `ms` milliseconds of its own budget and no lease bound. static Retry within(uint64_t ms) { return {.window_ms = ms, .lease_deadline_ms = std::nullopt, .single_attempt = false}; } - /// `within(90'000)` -- the default write policy. - static Retry standard() { return within(90'000); } + /// `within(kStandardWriteWindowMs)` -- the default write policy. + static Retry standard() { return within(kStandardWriteWindowMs); } /// A policy bound by the mount lease: `lease_deadline_ms` minus `margin`, clamped at 0 -- never /// risk a write landing after this node's fence may already be gone. `window_ms` defaults to the - /// standard 90 s write budget; a caller whose own budget is deliberately much smaller (the + /// default write window, `kStandardWriteWindowMs`; a caller whose own budget is deliberately much smaller (the /// graceful-shutdown farewell, whose window is derived from what ONE write costs, not from the /// standard policy) passes its own window explicitly, and `bind` still takes whichever of the two /// bounds is smaller. - static Retry untilLeaseSafe(uint64_t lease_deadline_ms, uint64_t margin, uint64_t window_ms = 90'000) + static Retry untilLeaseSafe(uint64_t lease_deadline_ms, uint64_t margin, uint64_t window_ms = kStandardWriteWindowMs) { return {.window_ms = window_ms, .lease_deadline_ms = lease_deadline_ms > margin ? lease_deadline_ms - margin : 0, .single_attempt = false}; } /// The standard policy, but at most one attempt is ever sent. - static Retry once() { return {.window_ms = 90'000, .lease_deadline_ms = std::nullopt, .single_attempt = true}; } + static Retry once() { return {.window_ms = kStandardWriteWindowMs, .lease_deadline_ms = std::nullopt, .single_attempt = true}; } /// No window and no lease bound: retried until a definitive answer, or until the fence or the /// caller's liveness ends it. `bind` saturates, so the only horizon is the range of the clock. diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.cpp index 7331d2bad29a..9e6883d8c775 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.cpp @@ -2,6 +2,7 @@ #include #include +#include #include #include #include @@ -82,7 +83,7 @@ constexpr std::string_view CAS_KEY_PREFIX = "cas_"; DECLARE(UInt64, gc_io_concurrency, 16, "Maximum number of threads in the GC I/O pool. Used for fold and rebuild read-ahead, orphan-manifest sweep planning reads, and pending_deletes HEAD plus conditional DELETE. Per-hash meta writes use gc_meta_pool_size; other GC requests run on the round thread. 1 disables parallel GC I/O", 0) \ DECLARE(UInt64, gc_bulk_delete_chunk_keys, 1000, "Keys per batch delete request in GC's write-once families (owner-removed manifest bodies, covered ref logs and snapshots); 1 to 1000", 0) \ DECLARE(UInt64, attempt_timeout_ms, 5000, "Budget for one HTTP attempt of a writable Native mount's control-plane requests (read, head, list, remove, conditional write), at least 1. With the connect cap it forms the attempt envelope the lease arithmetic reserves", 0) \ - DECLARE(UInt64, lease_safety_margin_ms, 2000, "Startup-only margin validated against the mount lease TTL: attempt envelope + this must be strictly less than the TTL, and renew period + 2 × envelope + this too", 0) \ + DECLARE(UInt64, lease_safety_margin_ms, Cas::kDefaultLeaseSafetyMarginMs, "Startup-only margin validated against the mount lease TTL: attempt envelope + this must be strictly less than the TTL, and renew period + 2 × envelope + this too", 0) \ DECLARE(String, staging_backend, "local", "Blob staging backend (local | s3); s3 is opt-in", 0) \ DECLARE_SETTINGS_TRAITS(ContentAddressedSettingsTraits, LIST_OF_CONTENT_ADDRESSED_SETTINGS, CONTENT_ADDRESSED_SETTINGS_SUPPORTED_TYPES) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h index 23c0fde83ce6..08eab63300c2 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h @@ -1,4 +1,5 @@ #pragma once +#include #include #include #include @@ -586,7 +587,7 @@ class MountLeaseRenewer uint64_t writer_epoch_, std::chrono::milliseconds ttl_, std::function now_ms_fn_, std::function min_active_build_sequence_fn_, CasEventSink event_sink_ = {}, - std::chrono::milliseconds lease_safety_margin_ = std::chrono::milliseconds(2000), + std::chrono::milliseconds lease_safety_margin_ = std::chrono::milliseconds(static_cast(kDefaultLeaseSafetyMarginMs)), /// boot-domain clock for the on_renew_ok anchor; empty = real CLOCK_BOOTTIME. Injectable for /// tests and wired by CasMountRuntime::installRenewer. std::function boot_ms_fn_ = {}); From 48fd2373c3c433cb0addc503993b32a367e14257 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 09:45:37 +0200 Subject: [PATCH 23/37] Correct the lease margin, grace, FORGET trip and renewal descriptions The safety margin is read on every admission and by the farewell, not only at startup. The materialization grace and the unclean-boundary marker no longer exist. FORGET's trip leaves a live pool `Live`: the published intent suppresses the loss transition. `CASMountLeaseLost`, the `MountRenewTerminalClassification` comment and several test comments named bounded renewals, parked paths and workers that are gone. Co-Authored-By: Claude Sonnet 5.5 --- docs/en/antalya/cas/configuration.md | 2 +- src/Common/ProfileEvents.cpp | 2 +- .../Backend/CasRequestBudget.h | 14 ++++------- .../ContentAddressedMetadataStorage.h | 2 +- .../ContentAddressedSettings.cpp | 2 +- .../ContentAddressed/Pool/CasMountRuntime.h | 5 ++-- .../ContentAddressed/Pool/CasPool.cpp | 11 ++++----- .../ContentAddressed/Pool/CasPool.h | 24 +++++++++---------- .../ContentAddressed/Pool/CasServerRoot.cpp | 1 - .../ContentAddressed/Pool/CasServerRoot.h | 5 ++-- src/Disks/tests/gtest_cas_heartbeat.cpp | 2 +- src/Disks/tests/gtest_cas_pool.cpp | 12 +++++----- 12 files changed, 36 insertions(+), 46 deletions(-) diff --git a/docs/en/antalya/cas/configuration.md b/docs/en/antalya/cas/configuration.md index 3e19b981fe2e..1d37513a8f0a 100644 --- a/docs/en/antalya/cas/configuration.md +++ b/docs/en/antalya/cas/configuration.md @@ -108,7 +108,7 @@ entirely before release. Treat this table as a snapshot of the current build, no | `cas_gc_meta_pool_size` | `16` | Bounded pool size for GC per-hash freshness-meta writes | | `cas_gc_io_concurrency` | `16` | Bounded pool size for GC object-storage requests that run in parallel: the fold's read-ahead (checkpoints, ref logs, manifests, zero-candidate HEADs), the orphan-manifest sweep planning reads, the `SYSTEM CAS GC REBUILD` read-ahead, and the `pending_deletes` blob `HEAD` + conditional `DELETE` fan-out. Not covered: meta writes (`cas_gc_meta_pool_size`) and all other GC requests, which run on the round thread. `1` runs the covered requests sequentially. `cas_gc_read_concurrency` is rejected without an alias; use `cas_gc_io_concurrency` instead | | `cas_attempt_timeout_ms` | `5000` | Budget for one HTTP attempt of a writable Native mount's control-plane requests (read, head, list, remove, conditional write), at least 1. Together with the connect cap it forms the attempt envelope (`cas_attempt_timeout_ms + 2 × cap`; the cap is `cas_attempt_timeout_ms` itself when the disk's `connect_timeout_ms` is `0`, else `min(connect_timeout_ms, cas_attempt_timeout_ms)`) that the lease arithmetic reserves: one TCP connect and one TLS handshake under the cap each, send/receive bounded per socket operation by `cas_attempt_timeout_ms`. With background renewal the cadence check requires `cas_mount_renew_period_ms + 2 × envelope + cas_lease_safety_margin_ms < cas_mount_lease_ttl_ms`, which puts an effective ceiling on the frozen connect cap: under the defaults (TTL 30000, period 10000, margin 2000) the envelope must stay under 9000, so a disk `connect_timeout_ms` of 2000 ms or more refuses to open writable — lower the connect timeout or raise the TTL if you hit this | -| `cas_lease_safety_margin_ms` | `2000` | Startup-only margin validated against the mount lease TTL: the attempt envelope + `cas_lease_safety_margin_ms` must be strictly less than the mount lease TTL, and `cas_mount_renew_period_ms` + 2 × envelope + `cas_lease_safety_margin_ms` too, or the disk refuses to open writable | +| `cas_lease_safety_margin_ms` | `2000` | Room kept between a request and the mount lease deadline: a write is admitted only while the requests it may send plus this margin fit in the remaining lease. Validated against the mount lease TTL when a writable mount opens: the attempt envelope + `cas_lease_safety_margin_ms` must be strictly less than the mount lease TTL, and `cas_mount_renew_period_ms` + 2 × envelope + `cas_lease_safety_margin_ms` too, or the disk refuses to open writable | | `cas_unsafe_remount_no_delay` | `0` | Reclaim a mount slot that carries this server's own uuid at once after a hard restart, without observing the slot's token for the lease TTL. Unsafe whenever two processes can hold the same `server_uuid` (a copied uuid file, a stalled predecessor). After such a reclaim the predecessor can still start conditional writes until its own cutoff (`confirmed deadline − cas_lease_safety_margin_ms − 2 × envelope`) or until its next renewal meets the token guard, and a request it already sent may still materialize later. That is not a data hazard: ref-log keys carry `(writer_epoch, sequence)` and creates are conditional, so two writers can never commit different bodies to one key, and recovery's epoch seal settles any straggler (recovery fails closed after 64 successive seal-create attempts displaced by newly materializing old-epoch transactions). The exposure is availability, not data. Intended for test stands and deployments that guarantee one process per uuid | | `cas_staging_backend` | `local` | Blob staging backend (`local` \| `s3`); `s3` is opt-in and requires native same-store copy on writable mount | diff --git a/src/Common/ProfileEvents.cpp b/src/Common/ProfileEvents.cpp index a2a6b40d7c1a..71f905b1a297 100644 --- a/src/Common/ProfileEvents.cpp +++ b/src/Common/ProfileEvents.cpp @@ -956,7 +956,7 @@ The server successfully detected this situation and will download merged part fr M(CASMountRenewalRetries, "Number of physical conditional renewal PUTs sent after the first attempt of one logical CAS mount-lease renewal.", ValueType::Number) \ M(CASMountRenewalResolved, "Number of CAS mount-lease renewals whose committed outcome was proved by an exact resolving GET.", ValueType::Number) \ M(CASMountRenewalRecovered, "Number of logical CAS mount-lease renewals that committed after a physical retry or exact resolving GET.", ValueType::Number) \ - M(CASMountLeaseLost, "Counts exactly once per operational CAS mount-lease Live-to-TransientNotLive loss/recovery generation. The initiating external loss or the first ordinary terminal renewal consumer owns the increment, including external lease-safety deadline exhaustion; parked/classification/shutdown paths do not duplicate it.", ValueType::Number) \ + M(CASMountLeaseLost, "Number of Live-to-TransientNotLive transitions of a CAS mount lease, one per loss. A terminal renewal and an interference report count one, as does any other trip of the fence while no terminal intent is published. A renewal ended by a pending remount request, a stop or a FORGET does not count one.", ValueType::Number) \ M(CASMountLeaseExpired, "Number of times a renewal restored a CAS mount lease that had expired. While the lease is expired this server refuses writes and system.cas_mounts shows lifecycle_reason = 'lease_expired'; the watermark_renew event of the restoring renewal carries expired_ms.", ValueType::Number) \ M(CASRemountAttempts, "Number of invocations of the CAS whole-chain remount attempt.", ValueType::Number) \ M(CASRemountSucceeded, "Number of CAS whole-chain remount attempts that restored Live under a fresh writer epoch.", ValueType::Number) \ diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.h index 59dc55366c32..06d8898a95f2 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Backend/CasRequestBudget.h @@ -20,9 +20,10 @@ struct CasRequestBudget /// every attempt it starts; the actual socket-level wait is configured on the object storage's /// client (the object storage backend's single-attempt client), not by this struct. uint64_t attempt_timeout_ms = 5000; - /// Startup-only margin folded into `validateCasRequestBudget`'s inequality against the mount lease - /// TTL. Not consulted at runtime by the engine itself -- the caller's fence (backed by the local - /// write fence's own deadline) is what actually gates lease-relative timing per attempt. + /// Room kept between the end of a request and the mount lease deadline. `CasMountRuntime::admit` + /// refuses a request unless its need plus this margin fits in the remaining lease, and the farewell's + /// lease bound stops short of the deadline by it. `validateCasRequestBudget` checks it against the + /// lease TTL when a writable mount opens. uint64_t lease_safety_margin_ms = kDefaultLeaseSafetyMarginMs; /// The cap the single-attempt client puts on one TCP connect and again on one TLS handshake, @@ -62,13 +63,6 @@ struct CasRequestBudget /// and, when `background_renewal` is true (the mount runs a background renewer): /// mount_renew_period_ms + 2 × attemptEnvelopeMs() + lease_safety_margin_ms < mount_lease_ttl_ms /// (a renewal is a write: two envelopes for the attempt and its settlement read). -/// -/// A successor mounting over an unclean predecessor waits at least one lease TTL, plus its -/// materialization grace period, before trusting recovery listings. This is long enough for any -/// conditional PUT still in flight at the predecessor to either land or be abandoned by its own -/// exhausted retry budget. The predecessor's budget is constrained by -/// `attemptEnvelopeMs() + lease_safety_margin_ms < mount_lease_ttl_ms`, so no additional handover -/// check is needed here. void validateCasRequestBudget(const CasRequestBudget & budget, uint64_t mount_lease_ttl_ms, uint64_t mount_renew_period_ms, bool background_renewal); diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.h index be5d5299e504..cba947a00bc1 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.h @@ -654,7 +654,7 @@ class ContentAddressedMetadataStorage final : public IMetadataStorage, public IC /// `Cas::PoolConfig::cas_request_budget.attempt_timeout_ms` and the backend's own /// `attemptTimeoutMs()`. const uint64_t cas_attempt_timeout_ms; - /// Startup-only margin validated against the mount lease TTL; feeds + /// Room kept between a request and the mount lease deadline; feeds /// `Cas::PoolConfig::cas_request_budget.lease_safety_margin_ms`. const uint64_t cas_lease_safety_margin_ms; /// Configured staging backend; `Local` preserves the existing write path. diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.cpp index 9e6883d8c775..eed81bee1aca 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.cpp @@ -83,7 +83,7 @@ constexpr std::string_view CAS_KEY_PREFIX = "cas_"; DECLARE(UInt64, gc_io_concurrency, 16, "Maximum number of threads in the GC I/O pool. Used for fold and rebuild read-ahead, orphan-manifest sweep planning reads, and pending_deletes HEAD plus conditional DELETE. Per-hash meta writes use gc_meta_pool_size; other GC requests run on the round thread. 1 disables parallel GC I/O", 0) \ DECLARE(UInt64, gc_bulk_delete_chunk_keys, 1000, "Keys per batch delete request in GC's write-once families (owner-removed manifest bodies, covered ref logs and snapshots); 1 to 1000", 0) \ DECLARE(UInt64, attempt_timeout_ms, 5000, "Budget for one HTTP attempt of a writable Native mount's control-plane requests (read, head, list, remove, conditional write), at least 1. With the connect cap it forms the attempt envelope the lease arithmetic reserves", 0) \ - DECLARE(UInt64, lease_safety_margin_ms, Cas::kDefaultLeaseSafetyMarginMs, "Startup-only margin validated against the mount lease TTL: attempt envelope + this must be strictly less than the TTL, and renew period + 2 × envelope + this too", 0) \ + DECLARE(UInt64, lease_safety_margin_ms, Cas::kDefaultLeaseSafetyMarginMs, "Room kept between a request and the mount lease deadline: a write is admitted only while the requests it may send plus this margin fit in the remaining lease. Validated against the mount lease TTL when a writable mount opens: attempt envelope + this must be strictly less than the TTL, and renew period + 2 × envelope + this too", 0) \ DECLARE(String, staging_backend, "local", "Blob staging backend (local | s3); s3 is opt-in", 0) \ DECLARE_SETTINGS_TRAITS(ContentAddressedSettingsTraits, LIST_OF_CONTENT_ADDRESSED_SETTINGS, CONTENT_ADDRESSED_SETTINGS_SUPPORTED_TYPES) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index 1a1b48ee2e3c..9e7d7e8b1344 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -117,7 +117,7 @@ struct MountFence /// Owns the live writer-incarnation mechanics shared by the pool's mount and recovery orchestration: /// the `MountLeaseRenewer`, local `MountFence`, build watermark and in-flight build registry, -/// `live_writer_epoch`, unclean-boundary marker, and the lease thread. `Pool` retains the higher-level +/// `live_writer_epoch`, and the lease thread. `Pool` retains the higher-level /// claim/recovery sequence and its `remount_mutex`; in particular, the runtime does not acquire or own /// the ref-ledger locks. The runtime receives its backend, layout, configuration, event sink, request /// budget, and a callback that performs one pool-level remount attempt, so it has no `Pool` back-reference. @@ -326,7 +326,7 @@ class CasMountRuntime /// `driver_mutex`. void noteRenewRequest(const MountRenewRequestEvent & event) noexcept; - /// Writes one `WARNING` per lease expiry, at the first request event after it. Lease thread only. + /// Writes one `WARNING` per lease expiry, at the first request event after it. Called from `noteRenewRequest`. void warnOnceIfLeaseExpired(const MountRenewRequestEvent & event) noexcept; /// TRUE once the pool has reached — or is being driven toward — a state on which the lease thread @@ -410,7 +410,6 @@ class CasMountRuntime void finishTeardown(bool drained); /// Sleep through the injected test hook when present; otherwise use the production thread sleep. - /// `Pool` claim observation and materialization grace waits share this seam so tests control both. void waitSleep(uint64_t ms) const; /// Swap the wait hook after construction -- a test that must change what a wait DOES partway /// through a scenario (e.g. driving a second incarnation's renewal from inside the observed diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp index 3a5fee93ac65..433911199e7b 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp @@ -659,10 +659,9 @@ void Pool::mountWritable(PoolPtr & store, UInt128 our_uuid, MountClaimPolicy pol /// observation window. Derived from existing config — no new knob. const uint64_t poll_interval_ms = std::max( 1, static_cast(store->config.mount_renew_period.count()) / 2); - /// routes through `mount_runtime.waitSleep` (which itself routes through - /// `config.wait_sleep_fn` when a test injected one) rather than a bare `sleep_for` directly, so - /// a test intercepting `wait_sleep_fn` observes every wait `open` can block on -- and since the - /// post-reclaim materialization grace was retired, this observation poll is the only one left. + /// Routes through `mount_runtime.waitSleep` (which itself routes through `config.wait_sleep_fn` when + /// a test injected one) rather than a bare `sleep_for`, so a test intercepting `wait_sleep_fn` + /// observes every wait `open` can block on. const auto sleep_ms = [s = store.get()](uint64_t ms) { s->mount_runtime.waitSleep(ms); }; /// Operator-visible log the moment startup decides to watch a stale-looking self-mount (the /// disk-open path blocks up to ~threshold_ms here, so a silent block would be confusing). May @@ -1151,8 +1150,8 @@ void Pool::forgetDisk(const std::function & stop_and_join_gc, const Stri mount_runtime.publishVanishedIntent(); /// (2) Trip the local fence — the deliberate decommission act (allowed on a live disk). No durable- - /// effect write admits past this point (the fence-generation gate), and a live pool moves to - /// `TransientNotLive`, so store-class access already fails loud during the teardown window below. + /// effect write admits past this point (the fence-generation gate). The published intent keeps the + /// trip from counting a lease loss, so a live pool stays `Live` until step (6). mount_runtime.tripMountLost(); /// (3+4) Stop the GC scheduler (clears its leadership and JOINS its worker + heartbeat threads) BEFORE diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h index 45048640540a..d60113225384 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h @@ -216,10 +216,11 @@ struct PoolConfig std::function gc_redelete_apply_hook_for_test = {}; std::optional gc_io_pool_refuse_at_for_test = std::nullopt; - /// Mount-lease TTL: how long a freshly-renewed mount lease is valid. The local - /// write fence's monotonic deadline is `renew_time + this`, so a superseded/paused writer is fenced - /// once `this` elapses with no successful renew. The background renewer runs every - /// `mount_renew_period` (default ttl/3) so a healthy mount renews well before expiry. + /// Mount-lease TTL: how long a claim or renewal keeps the lease valid. The local write fence's + /// deadline is the start of the last committed claim or renewal attempt plus this, so a superseded + /// or paused writer is fenced once `this` elapses with no successful renewal. The background renewer + /// starts a renewal every `mount_renew_period` (default ttl/3) so a healthy mount renews well before + /// expiry. std::chrono::milliseconds mount_lease_ttl_ms{30000}; std::chrono::milliseconds mount_renew_period{10000}; /// = ttl/3 by default @@ -1259,20 +1260,17 @@ class Pool : public std::enable_shared_from_this /// The mount / write-fence / build-watermark / self-remount runtime, extracted /// from Pool. Owns the `MountLeaseRenewer`, the local `MountFence`, the per-server /// build watermark (`process_epoch` + the `builds_mutex`-guarded seq/registry) and its in-flight-build - /// map, the live-incarnation `live_writer_epoch`, the unclean-epoch high-water-mark, and the + /// map, the live-incarnation `live_writer_epoch`, and the /// lease thread (with one driver mutex/condition pair). Injected with backend/layout /// the `MountConfig` slice + `server_root_id` + the event-sink reference + the pool `cas_request_budget` /// + a `remount_attempt` callback (== `Pool::tryRemountOnce`, which STAYS on Pool: the claim/recovery /// ORCHESTRATION drives these owned primitives). /// - /// Declared AFTER `ref_ledger` -- preserving the pre-3.5 relative order VERBATIM (the mount raw-members - /// this component replaces all sat after `ref_ledger`), so `mount_runtime` is destroyed FIRST and - /// `ref_ledger` LAST. Both orders were proven equally safe -- `~Pool` - /// quiesces both subsystems before ANY member dtor runs (`stopBackgroundWorkers` -> - /// ref_ledger.drainRefLanesForShutdown -> mount_runtime.finishTeardown), and the ledger's async paths - /// pin `Pool::shared_from_this`, so no ledger->mount callback can fire during destruction in either - /// order. Both safe ⇒ this is a pure behavior-preserving relocation, so the ORIGINAL order is kept and - /// NO member-order change is introduced. Declared after event_sink_, pool_backend, config and + /// Declared AFTER `ref_ledger`, so `mount_runtime` is destroyed FIRST and `ref_ledger` LAST. Either + /// order is safe -- `~Pool` quiesces both subsystems before ANY member dtor runs + /// (`stopBackgroundWorkers` -> ref_ledger.drainRefLanesForShutdown -> mount_runtime.finishTeardown), and + /// the ledger's async paths pin `Pool::shared_from_this`, so no ledger->mount callback can fire during + /// destruction in either order. Declared after event_sink_, pool_backend, config and /// pool_layout so it is constructed after every dependency it is injected with. CasMountRuntime mount_runtime; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp index edb057ffce9c..e37c4ddf7da0 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp @@ -1281,7 +1281,6 @@ uint64_t MountLeaseRenewer::start(Liveness liveness) /// This decoded authoritative observation is the exact point at which this incarnation learns /// that a foreign successor owns the slot. Terminal teardown intentionally performs no release /// I/O, so account the skipped farewell here, once, before the renewer enters its terminal state. - /// The renewal may run under `remount_mutex`; keep the increment trace-free. ProfileEvents::incrementNoTrace(ProfileEvents::CASMountReleaseSkippedForeignOccupant); emitMountEvent( event_sink, CasEventType::MountConflict, srid, "foreign_writer", ¤t, diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h index 08eab63300c2..4554e76c03b7 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h @@ -47,8 +47,9 @@ enum class MountRenewOutcome : uint8_t Terminal, }; -/// Why a renewal ended without retaining the lease, in the words of its audit event. `renew` sets it in -/// the arm of the write's verdict that ended the renewal. +/// Why a renewal ended without committing, in the words of its audit event. `renew` sets it in the arm +/// of the write's verdict that ended the renewal. `Cancelled` is also set when nothing was attempted and +/// the lease is kept. enum class MountRenewTerminalClassification : uint8_t { Unclassified, diff --git a/src/Disks/tests/gtest_cas_heartbeat.cpp b/src/Disks/tests/gtest_cas_heartbeat.cpp index 5d02938b72dd..1db269e2476c 100644 --- a/src/Disks/tests/gtest_cas_heartbeat.cpp +++ b/src/Disks/tests/gtest_cas_heartbeat.cpp @@ -1728,7 +1728,7 @@ TEST(CASHeartbeat, DefinitiveAnswersStayTerminalPastTheDeadline) } f.backend->attempts.clear(); f.events.clear(); - /// Past the lease: a bounded renewal would send nothing here. + /// Past the lease: the renewal still sends, because it has no lease bound. f.boot_ms = f.anchor + UnboundedRenewalFixture::ttl_ms + 1'000; /// A definitive answer that was retried would never end this renewal; the bound makes it fail instead. const uint64_t live_until = f.boot_ms + 600'000; diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp index e9bdbaf4b30e..c88a5895d34c 100644 --- a/src/Disks/tests/gtest_cas_pool.cpp +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -2084,7 +2084,7 @@ class RuntimeUnderTest /// deadline against real boottime, finds it long past, and refuses every request unsent. farewell.setNowFnForTest([this] { return runtime.bootMsNow(); }); lease.setNowFnForTest([this] { return runtime.bootMsNow(); }); - /// As `Pool` wires it: a stop wakes the worker renewal's wait. + /// As `Pool` wires it: a stop wakes the lease thread's renewal wait. lease.setSleepFnForTest([this](uint64_t ms) { runtime.sleepInterruptibly(ms); }); } @@ -3994,7 +3994,7 @@ TEST(CASPoolRemount, ImmediatePostRemountRenewalFailureIsNotDropped) const uint64_t fresh_anchor = runtime_ptr->startRenewer(); EXPECT_TRUE(runtime_ptr->armIfAdmissible(fresh_anchor + 10'000)); boot_ms = 2'000; - /// A definitive answer for the fresh incarnation's first worker renewal; a transient + /// A definitive answer for the fresh incarnation's first renewal on the lease thread; a transient /// fault would only be retried. fenceOutMount(*backend, layout.mountKey("test")); first.arriveAndWait(); @@ -4028,8 +4028,8 @@ TEST(CASPoolRemount, ThrowingEventSinkAfterCommitLeavesRuntimeLive) .pool_prefix = "throwing-remount-event", .server_root_id = "test", .background_watermark = true}); /// Heap-owned, not a plain local declared after `store`: if `waitUntilArrived` below throws on its /// own internal timeout, unwinding would destroy a stack-local barrier before `store`'s destructor - /// joins the remount worker, and that worker can still be inside `arriveAndWait` on the dangling - /// reference. A `shared_ptr` capture keeps the barrier alive for as long as the worker needs it, + /// joins the lease thread, and that thread can still be inside `arriveAndWait` on the dangling + /// reference. A `shared_ptr` capture keeps the barrier alive for as long as the thread needs it, /// independent of declaration order. auto committed = std::make_shared(); store->setEventSink([committed](const CasEvent & event) @@ -4217,7 +4217,7 @@ TEST(CASPool, DeterministicWorkerFailureFencesWithoutWaitingForCadence) runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); - /// A definitive answer ends the worker's renewal; a transient fault would only be retried. + /// A definitive answer ends the lease thread's renewal; a transient fault would only be retried. fenceOutMount(*backend, layout.mountKey("test")); runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); remount_entered.waitUntilArrived(); @@ -4228,7 +4228,7 @@ TEST(CASPool, DeterministicWorkerFailureFencesWithoutWaitingForCadence) runtime.finishTeardown(false); } -/// FORGET while the worker's renewal retries past its lease: the intent then the trip, in the order +/// FORGET while the lease thread's renewal retries past its lease: the intent then the trip, in the order /// `Pool::forgetDisk` uses, end the renewal inside the wait it is in, the lease thread exits, and no remount /// generation is raised. TEST(CASMountRuntime, ForgetEndsAnUnboundedRenewal) From 7a3d9f69ec1191e190732babcd3d34c0e17de4c6 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 09:47:21 +0200 Subject: [PATCH 24/37] Make warnOnceIfLeaseExpired private Only `noteRenewRequest` calls it; no test does. Co-Authored-By: Claude Sonnet 5.5 --- .../MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index 9e7d7e8b1344..9325d89529c0 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -326,9 +326,6 @@ class CasMountRuntime /// `driver_mutex`. void noteRenewRequest(const MountRenewRequestEvent & event) noexcept; - /// Writes one `WARNING` per lease expiry, at the first request event after it. Called from `noteRenewRequest`. - void warnOnceIfLeaseExpired(const MountRenewRequestEvent & event) noexcept; - /// TRUE once the pool has reached — or is being driven toward — a state on which the lease thread /// must stop: a published terminal `Vanished` intent (`vanished_intent` — set early by /// FORGET, or by a natural `enterVanished`, and already subsuming every settled `Vanished*` state since @@ -453,6 +450,8 @@ class CasMountRuntime /// caused by the stop cannot be mistaken for one that preceded it. bool renewalCancelled() const; void tripFenceWithoutOperationalLoss(); + /// Writes one `WARNING` per lease expiry, at the first request event after it. Called from `noteRenewRequest`. + void warnOnceIfLeaseExpired(const MountRenewRequestEvent & event) noexcept; /// Publish `deadline_boot_ms` as a fresh lease incarnation and open the fence. With `report_live`, /// `Live` is published before the fence opens, so a reader that sees the fence armed reads `Live`. void armFence(uint64_t deadline_boot_ms, bool report_live); From 45aac314bb3ad0ae5c290ace2dcb3e7b8da771a0 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 09:50:58 +0200 Subject: [PATCH 25/37] Delete process_epoch and the epoch accessors nothing reads Identity uses the allocated writer epoch and incarnation checks use `liveWriterEpoch`. The 64 test lines that read `writerEpoch` read `liveWriterEpoch`; none runs on a read-only pool, where the value would change from a random number to 0. Co-Authored-By: Claude Sonnet 5.5 --- .../ContentAddressed/Pool/CasMountRuntime.cpp | 18 -------- .../ContentAddressed/Pool/CasMountRuntime.h | 28 +++---------- .../ContentAddressed/Pool/CasPartWriteTxn.h | 2 +- .../ContentAddressed/Pool/CasPool.cpp | 22 +--------- .../ContentAddressed/Pool/CasPool.h | 11 +---- .../ContentAddressed/Pool/CasRefLedger.h | 4 +- src/Disks/tests/gtest_cas_decommission.cpp | 8 ++-- src/Disks/tests/gtest_cas_event_log.cpp | 2 +- .../tests/gtest_cas_fence_generation.cpp | 2 +- src/Disks/tests/gtest_cas_mount.cpp | 8 ++-- src/Disks/tests/gtest_cas_part_write.cpp | 4 +- src/Disks/tests/gtest_cas_pool.cpp | 1 - .../gtest_cas_ref_catalog_birth_wiring.cpp | 8 ++-- .../tests/gtest_cas_ref_chunked_flush.cpp | 2 +- src/Disks/tests/gtest_cas_ref_ckpt.cpp | 20 ++++----- .../tests/gtest_cas_ref_contiguous_alloc.cpp | 22 +++++----- ...test_cas_ref_snapshot_publish_ordering.cpp | 42 +++++++++---------- src/Disks/tests/gtest_cas_ref_writer.cpp | 11 +++-- src/Disks/tests/gtest_cas_writer_duties.cpp | 8 ++-- 19 files changed, 81 insertions(+), 142 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index 0db29a68fda1..f9f25a6bbed4 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -4,7 +4,6 @@ #include #include #include -#include #include #include #include @@ -439,23 +438,6 @@ void CasMountRuntime::cancelInflightBuildsForNamespace(const RootNamespace & ns) build->cancelForNamespaceRemoval(ns); } -void CasMountRuntime::mintRandomProcessEpoch() -{ - /// Mint a nonzero equality-only identity. Keep it away from the zero/unarmed and UINT64_MAX/retired - /// sentinels; 52 random bits are sufficient for the expected collision risk of this token. - constexpr uint64_t EPOCH_MASK = (1ULL << 52) - 1; - process_epoch.store( - (thread_local_rng() ^ (static_cast(thread_local_rng()) << 32)) & EPOCH_MASK, - std::memory_order_relaxed); - if (process_epoch.load(std::memory_order_relaxed) == 0) - process_epoch.store(1, std::memory_order_relaxed); -} - -void CasMountRuntime::setProcessEpoch(uint64_t v, std::memory_order order) -{ - process_epoch.store(v, order); -} - void CasMountRuntime::setLiveWriterEpoch(uint64_t v) { live_writer_epoch.store(v, std::memory_order_release); diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index 9325d89529c0..47ae9d4da33c 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -141,11 +141,7 @@ class CasMountRuntime /// after construction, from the lease thread. std::function remount_attempt_); - /// ---- per-server watermark and identity ---- - /// `process_epoch` is random and nonzero for this pool incarnation. GC compares it for equality, - /// never ordering; a different value means that the previous writer incarnation is no longer live. - uint64_t epoch() const { return process_epoch.load(std::memory_order_acquire); } - uint64_t writerEpoch() const { return process_epoch.load(std::memory_order_acquire); } + /// ---- per-server watermark ---- /// The GC floor: the oldest in-flight build_seq, or next_build_seq when no build is active (so a /// quiescent server's watermark floor advances to the next-to-be-allocated seq). Locks builds_mutex. uint64_t minActive(); @@ -370,12 +366,7 @@ class CasMountRuntime /// cancellation may take a different path and must not run under the registry lock. void cancelInflightBuildsForNamespace(const RootNamespace & ns); - /// ---- process epoch (identity) ---- - /// Mint the random nonzero process identity used by GC's equality check. - void mintRandomProcessEpoch(); - /// Set `process_epoch` to the durable `writer_epoch`. The caller supplies the memory order because - /// the initial writable claim and a later self-remount have different publication requirements. - void setProcessEpoch(uint64_t v, std::memory_order order); + /// ---- live writer epoch ---- /// Publish the live-incarnation `live_writer_epoch` with release ordering. void setLiveWriterEpoch(uint64_t v); @@ -480,17 +471,10 @@ class CasMountRuntime CasRequestBudget cas_request_budget; std::function remount_attempt; - /// Per-server build watermark. `process_epoch` is a random - /// nonzero u64 minted once at open: GC checks it for EQUALITY (an object stamped with a different - /// epoch is from a dead incarnation), never for ordering. next_build_seq is a strictly-increasing - /// per-process counter (monotonicity is load-bearing — a seq is never reused or lowered); - /// active_build_seqs holds the seqs of in-flight builds, so `minActive` yields the GC floor. The floor - /// is published by the merged `mount_renewer` - /// beat (there is no standalone watermark object anymore). ATOMIC because a self-remount re-stamps it - /// (kept equal to `live_writer_epoch`) from the lease thread while `epoch`/`writerEpoch` - /// may observe it; the ref-lane hot readers were moved to `liveWriterEpoch`, so this now backs only - /// the identity accessors. - std::atomic process_epoch{0}; + /// Per-server build watermark. `next_build_seq` is a strictly-increasing per-process counter + /// (monotonicity is load-bearing: a seq is never reused or lowered); `active_build_seqs` holds the + /// seqs of in-flight builds, so `minActive` yields the GC floor. The floor is published by the + /// merged `mount_renewer` beat (there is no standalone watermark object). std::mutex builds_mutex; uint64_t next_build_seq = 1; std::set active_build_seqs; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPartWriteTxn.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPartWriteTxn.h index 953059a76ebc..d84ccdf427fe 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPartWriteTxn.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPartWriteTxn.h @@ -358,7 +358,7 @@ class PartWriteTxn uint64_t txn_generation{}; UInt128 build_id{}; uint64_t build_seq{}; /// per-process monotone sequence - uint64_t epoch{}; /// owning Pool's process_epoch + uint64_t epoch{}; /// the owning Pool's live writer epoch when the build began uint32_t next_manifest_ordinal = 1; /// per-build monotone manifest ordinal PartWriteInfo info; bool alive = true; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp index 433911199e7b..a5939bc31c0c 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp @@ -568,11 +568,6 @@ PoolPtr Pool::open(BackendPtr backend, PoolConfig config) /// durably admitted (the invariant this whole design rests on). chassert(store->isAlgoAdmitted(write_algo)); - /// Per-server watermark: mint the random NONZERO `process_epoch` - /// once per Pool (GC checks it for equality only -- a different epoch == a dead incarnation). The - /// masking/redraw detail lives in `CasMountRuntime::mintRandomProcessEpoch`. - store->mount_runtime.mintRandomProcessEpoch(); - /// W-ANCHOR: the per-server watermark must be durable BEFORE any object PUT. A read-only open /// must never mutate the pool (the probe is skipped above for the same reason), so the watermark /// — which rides inside the `gc/server-roots//mount` lease object — is only @@ -630,10 +625,7 @@ void Pool::mountWritable(PoolPtr & store, UInt128 our_uuid, MountClaimPolicy pol std::chrono::system_clock::now().time_since_epoch()).count()); }; - /// 3. Durable-monotone writer_epoch — CAS-bump the sticky `epoch` object. THE BRIDGE: this - /// durable value REPLACES the random `process_epoch` for identity, so the watermark + every - /// manifest ref carries it (the random mint above stays for the read-only - /// path, which never reaches here). The epoch-aware sweep reads this value. + /// 3. Durable-monotone writer_epoch — CAS-bump the sticky `epoch` object. /// Mutable: a GC fence of our fresh lease during open (expiry mid-open racing a GC round) is /// recoverable — a fence costs an epoch, so the fence-recovery loop below re-allocates a fresh /// writer_epoch and re-claims (the TLA+-checked `NoPermanentWedge` invariant). @@ -647,7 +639,6 @@ void Pool::mountWritable(PoolPtr & store, UInt128 our_uuid, MountClaimPolicy pol CasOperation epoch_op = store->gc_requests.admit(); uint64_t writer_epoch = allocateWriterEpoch( epoch_op, store->pool_layout, srid, epoch_policy, now_ms(), observe_catalog); - store->mount_runtime.setProcessEpoch(writer_epoch, std::memory_order_relaxed); /// 4. Mount lease — LIVENESS. Decide over the current mount object using the wall-clock `now_ms` /// hoisted above. @@ -762,7 +753,6 @@ void Pool::mountWritable(PoolPtr & store, UInt128 our_uuid, MountClaimPolicy pol CasOperation reallocate_op = store->gc_requests.admit(); writer_epoch = allocateWriterEpoch( reallocate_op, store->pool_layout, srid, epoch_policy, now_ms(), observe_catalog); - store->mount_runtime.setProcessEpoch(writer_epoch, std::memory_order_relaxed); continue; } if (claim.kind != MountClaimResult::Claimed) @@ -809,7 +799,6 @@ void Pool::mountWritable(PoolPtr & store, UInt128 our_uuid, MountClaimPolicy pol CasOperation refenced_epoch_op = store->gc_requests.admit(); writer_epoch = allocateWriterEpoch( refenced_epoch_op, store->pool_layout, srid, epoch_policy, now_ms(), observe_catalog); - store->mount_runtime.setProcessEpoch(writer_epoch, std::memory_order_relaxed); continue; } break; @@ -962,11 +951,6 @@ PoolPtr Pool::openForDecommission(BackendPtr backend, PoolConfig config, const S /// Register-before-first-write, belt-and-braces: same invariant `open` asserts. chassert(store->isAlgoAdmitted(write_algo)); - /// No random `process_epoch` mint here: `open` pays that prologue because its read-only path - /// never reaches `mountWritable` and so needs SOME nonzero epoch, but this factory is - /// writer-only -- `mountWritable` below unconditionally overwrites `process_epoch` with the - /// freshly allocated durable `writer_epoch` before anything could observe the zero-initialized - /// default. mountWritable(store, *victim_uuid, MountClaimPolicy::NoWait); return store; } @@ -1487,11 +1471,9 @@ bool Pool::tryRemountOnce() /// swap. /// 1. Bump the live epoch so every subsequent `allocateRefTxnId` sorts strictly above any older /// (dead-incarnation or twin) durable log. Do this BEFORE `armMountFence` so there is no window - /// where the gate is open while the epoch is still stale. Keep `process_epoch` (the identity - /// accessors) equal to it. + /// where the gate is open while the epoch is still stale. step = "publish_writer_epoch"; mount_runtime.setLiveWriterEpoch(writer_epoch); - mount_runtime.setProcessEpoch(writer_epoch, std::memory_order_release); /// 2. CANCEL OR JOIN every in-flight ref-table recovery, and BLOCK here until none is left (spec /// §3: "self-remount cancels or waits out recovery before rearming"). A recovery admitted under /// the outgoing incarnation WRITES -- its seal CAS-walk mints epoch seals and advances the diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h index d60113225384..fb3e9641167a 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h @@ -455,14 +455,6 @@ class Pool : public std::enable_shared_from_this void setDetachedDrainDeadlineBudgetForTest(const CasRequestBudget & budget); /// ---- per-server watermark surface ---- - /// process_epoch: random nonzero per Pool (process). GC checks epoch EQUALITY, never ordering. - uint64_t epoch() const { return mount_runtime.epoch(); } - /// The durable-monotone writer_epoch allocated at writable open. On a - /// writable Pool this is the value bridged into `process_epoch` (so the watermark + the manifest - /// manifest ref carries it); on a read-only open the random `process_epoch` is unchanged and - /// no durable epoch is allocated. A self-remount re-establishes this to the fresh incarnation's - /// writer_epoch (kept equal to `liveWriterEpoch`). The epoch-aware sweep reads this value. - uint64_t writerEpoch() const { return mount_runtime.writerEpoch(); } /// The GC floor: the oldest in-flight build_seq, or next_build_seq when no build is active (so a /// quiescent server's watermark floor advances to the next-to-be-allocated seq). Locks builds_mutex. uint64_t minActive(); @@ -807,6 +799,7 @@ class Pool : public std::enable_shared_from_this /// The writer_epoch of the LIVE mount incarnation. Bumped by `tryRemountOnce` (self-remount /// after a GC fence-out) — a `PartWriteTxn` minted under an older epoch fails closed on its next step. + /// Zero on a read-only pool, which claims no mount. uint64_t liveWriterEpoch() const { return mount_runtime.liveWriterEpoch(); } /// Test seam: publish a new live-incarnation writer epoch WITHOUT running a self-remount -- the @@ -1259,7 +1252,7 @@ class Pool : public std::enable_shared_from_this CasRefLedger ref_ledger; /// The mount / write-fence / build-watermark / self-remount runtime, extracted /// from Pool. Owns the `MountLeaseRenewer`, the local `MountFence`, the per-server - /// build watermark (`process_epoch` + the `builds_mutex`-guarded seq/registry) and its in-flight-build + /// build watermark (the `builds_mutex`-guarded seq/registry) and its in-flight-build /// map, the live-incarnation `live_writer_epoch`, and the /// lease thread (with one driver mutex/condition pair). Injected with backend/layout /// the `MountConfig` slice + `server_root_id` + the event-sink reference + the pool `cas_request_budget` diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefLedger.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefLedger.h index 0f53238dfdd3..2414a621216e 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefLedger.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefLedger.h @@ -982,8 +982,8 @@ class CasRefLedger /// A post-durable install failure moves the lane to `NeedsRecovery`, so this function is not called /// again until replay has installed that durable transaction and advanced `greatest_applied`. /// - /// The epoch component is the live mount incarnation's writer epoch, not the open-time - /// `process_epoch`: a self-remount allocates a strictly-greater durable writer_epoch, so every ref + /// The epoch component is the live mount incarnation's writer epoch: a self-remount allocates a + /// strictly-greater durable writer_epoch, so every ref /// transaction stamped after the remount sorts strictly ABOVE any (dead-incarnation or twin) log /// still durable under an older epoch. `RefTxnId` compares epoch first, so the epoch bump alone /// guarantees that a new log is never inserted at or below an already durable table log id. diff --git a/src/Disks/tests/gtest_cas_decommission.cpp b/src/Disks/tests/gtest_cas_decommission.cpp index d590eaed7a59..b662cf20a48f 100644 --- a/src/Disks/tests/gtest_cas_decommission.cpp +++ b/src/Disks/tests/gtest_cas_decommission.cpp @@ -473,7 +473,7 @@ void makeTableWithRefs(Pool & victim, const String & ns_str, uint64_t committed, ManifestId seedOrphanManifestBody(Pool & victim, const String & ns_str) { const RootNamespace ns(ns_str); - const ManifestRef ref{.writer_epoch = victim.writerEpoch(), .build_sequence = 99, .manifest_ordinal = 1}; + const ManifestRef ref{.writer_epoch = victim.liveWriterEpoch(), .build_sequence = 99, .manifest_ordinal = 1}; const ManifestId id = writeManifestRaw(*victim.poolBackendPtr(), victim.layout(), ns, ref, {}); /// EXPECT, not ASSERT: this function returns a value now, and ASSERT_* expands to a bare `return;` /// -- invalid in a non-void function. @@ -511,12 +511,12 @@ TEST(CASDecommission, ClaimsDeadMemberAndBumpsEpoch) uint64_t victim_epoch = 0; { auto victim = openVictim(backend); - victim_epoch = victim->writerEpoch(); + victim_epoch = victim->liveWriterEpoch(); } /// graceful close: lease stamped already-expired + farewell — the slot is claimable auto admin = Pool::openForDecommission(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "admin"}, "victim"); ASSERT_TRUE(admin != nullptr); - EXPECT_GT(admin->writerEpoch(), victim_epoch); + EXPECT_GT(admin->liveWriterEpoch(), victim_epoch); /// The admin store IS the victim server root now (impersonation). EXPECT_EQ(admin->poolConfig().server_root_id, "victim"); } @@ -795,7 +795,7 @@ TEST(CASDecommission, CountsRealisticEpochPrecommit) uint64_t victim_epoch = 0; { auto victim = openVictim(backend); - victim_epoch = victim->writerEpoch(); + victim_epoch = victim->liveWriterEpoch(); makeTableWithRefs(*victim, "victim/db/t1", /*committed=*/1, /*precommits=*/0); const RootNamespace ns("victim/db/t1"); diff --git a/src/Disks/tests/gtest_cas_event_log.cpp b/src/Disks/tests/gtest_cas_event_log.cpp index 59041135989a..153839908b85 100644 --- a/src/Disks/tests/gtest_cas_event_log.cpp +++ b/src/Disks/tests/gtest_cas_event_log.cpp @@ -245,7 +245,7 @@ TEST(CASEvent, WatermarkRenewEventsAreBoundedAndComplete) EXPECT_EQ(renewals[0].detail.at("attempts_sent"), "2"); EXPECT_EQ(renewals[0].detail.at("classification"), "committed_after_retry"); EXPECT_EQ(renewals[0].detail.at("server_root_id"), "test"); - EXPECT_EQ(renewals[0].detail.at("writer_epoch"), std::to_string(store->writerEpoch())); + EXPECT_EQ(renewals[0].detail.at("writer_epoch"), std::to_string(store->liveWriterEpoch())); EXPECT_EQ(renewals[0].detail.at("seq"), "2"); EXPECT_FALSE(renewals[0].detail.at("write_attempt_id").empty()); EXPECT_LT(renewals[0].detail.at("write_attempt_id").size(), 32u); diff --git a/src/Disks/tests/gtest_cas_fence_generation.cpp b/src/Disks/tests/gtest_cas_fence_generation.cpp index f4f060d576f9..ac726ec89e12 100644 --- a/src/Disks/tests/gtest_cas_fence_generation.cpp +++ b/src/Disks/tests/gtest_cas_fence_generation.cpp @@ -216,7 +216,7 @@ TEST(CASFenceGeneration, RearmPublishesTheNewGenerationBeforeOpeningTheFence) << "no runtime may be published in the re-arm interposition"; }); - store->armMountFence(DB::UInt128{0, 1}, store->writerEpoch(), store->bootMsNow() + 600000); + store->armMountFence(DB::UInt128{0, 1}, store->liveWriterEpoch(), store->bootMsNow() + 600000); store->setArmMountFenceInterpositionHookForTest(nullptr); EXPECT_FALSE(admitted_in_interposition); diff --git a/src/Disks/tests/gtest_cas_mount.cpp b/src/Disks/tests/gtest_cas_mount.cpp index 8182a61d8204..bff498b8f0a8 100644 --- a/src/Disks/tests/gtest_cas_mount.cpp +++ b/src/Disks/tests/gtest_cas_mount.cpp @@ -1204,7 +1204,7 @@ TEST(CASMountStartup, WriterEpochStrictlyIncreasesAcrossReopen) auto b = std::make_shared(); auto s1 = Pool::open(b, PoolConfig{ .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "r"}); - const uint64_t e1 = s1->writerEpoch(); + const uint64_t e1 = s1->liveWriterEpoch(); /// Simulate shutdown: the Pool dtor stops the renewer, whose terminate() retires the lease /// (stamps it already-expired). The owner + the durable epoch object stay sticky. @@ -1214,7 +1214,7 @@ TEST(CASMountStartup, WriterEpochStrictlyIncreasesAcrossReopen) /// higher durable writer_epoch. auto s2 = Pool::open(b, PoolConfig{ .pool_prefix = "p", .server_id = UInt128(1), .server_root_id = "r"}); - const uint64_t e2 = s2->writerEpoch(); + const uint64_t e2 = s2->liveWriterEpoch(); EXPECT_GT(e2, e1); } @@ -1400,7 +1400,7 @@ TEST(CASMountStartup, StaleSelfMountReclaimedAfterWait) .mount_renew_period = std::chrono::milliseconds(100), .cas_request_budget = tiny_budget}); ASSERT_NE(a, nullptr); - const uint64_t e1 = a->writerEpoch(); + const uint64_t e1 = a->liveWriterEpoch(); const String mount_key = a->layout().mountKey("r"); Ops ops(b); const auto stale_mount = ops.op.read(mount_key, Retry::standard()); @@ -1441,7 +1441,7 @@ TEST(CASMountStartup, StaleSelfMountReclaimedAfterWait) *a2_fake_boot += ms; }})); ASSERT_NE(a2, nullptr); - EXPECT_GT(a2->writerEpoch(), e1); + EXPECT_GT(a2->liveWriterEpoch(), e1); /// The original live-object overlap: a first Pool is still alive when a replacement reclaims its /// slot, so the first one's release meets a stranger. This was an `EXPECT_DEATH` pinning a diff --git a/src/Disks/tests/gtest_cas_part_write.cpp b/src/Disks/tests/gtest_cas_part_write.cpp index 7259f13360ca..963e166f104e 100644 --- a/src/Disks/tests/gtest_cas_part_write.cpp +++ b/src/Disks/tests/gtest_cas_part_write.cpp @@ -418,14 +418,14 @@ TEST(CASPartWriteTxn, StageManifestUsesPerBuildOrdinals) const ManifestId first = build->stageManifest({blobManifestEntry("a.bin", "a")}); const ManifestId second = build->stageManifest({blobManifestEntry("b.bin", "b")}); - EXPECT_EQ(first.ref.writer_epoch, s->writerEpoch()); + EXPECT_EQ(first.ref.writer_epoch, s->liveWriterEpoch()); EXPECT_EQ(first.ref.build_sequence, build->buildSeq()); EXPECT_EQ(first.ref.manifest_ordinal, 1u); EXPECT_EQ(second.ref.writer_epoch, first.ref.writer_epoch); EXPECT_EQ(second.ref.build_sequence, first.ref.build_sequence); EXPECT_EQ(second.ref.manifest_ordinal, 2u); /// Canonical hex build directory (spec §Manifest Identifier): `-/`. - const String build_segment = renderRefTxnId(RefTxnId{s->writerEpoch(), build->buildSeq()}); + const String build_segment = renderRefTxnId(RefTxnId{s->liveWriterEpoch(), build->buildSeq()}); EXPECT_EQ(s->layout().manifestKey(first), "p/cas/manifests/test/tbl@cas@/" + build_segment + "/000001.zst"); EXPECT_EQ(s->layout().manifestKey(second), "p/cas/manifests/test/tbl@cas@/" + build_segment + "/000002.zst"); diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp index c88a5895d34c..70be8e9bc275 100644 --- a/src/Disks/tests/gtest_cas_pool.cpp +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -3891,7 +3891,6 @@ TEST(CASPoolRemount, ExternalLossDuringRenewalUsesOneRecoveryGeneration) return false; runtime_ptr->installRenewer(uuid, 2, [&] { return wall_ms; }); const uint64_t fresh_anchor = runtime_ptr->startRenewer(); - runtime_ptr->setProcessEpoch(2, std::memory_order_release); runtime_ptr->setLiveWriterEpoch(2); EXPECT_TRUE(runtime_ptr->armIfAdmissible(fresh_anchor + 1000)); remount_barrier.arriveAndWait(); diff --git a/src/Disks/tests/gtest_cas_ref_catalog_birth_wiring.cpp b/src/Disks/tests/gtest_cas_ref_catalog_birth_wiring.cpp index 2c094fa122af..d06bbbdd2e8e 100644 --- a/src/Disks/tests/gtest_cas_ref_catalog_birth_wiring.cpp +++ b/src/Disks/tests/gtest_cas_ref_catalog_birth_wiring.cpp @@ -178,7 +178,7 @@ TEST(CASRefCatalogBirthWiring, FirstOpenMintsALiveCatalogEntryAndKeysTheBirthAtI const RootNamespace ns{"srv1/birth_wiring"}; const RefTxnId id = publishBirth(store, ns, "a"); - EXPECT_EQ(id, (RefTxnId{store->writerEpoch(), 1})); + EXPECT_EQ(id, (RefTxnId{store->liveWriterEpoch(), 1})); const CasRefCatalog::Snapshot snap = CasRefCatalog::read(op, store->layout()); const CatalogEntry * entry = findEntry(snap.catalog, ns); @@ -346,14 +346,14 @@ TEST(CASRefCatalogBirthWiring, AnExistingLiveEntryIsAdoptedRatherThanReminted) .creator = std::nullopt}; CasRefCatalog::casAdmitEntry(op, layout, 1, entry); DB::Cas::tests::writeRecoverableCkptForRawFixture(*backend, layout, ns, RefCkpt{ - .life_epoch = store->writerEpoch(), + .life_epoch = store->liveWriterEpoch(), .committed_through = std::nullopt, .checkpoint_snapshot_id = std::nullopt, .last_epoch_seal = std::nullopt, }); const RefTxnId id = publishBirth(store, ns, "a"); - EXPECT_EQ(id, (RefTxnId{store->writerEpoch(), 1})); + EXPECT_EQ(id, (RefTxnId{store->liveWriterEpoch(), 1})); /// The read result must outlive the returned pointer -- findEntry points into its entries. const auto after_cut = CasRefCatalog::read(op, layout); @@ -468,7 +468,7 @@ TEST(CASRefCatalogBirthWiring, AStaleCreatingEntryFromATerminatedForeignFenceIsR /// The production path resumes creation itself: reconciles the stale entry onto THIS mount's own /// fence and completes it to `Live`, over the SAME incarnation the dead creator minted. const RefTxnId id = publishBirth(store, ns, "a"); - EXPECT_EQ(id, (RefTxnId{store->writerEpoch(), 1})); + EXPECT_EQ(id, (RefTxnId{store->liveWriterEpoch(), 1})); /// The read result must outlive the returned pointer -- findEntry points into its entries. const auto live_cut = CasRefCatalog::read(op, layout); diff --git a/src/Disks/tests/gtest_cas_ref_chunked_flush.cpp b/src/Disks/tests/gtest_cas_ref_chunked_flush.cpp index a79f58fdf424..17c41bc7c956 100644 --- a/src/Disks/tests/gtest_cas_ref_chunked_flush.cpp +++ b/src/Disks/tests/gtest_cas_ref_chunked_flush.cpp @@ -406,7 +406,7 @@ TEST(CASRefWriterChunkedFlush, DropNamespaceOverOpCapSucceeds) /// fixture directly to `backend` after open, but before `ns` is ever touched, is observed identically /// to writing it before open. auto store = openPool(backend); - const uint64_t epoch = store->writerEpoch(); + const uint64_t epoch = store->liveWriterEpoch(); /// Stage B (Task 4-C): pin `ns` to the sentinel now, before the raw snapshot below -- `listRefs`/ /// `dropNamespace` further down are real production reads that trigger `resolveNamespaceLife`, /// which for an UNADMITTED namespace mints a fresh RANDOM incarnation rather than adopting the diff --git a/src/Disks/tests/gtest_cas_ref_ckpt.cpp b/src/Disks/tests/gtest_cas_ref_ckpt.cpp index f3f7e08efbd7..5d74d1f5a9a9 100644 --- a/src/Disks/tests/gtest_cas_ref_ckpt.cpp +++ b/src/Disks/tests/gtest_cas_ref_ckpt.cpp @@ -1102,12 +1102,12 @@ TEST(CASRefCheckpoint, NamespaceBirthCreatesTheCheckpointCarryingItsLifeEpoch) /// test cannot yet compute. EXPECT_TRUE(CasRefCatalog::read(op, store->layout()).catalog.entries.empty()) << "nothing exists before the birth"; - ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->writerEpoch(), 1})); + ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->liveWriterEpoch(), 1})); const NamespaceLifeId life = liveLifeOrFail(op, store->layout(), ns); const auto sample = readCkpt(op, store->layout(), life); ASSERT_TRUE(sample.has_value()) << "spec §3 creates the _ckpt before the namespace becomes Live"; - EXPECT_EQ(sample->ckpt.life_epoch, store->writerEpoch()); + EXPECT_EQ(sample->ckpt.life_epoch, store->liveWriterEpoch()); EXPECT_FALSE(sample->ckpt.checkpoint_snapshot_id.has_value()) << "a newborn namespace has no base yet"; EXPECT_FALSE(sample->ckpt.last_epoch_seal.has_value()); } @@ -1119,7 +1119,7 @@ TEST(CASRefCheckpoint, ACommittedSnapshotPublishAdvancesTheCheckpoint) auto store = openPool(backend); CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); CasOperation op = requests.admit(); - const uint64_t epoch = store->writerEpoch(); + const uint64_t epoch = store->liveWriterEpoch(); const RootNamespace ns{"srv1/ckpt_publish"}; ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{epoch, 1})); @@ -1149,7 +1149,7 @@ TEST(CASRefCheckpoint, CleanupPlannedBetweenTheBodyPutAndTheCkptCasCannotDeleteT auto store = openPool(backend); CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); CasOperation op = requests.admit(); - const uint64_t epoch = store->writerEpoch(); + const uint64_t epoch = store->liveWriterEpoch(); const RootNamespace ns{"srv1/ckpt_race"}; ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{epoch, 1})); @@ -1185,7 +1185,7 @@ TEST(CASRefCheckpoint, TheCheckpointIsWrittenOncePerPublicationAndNotOnIdleAttem auto store = openPool(backend); CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); CasOperation op = requests.admit(); - const uint64_t epoch = store->writerEpoch(); + const uint64_t epoch = store->liveWriterEpoch(); const RootNamespace ns{"srv1/ckpt_republish"}; ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{epoch, 1})); @@ -1218,12 +1218,12 @@ TEST(CASRefCheckpoint, SnapshotPublisherRefusesEpochSealCandidateWithoutAnyWrite uint64_t predecessor_epoch = 0; { auto predecessor = openPool(backend); - predecessor_epoch = predecessor->writerEpoch(); + predecessor_epoch = predecessor->liveWriterEpoch(); ASSERT_EQ(publishRef(predecessor, ns, "ref_1", 1), (RefTxnId{predecessor_epoch, 1})); } auto store = openPool(backend); - ASSERT_GT(store->writerEpoch(), predecessor_epoch); + ASSERT_GT(store->liveWriterEpoch(), predecessor_epoch); ASSERT_EQ(store->listRefs(ns).size(), 1u) << "recovery must close the predecessor epoch before publishing"; const RefTxnId seal_id{predecessor_epoch, 2}; @@ -1242,7 +1242,7 @@ TEST(CASRefCheckpoint, SnapshotPublisherRefusesEpochSealCandidateWithoutAnyWrite EXPECT_EQ(backend->writes(ckpt_key), ckpt_writes_before); /// Once an ordinary transaction advances the candidate beyond the seal, normal publication resumes. - ASSERT_EQ(publishRef(store, ns, "ref_2", 2), (RefTxnId{store->writerEpoch(), 1})); + ASSERT_EQ(publishRef(store, ns, "ref_2", 2), (RefTxnId{store->liveWriterEpoch(), 1})); EXPECT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); } @@ -1255,7 +1255,7 @@ TEST(CASRefCheckpoint, NeedsRecoveryReplaysBeforeCheckpointAdvance) CasOperation op = requests.admit(); const RootNamespace ns{"srv1/ckpt_poisoned"}; - ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->writerEpoch(), 1})); + ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->liveWriterEpoch(), 1})); const NamespaceLifeId life = liveLifeOrFail(op, store->layout(), ns); const String key = store->layout().refCkptKey(life); const auto before = readCkpt(op, store->layout(), life); @@ -1305,7 +1305,7 @@ TEST(CASRefCheckpoint, APublishFencedOutMidAttemptDoesNotAdvanceTheCheckpoint) CasRequests requests = DB::Cas::tests::openRequestsForTest(backend); CasOperation op = requests.admit(); - ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->writerEpoch(), 1})); + ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->liveWriterEpoch(), 1})); const NamespaceLifeId life = liveLifeOrFail(op, store->layout(), ns); const String ckpt_key = store->layout().refCkptKey(life); const auto before = readCkpt(op, store->layout(), life); diff --git a/src/Disks/tests/gtest_cas_ref_contiguous_alloc.cpp b/src/Disks/tests/gtest_cas_ref_contiguous_alloc.cpp index 31648e992ceb..c6bef654ef15 100644 --- a/src/Disks/tests/gtest_cas_ref_contiguous_alloc.cpp +++ b/src/Disks/tests/gtest_cas_ref_contiguous_alloc.cpp @@ -176,7 +176,7 @@ TEST(CASRefContiguousAlloc, TwoNamespacesAllocateIndependently) { auto backend = std::make_shared(); auto store = openPool(backend); - const uint64_t epoch = store->writerEpoch(); + const uint64_t epoch = store->liveWriterEpoch(); const RootNamespace ns_a{"srv1/contig_ns_a"}; const RootNamespace ns_b{"srv1/contig_ns_b"}; @@ -204,7 +204,7 @@ TEST(CASRefContiguousAlloc, PreAttemptRefusalConsumesNoId) { auto backend = std::make_shared(); auto store = openPoolFenceControlled(backend); - const uint64_t epoch = store->writerEpoch(); + const uint64_t epoch = store->liveWriterEpoch(); const RootNamespace ns{"srv1/contig_no_gap"}; ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{epoch, 1})); @@ -239,13 +239,13 @@ TEST(CASRefContiguousAlloc, EpochChangeRestartsTheSequenceAtOne) uint64_t e1 = 0; { auto predecessor = openPool(backend); - e1 = predecessor->writerEpoch(); + e1 = predecessor->liveWriterEpoch(); ASSERT_EQ(publishRef(predecessor, ns, "ref_1", 1), (RefTxnId{e1, 1})); ASSERT_EQ(publishRef(predecessor, ns, "ref_2", 2), (RefTxnId{e1, 2})); } /// predecessor destroyed: its mount lease is released auto successor = openPool(backend); - const uint64_t e2 = successor->writerEpoch(); + const uint64_t e2 = successor->liveWriterEpoch(); ASSERT_GT(e2, e1); EXPECT_EQ(publishRef(successor, ns, "ref_3", 3), (RefTxnId{e2, 1})) << "a fresh incarnation starts this table's sequence over at 1"; @@ -324,7 +324,7 @@ TEST(CASRefContiguousAlloc, NeedsRecoveryReplaysBeforeAllocatingTheNextId) { auto backend = std::make_shared(); auto store = openPool(backend); - const uint64_t epoch = store->writerEpoch(); + const uint64_t epoch = store->liveWriterEpoch(); const RootNamespace ns{"srv1/contig_durable_floor"}; ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{epoch, 1})); @@ -373,7 +373,7 @@ TEST(CASRefContiguousAlloc, NeedsRecoveryReplaysBeforeSnapshotPublication) { auto backend = std::make_shared(); auto store = openPool(backend); - const uint64_t epoch = store->writerEpoch(); + const uint64_t epoch = store->liveWriterEpoch(); const RootNamespace ns{"srv1/contig_poison_publish"}; ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{epoch, 1})); @@ -420,7 +420,7 @@ TEST(CASRefContiguousAlloc, RecreationRefusedWhileAMountSlotIsStillHeld) auto backend = std::make_shared(); auto holder = openPool(backend); const RootNamespace ns{"srv1/contig_quiesce"}; - ASSERT_EQ(publishRef(holder, ns, "ref_1", 1), (RefTxnId{holder->writerEpoch(), 1})); + ASSERT_EQ(publishRef(holder, ns, "ref_1", 1), (RefTxnId{holder->liveWriterEpoch(), 1})); /// The operator removes the pool identity, intending to recreate -- but the holder is still up. ASSERT_EQ(eraseKeysContaining(*backend, "_pool_meta"), 1u); @@ -434,7 +434,7 @@ TEST(CASRefContiguousAlloc, RecreationRefusedWhileAMountSlotIsStillHeld) << "the holder must be identified so the operator knows what to stop: " << message; /// And the holder is untouched by the refused recreation: its own stream continues contiguously. - EXPECT_EQ(publishRef(holder, ns, "ref_2", 2), (RefTxnId{holder->writerEpoch(), 2})); + EXPECT_EQ(publishRef(holder, ns, "ref_2", 2), (RefTxnId{holder->liveWriterEpoch(), 2})); } /// Recreation quiesce, acceptance leg. Once the holder is gone its slot carries the graceful-farewell @@ -463,7 +463,7 @@ TEST(CASRefContiguousAlloc, RecreationProceedsOnceTheHolderIsTerminal) /// mints a fresh pool that starts its own ref stream at 1. ASSERT_GT(eraseKeysContaining(*backend, ""), 0u); auto recreated = openPoolWithoutSeeding(backend, "test"); - EXPECT_EQ(publishRef(recreated, ns, "ref_1", 1), (RefTxnId{recreated->writerEpoch(), 1})); + EXPECT_EQ(publishRef(recreated, ns, "ref_1", 1), (RefTxnId{recreated->liveWriterEpoch(), 1})); } /// The other half of the rule: if the prefix IS cleared while a writer survives (the mistake the @@ -491,7 +491,7 @@ TEST(CASRefContiguousAlloc, SurvivingWriterIsFencedByTheRecreatedPoolsMount) DB::Cas::tests::seedPoolMetaForRestart(*backend); auto survivor = Pool::open(backend, survivor_cfg); const RootNamespace ns{"srv1/contig_survivor"}; - ASSERT_EQ(publishRef(survivor, ns, "ref_1", 1), (RefTxnId{survivor->writerEpoch(), 1})); + ASSERT_EQ(publishRef(survivor, ns, "ref_1", 1), (RefTxnId{survivor->liveWriterEpoch(), 1})); const uint64_t skipped_before = ProfileEvents::global_counters[ProfileEvents::CASMountReleaseSkippedForeignOccupant].load(); const uint64_t violations_before @@ -519,7 +519,7 @@ TEST(CASRefContiguousAlloc, SurvivingWriterIsFencedByTheRecreatedPoolsMount) << "the survivor's queued write must be refused"; /// The recreated pool is unaffected and owns the stream from 1. - EXPECT_EQ(publishRef(recreated, ns, "ref_1", 1), (RefTxnId{recreated->writerEpoch(), 1})); + EXPECT_EQ(publishRef(recreated, ns, "ref_1", 1), (RefTxnId{recreated->liveWriterEpoch(), 1})); /// The survivor's TEARDOWN is the other half, and it is asserted here rather than left to the /// destructor at scope exit. A terminal renewer must skip release without backend I/O: the renewal diff --git a/src/Disks/tests/gtest_cas_ref_snapshot_publish_ordering.cpp b/src/Disks/tests/gtest_cas_ref_snapshot_publish_ordering.cpp index dbb53e10fe0a..2d02c68dd784 100644 --- a/src/Disks/tests/gtest_cas_ref_snapshot_publish_ordering.cpp +++ b/src/Disks/tests/gtest_cas_ref_snapshot_publish_ordering.cpp @@ -142,7 +142,7 @@ RefTxnId publishRef(const PoolPtr & store, const RootNamespace & ns, const Strin void forceAdoptablePublishWedge( const PoolPtr & store, const RootNamespace & ns, uint64_t ref_sequence, const String & ref, uint64_t ordinal) { - const RefTxnId txn_id{store->writerEpoch(), ref_sequence}; + const RefTxnId txn_id{store->liveWriterEpoch(), ref_sequence}; RefLogTxn txn; txn.ns = ns.string(); txn.txn_id = txn_id; @@ -165,9 +165,9 @@ TEST(CASRefSnapshotPublishOrdering, SnapshotBodyIsDurableBeforeCheckpointAdvance auto store = openPool(backend); const RootNamespace ns{"srv1/order_body_before_ckpt"}; - ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->writerEpoch(), 1})); + ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->liveWriterEpoch(), 1})); const NamespaceLifeId life = *store->refTableLifeForTest(ns); - const String snapshot_key = store->layout().refSnapshotKey(life, RefTxnId{store->writerEpoch(), 1}); + const String snapshot_key = store->layout().refSnapshotKey(life, RefTxnId{store->liveWriterEpoch(), 1}); const String ckpt_key = store->layout().refCkptKey(life); /// The birth transaction above already CAS'd `_ckpt` itself (once for its own `life_epoch`, once for @@ -207,9 +207,9 @@ TEST(CASRefSnapshotPublishOrdering, AdoptionHappensLastAndOnlyAfterBothDurableEf auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/order_adoption_after_both"}; - ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->writerEpoch(), 1})); + ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->liveWriterEpoch(), 1})); const NamespaceLifeId life = *store->refTableLifeForTest(ns); - const String snapshot_key = store->layout().refSnapshotKey(life, RefTxnId{store->writerEpoch(), 1}); + const String snapshot_key = store->layout().refSnapshotKey(life, RefTxnId{store->liveWriterEpoch(), 1}); const String ckpt_key = store->layout().refCkptKey(life); /// Refuse the `_ckpt` CAS for as long as `publishCkpt` keeps reissuing, so it ends at its own retry @@ -236,7 +236,7 @@ TEST(CASRefSnapshotPublishOrdering, AdoptionHappensLastAndOnlyAfterBothDurableEf EXPECT_EQ(backend->putCount(snapshot_key), 2u) << "the retry's body create is its own attempt, accepted against identical, already-durable " "bytes rather than writing a second object"; - EXPECT_EQ(store->newestPublishedSnapshotIdForTest(ns), std::make_optional(RefTxnId{store->writerEpoch(), 1})) + EXPECT_EQ(store->newestPublishedSnapshotIdForTest(ns), std::make_optional(RefTxnId{store->liveWriterEpoch(), 1})) << "adoption happens exactly once, after both effects are durable"; } @@ -278,11 +278,11 @@ TEST(CASRefSnapshotPublishOrdering, NeedsRecoveryLaneRecoversBeforeAnySnapshotPu auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/order_poisoned_refuses"}; - ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->writerEpoch(), 1})); + ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->liveWriterEpoch(), 1})); const NamespaceLifeId life = *store->refTableLifeForTest(ns); const String ckpt_key = store->layout().refCkptKey(life); /// The durable transaction the stale cache will be missing: `dropRef`'s removal, sequence 2. - const RefTxnId missing_durable_txn{store->writerEpoch(), 2}; + const RefTxnId missing_durable_txn{store->liveWriterEpoch(), 2}; const String next_snapshot_key = store->layout().refSnapshotKey(life, missing_durable_txn); /// Drive the very next mutation's OWN checkpoint-frontier CAS into persistent conflict: the log PUT @@ -388,7 +388,7 @@ TEST(CASRefSnapshotPublishOrdering, PublishBackoffDecisionsAreCharacterized) auto clock = VirtualRetryClock::installOn(store); const RootNamespace ns{"srv1/order_backoff"}; - ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->writerEpoch(), 1})); + ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->liveWriterEpoch(), 1})); store->waitForSnapshotPublishSettleForTest(ns); /// drain the birth's own auto-dispatched publish /// The birth's own auto-dispatch already published a snapshot at this point (threshold 0); the /// baseline every "no new publish yet" check below compares against. @@ -406,7 +406,7 @@ TEST(CASRefSnapshotPublishOrdering, PublishBackoffDecisionsAreCharacterized) const auto dispatchCount = [&] { return global_counters[ProfileEvents::CASRefSnapshotPublishDispatched].load(); }; /// Attempt 1: admitted immediately (no backoff armed yet). Fails -> backoff armed at the initial 1000ms. - ASSERT_EQ(publishRef(store, ns, "ref_2", 2), (RefTxnId{store->writerEpoch(), 2})); + ASSERT_EQ(publishRef(store, ns, "ref_2", 2), (RefTxnId{store->liveWriterEpoch(), 2})); store->waitForSnapshotPublishSettleForTest(ns); const uint64_t d1 = dispatchCount(); EXPECT_EQ(store->newestPublishedSnapshotIdForTest(ns), snapshot_after_birth) @@ -458,7 +458,7 @@ TEST(CASRefSnapshotPublishOrdering, PublishBackoffDecisionsAreCharacterized) EXPECT_NE(store->newestPublishedSnapshotIdForTest(ns), snapshot_after_birth) << "the fault is disarmed, so this attempt actually advances the published snapshot"; - ASSERT_EQ(publishRef(store, ns, "ref_3", 3), (RefTxnId{store->writerEpoch(), 3})); + ASSERT_EQ(publishRef(store, ns, "ref_3", 3), (RefTxnId{store->liveWriterEpoch(), 3})); store->waitForSnapshotPublishSettleForTest(ns); EXPECT_EQ(dispatchCount(), d1 + 4) << "resetPublishBackoff must have cleared the cooldown: the very next over-threshold trigger, at " @@ -471,7 +471,7 @@ TEST(CASRefSnapshotPublishOrdering, PublishBackoffDecisionsAreCharacterized) /// 1000ms, admitted at 1000ms -- which a no-op reset cannot produce (it would refuse both probes, /// since the stale deadline is still far in the future). backend->armWriteFailure("_snap/", kFaultsBeyondTheRetryWindow); - ASSERT_EQ(publishRef(store, ns, "ref_4", 4), (RefTxnId{store->writerEpoch(), 4})); + ASSERT_EQ(publishRef(store, ns, "ref_4", 4), (RefTxnId{store->liveWriterEpoch(), 4})); store->waitForSnapshotPublishSettleForTest(ns); const uint64_t d2 = dispatchCount(); *fake_now += 500; @@ -530,11 +530,11 @@ TEST(CASRefSnapshotPublishOrdering, NotReadyRefusalBacksOffAndResetsAfterDurable return global_counters[ProfileEvents::CASRefSnapshotPublishBackoff].load(); }; - ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->writerEpoch(), 1})); + ASSERT_EQ(publishRef(store, ns, "ref_1", 1), (RefTxnId{store->liveWriterEpoch(), 1})); store->waitForSnapshotPublishSettleForTest(ns); ASSERT_EQ( store->newestPublishedSnapshotIdForTest(ns), - std::make_optional(RefTxnId{store->writerEpoch(), 1})); + std::make_optional(RefTxnId{store->liveWriterEpoch(), 1})); /// Direct calls remain one attempt per invocation even while a cooldown is armed. This first /// refusal is also the non-hanging RED discriminator: without the production fix the backoff @@ -552,7 +552,7 @@ TEST(CASRefSnapshotPublishOrdering, NotReadyRefusalBacksOffAndResetsAfterDurable /// Resolve the exact wedge through the real append-lane adoption path. Its adopted txn and /// the caller's own txn raise the table above threshold, but the warm-up cooldown prevents an /// automatic publish while the fixture prepares one uncovered tail entry. - ASSERT_EQ(publishRef(store, ns, "ref_3", 3), (RefTxnId{store->writerEpoch(), 3})); + ASSERT_EQ(publishRef(store, ns, "ref_3", 3), (RefTxnId{store->liveWriterEpoch(), 3})); ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); bool appended_during_capture = false; @@ -561,14 +561,14 @@ TEST(CASRefSnapshotPublishOrdering, NotReadyRefusalBacksOffAndResetsAfterDurable if (appended_during_capture) return; appended_during_capture = true; - EXPECT_EQ(publishRef(store, ns, "ref_4", 4), (RefTxnId{store->writerEpoch(), 4})); + EXPECT_EQ(publishRef(store, ns, "ref_4", 4), (RefTxnId{store->liveWriterEpoch(), 4})); }); ASSERT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); store->setSnapshotAfterCaptureHookForTest(nullptr); ASSERT_TRUE(appended_during_capture); ASSERT_EQ( store->newestPublishedSnapshotIdForTest(ns), - std::make_optional(RefTxnId{store->writerEpoch(), 3})); + std::make_optional(RefTxnId{store->liveWriterEpoch(), 3})); /// One uncovered tail entry now exists with no cooldown. Make the lane non-Ready before the read /// trigger, so the first production dispatch is an admitted refusal rather than a body PUT. @@ -622,7 +622,7 @@ TEST(CASRefSnapshotPublishOrdering, NotReadyRefusalBacksOffAndResetsAfterDurable /// Adopt the outstanding wedge through production and publish durably. The hook commits one later /// txn after capture while the capped cooldown is still armed; a correct durable publication resets /// that cooldown, so an immediate same-clock read dispatches the leftover tail without waiting. - ASSERT_EQ(publishRef(store, ns, "ref_6", 6), (RefTxnId{store->writerEpoch(), 6})); + ASSERT_EQ(publishRef(store, ns, "ref_6", 6), (RefTxnId{store->liveWriterEpoch(), 6})); ASSERT_EQ(store->laneStateForTest(ns), RefLaneState::Ready); bool appended_after_reset_capture = false; store->setSnapshotAfterCaptureHookForTest([&] @@ -630,14 +630,14 @@ TEST(CASRefSnapshotPublishOrdering, NotReadyRefusalBacksOffAndResetsAfterDurable if (appended_after_reset_capture) return; appended_after_reset_capture = true; - EXPECT_EQ(publishRef(store, ns, "ref_7", 7), (RefTxnId{store->writerEpoch(), 7})); + EXPECT_EQ(publishRef(store, ns, "ref_7", 7), (RefTxnId{store->liveWriterEpoch(), 7})); }); ASSERT_TRUE(store->tryPublishSnapshotAndAdvanceCheckpointOnce(ns)); store->setSnapshotAfterCaptureHookForTest(nullptr); ASSERT_TRUE(appended_after_reset_capture); ASSERT_EQ( store->newestPublishedSnapshotIdForTest(ns), - std::make_optional(RefTxnId{store->writerEpoch(), 6})); + std::make_optional(RefTxnId{store->liveWriterEpoch(), 6})); const uint64_t dispatches_before_reset_probe = dispatch_count(); store->resolveRef(ns, "ref_1"); @@ -646,5 +646,5 @@ TEST(CASRefSnapshotPublishOrdering, NotReadyRefusalBacksOffAndResetsAfterDurable << "durable publication must clear the capped cooldown for an immediate same-clock trigger"; EXPECT_EQ( store->newestPublishedSnapshotIdForTest(ns), - std::make_optional(RefTxnId{store->writerEpoch(), 7})); + std::make_optional(RefTxnId{store->liveWriterEpoch(), 7})); } diff --git a/src/Disks/tests/gtest_cas_ref_writer.cpp b/src/Disks/tests/gtest_cas_ref_writer.cpp index 159b12fb99e9..b5d38fe584d6 100644 --- a/src/Disks/tests/gtest_cas_ref_writer.cpp +++ b/src/Disks/tests/gtest_cas_ref_writer.cpp @@ -3388,7 +3388,7 @@ TEST(CASRefWriterStalePrecommitSweep, BoundedBatchesAndInterruptionResumeAcrossM uint64_t e1 = 0; { auto predecessor = openPool(backend); - e1 = predecessor->writerEpoch(); + e1 = predecessor->liveWriterEpoch(); } /// predecessor released; only its epoch is needed -- the stale precommits are seeded raw below /// Seed kTotalStale precommits directly (bypassing any Pool) under the predecessor's epoch, @@ -3676,11 +3676,10 @@ TEST(CASRefWriterStalePrecommitSweep, VerifiedCleanSweepClearsFlagWithoutEvents) } /// =================================================================================== -/// C1: self-remount establishes a fresh ref-protocol incarnation (spec §Startup And Recovery / -/// §write-fence). A self-remount bumps the durable writer_epoch, so every ref transaction it stamps -/// afterward sorts strictly above any log a dead-incarnation or same-uuid twin left durable under an -/// older epoch, and it drops its stale in-memory cache so the next touch re-recovers under the new -/// epoch. The unfixed code kept the open-time `process_epoch` and the cached tables across the fence-out. +/// A self-remount establishes a fresh ref-protocol incarnation. It bumps the durable writer_epoch, so +/// every ref transaction it stamps afterward sorts strictly above any log a dead-incarnation or +/// same-uuid twin left durable under an older epoch, and it drops its stale in-memory cache so the next +/// touch re-recovers under the new epoch. /// =================================================================================== namespace diff --git a/src/Disks/tests/gtest_cas_writer_duties.cpp b/src/Disks/tests/gtest_cas_writer_duties.cpp index 6706df7e0109..9b5937cf3b2c 100644 --- a/src/Disks/tests/gtest_cas_writer_duties.cpp +++ b/src/Disks/tests/gtest_cas_writer_duties.cpp @@ -392,7 +392,7 @@ TEST(CASWriterDuties, PendingDutySkipsCleanFarewellAndSuccessorSweepsTheCrashRem ManifestId abandoned_id; auto abandoned = stageEmptyManifest(predecessor, ns, "abandoned", abandoned_id); abandoned->precommitAdd(ns, "abandoned", abandoned_id); - const uint64_t predecessor_epoch = predecessor->writerEpoch(); + const uint64_t predecessor_epoch = predecessor->liveWriterEpoch(); const Layout layout = predecessor->layout(); const String mount_key = layout.mountKey("test"); @@ -428,7 +428,7 @@ TEST(CASWriterDuties, PendingDutySkipsCleanFarewellAndSuccessorSweepsTheCrashRem waits->push(ms); }, }); - ASSERT_GT(successor_store->writerEpoch(), predecessor_epoch); + ASSERT_GT(successor_store->liveWriterEpoch(), predecessor_epoch); ASSERT_FALSE(waits->empty()) << "the predecessor supplied no clean-death certificate"; ManifestId successor_id; @@ -525,7 +525,7 @@ TEST(CASWriterDuties, RejectedAttemptBodyIsEventuallyNominatedAndSwept) driveToNetworkErrorGiveUp(*backend, *clock, [&] { rejected->precommitAdd(ns, "rejected", rejected_id); }); ASSERT_TRUE(predecessor->refLaneWedgedForTest(ns)); ASSERT_EQ(rejected->precommitState(), PartWriteTxn::PrecommitState::Uncertain); - const uint64_t predecessor_epoch = predecessor->writerEpoch(); + const uint64_t predecessor_epoch = predecessor->liveWriterEpoch(); rejected.reset(); predecessor.reset(); @@ -556,7 +556,7 @@ TEST(CASWriterDuties, RejectedAttemptBodyIsEventuallyNominatedAndSwept) waits->push(ms); }, }); - ASSERT_GT(successor_store->writerEpoch(), predecessor_epoch); + ASSERT_GT(successor_store->liveWriterEpoch(), predecessor_epoch); ASSERT_FALSE(waits->empty()) << "the predecessor supplied no clean-death certificate"; /// An ordinary successor mutation both drains the inherited duty as a no-op (the rejected grant From 97953a2f47352e820469d3216438d833e09ba2d0 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 09:52:43 +0200 Subject: [PATCH 26/37] Test the Vanished lifecycle instead of a shadow flag in enterVanished `terminal_state_published` repeated what the `Vanished*` lifecycle values already say. `enterVanished` and `setLifecycleForTest` test `isVanished` under `driver_mutex`; the publication order is unchanged. Co-Authored-By: Claude Sonnet 5.5 --- .../ContentAddressed/Pool/CasMountRuntime.cpp | 14 +++++--------- .../ContentAddressed/Pool/CasMountRuntime.h | 10 ++-------- 2 files changed, 7 insertions(+), 17 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index f9f25a6bbed4..732221fdae02 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -811,9 +811,6 @@ void CasMountRuntime::setLifecycleForTest(PoolLifecycle lc) || lc == PoolLifecycle::VanishedForgotten) { vanished_intent.store(true, std::memory_order_release); - /// Keep the terminal-state guard consistent with the forced state, so a later `enterVanished` - /// (unusual, but not forbidden) is a clean no-op rather than re-storing / re-logging. - terminal_state_published.store(true, std::memory_order_release); } } @@ -929,7 +926,7 @@ void CasMountRuntime::enterVanished(PoolLifecycle which, const String & reason) "CasMountRuntime::enterVanished called with a non-terminal lifecycle value"); } - if (terminal_state_published.load(std::memory_order_acquire)) + if (isVanished()) return; if (config.vanished_reason_prepare_hook_for_test) @@ -939,12 +936,12 @@ void CasMountRuntime::enterVanished(PoolLifecycle which, const String & reason) static_assert(std::is_nothrow_move_assignable_v); bool transitioned = false; - /// The guard is published LAST. Reason preparation above is the only potentially-throwing step; - /// once this block begins, the statically-proven-noexcept move and atomic stores either publish - /// one complete terminal transition or observe that an earlier transition already completed. + /// The lifecycle store is the LAST step. Reason preparation above is the only potentially-throwing + /// step; once this block begins, the statically-proven-noexcept move and atomic stores either + /// publish one complete terminal transition or observe that an earlier transition already completed. { auto lock = lockTerminalPublication(); - if (!terminal_state_published.load(std::memory_order_acquire)) + if (!isVanished()) { transitioned = true; @@ -968,7 +965,6 @@ void CasMountRuntime::enterVanished(PoolLifecycle which, const String & reason) /// a `Vanished` state (their compare-exchanges are keyed on `Live`/`TransientNotLive`), so this /// value is absorbing. pool_lifecycle.store(which, std::memory_order_release); - terminal_state_published.store(true, std::memory_order_release); } } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index 47ae9d4da33c..813de39c3b5f 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -262,8 +262,8 @@ class CasMountRuntime /// One-way transition to a fully-terminal `Vanished` value (spec §3). Publishes the terminal-intent /// latch (so the runtime stops scheduling remount work and the lease loop exits at its next step /// boundary) if it is not already published, records `reason`, stores the state, then emits ONE WARN + - /// one `CASDataRootVanished` ProfileEvent. Idempotent: the first terminal STATE transition wins (a - /// dedicated latch keyed separately from `vanished_intent`, because FORGET publishes that intent latch + /// one `CASDataRootVanished` ProfileEvent. Idempotent: the first terminal STATE transition wins (keyed + /// on the `Vanished*` lifecycle value, not on `vanished_intent`, because FORGET publishes that intent /// early at step 1). `which` MUST be one of the two `Vanished*` values (`VanishedReplaced` or /// `VanishedForgotten`). `reason` is retained and /// surfaced verbatim in the `VanishedForgotten` [D5] error message (see `vanishedReason`). Threads exit @@ -540,12 +540,6 @@ class CasMountRuntime /// `remountTerminal`, so a terminal pool's runtime consumer never schedules a remount and the lease /// loop exits at its next step boundary — no claim/allocate/write after the pool is (being driven) terminal. std::atomic vanished_intent{false}; - /// Idempotency guard for the terminal STATE transition (`enterVanished`'s body). Distinct from - /// `vanished_intent`: FORGET publishes that intent latch at step 1, so it can no longer serve as the - /// "state transition already done" flag. Published last, after the no-throw reason move and terminal - /// stores complete under `driver_mutex`; a preparation exception therefore leaves it clear so a later - /// `enterVanished` can retry the whole transition. - std::atomic terminal_state_published{false}; /// The reason recorded by the winning `enterVanished` (see `vanishedReason`). Written once, BEFORE the /// `pool_lifecycle` release-store, and immutable thereafter — so a reader that acquire-observes a /// terminal state also observes this string. Empty when no terminal transition has run. From e488f976389a30a23a19e9bf895607860e4e0d52 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 09:55:24 +0200 Subject: [PATCH 27/37] Read the renewal period and the boot clock from one place `startBackgroundWorkers` reads `MountConfig::mount_renew_period` instead of keeping a copy of its argument. `MountRenewOperationEnvironment` loses `boot_ms`: the renewer already holds the same clock as `boot_ms_fn`. Tests set the period in their `MountConfig`. `CasServerRoot.cpp` also drops the `` and `` includes, which nothing in the file uses. Co-Authored-By: Claude Sonnet 5.5 --- .../ContentAddressed/Pool/CasMountRuntime.cpp | 6 +- .../ContentAddressed/Pool/CasMountRuntime.h | 3 +- .../ContentAddressed/Pool/CasPool.cpp | 2 +- .../ContentAddressed/Pool/CasServerRoot.cpp | 9 +- .../ContentAddressed/Pool/CasServerRoot.h | 1 - src/Disks/tests/gtest_cas_event_log.cpp | 1 - src/Disks/tests/gtest_cas_heartbeat.cpp | 53 +++--- src/Disks/tests/gtest_cas_pool.cpp | 151 +++++++++--------- 8 files changed, 107 insertions(+), 119 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index 732221fdae02..ded2a2ea9941 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -480,7 +480,6 @@ uint64_t CasMountRuntime::startRenewer() MountRenewOperationEnvironment CasMountRuntime::renewalEnvironment() { return MountRenewOperationEnvironment{ - .boot_ms = [this] { return bootMsNow(); }, .live = [this] { return renewalLive() && (!config.renewal_live_for_test || config.renewal_live_for_test()); @@ -617,7 +616,7 @@ ThreadFromGlobalPool CasMountRuntime::makeWorker(std::function body) return ThreadFromGlobalPool(std::move(body)); } -void CasMountRuntime::startBackgroundWorkers(std::chrono::milliseconds period) +void CasMountRuntime::startBackgroundWorkers() { { std::lock_guard lock(driver_mutex); @@ -628,7 +627,6 @@ void CasMountRuntime::startBackgroundWorkers(std::chrono::milliseconds period) workers_started = true; workers_stop_requested = false; lease_thread_id = {}; - renewal_period = period; } ThreadFromGlobalPool worker; @@ -702,7 +700,7 @@ void CasMountRuntime::renewalLoop() else { const uint64_t last_anchor = mount_renewer->lastCommittedAttemptStartBootMs(); - const uint64_t period_ms = static_cast(std::max(0, renewal_period.count())); + const uint64_t period_ms = static_cast(std::max(0, config.mount_renew_period.count())); const uint64_t due = last_anchor > std::numeric_limits::max() - period_ms ? std::numeric_limits::max() : last_anchor + period_ms; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index 813de39c3b5f..91487e93ecf9 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -374,7 +374,7 @@ class CasMountRuntime void installRenewer(UInt128 our_uuid, uint64_t writer_epoch, const std::function & now_ms); uint64_t startRenewer(); void renewerReset(); - void startBackgroundWorkers(std::chrono::milliseconds period); + void startBackgroundWorkers(); void stopBackgroundWorkers(); /// Latch a recovery generation for the lease thread. It never constructs a thread. void scheduleRemount(); @@ -501,7 +501,6 @@ class CasMountRuntime bool workers_stop_requested = false; /// The lease thread, from its first statement until `stopBackgroundWorkers` joins it. std::thread::id lease_thread_id; - std::chrono::milliseconds renewal_period{0}; uint64_t remount_requested_generation = 0; uint64_t remount_handled_generation = 0; /// The requested generation `beginReclaim` recorded; `armIfAdmissible` acknowledges it. diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp index a5939bc31c0c..861ab3e426eb 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp @@ -864,7 +864,7 @@ void Pool::mountWritable(PoolPtr & store, UInt128 our_uuid, MountClaimPolicy pol "mount has no lease thread to renew it; retry the open", srid, ttl_ms_u); return; } - store->mount_runtime.startBackgroundWorkers(store->config.mount_renew_period); + store->mount_runtime.startBackgroundWorkers(); if (armed) return; LOG_WARNING(getLogger("CasPool"), diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp index e37c4ddf7da0..ff88b3f3ae27 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp @@ -15,13 +15,11 @@ #include #include -#include #include #include #include #include #include -#include #include #include #include @@ -1345,14 +1343,13 @@ MountRenewResult MountLeaseRenewer::renew(const MountRenewOperationEnvironment & key, static_cast(renewer_state)); - const auto boot_clock = environment.boot_ms ? environment.boot_ms : boot_ms_fn; /// Sampled BEFORE the write. A refused admission is reported as "never attempted" only when this /// node had already been asked to stop, and reading the flag afterwards could not tell that apart /// from a flag the refusal itself set. const bool cancelled = environment.cancelled && environment.cancelled(); const uint64_t wall_ms = now_ms_fn(); - const uint64_t attempt_start_boot_ms = boot_clock(); + const uint64_t attempt_start_boot_ms = boot_ms_fn(); const uint64_t next_seq = seq + 1; const UInt128 write_attempt_id = newMountWriteAttemptId(); const String body = encodeBody(next_seq, wall_ms, min_active_build_sequence_fn(), write_attempt_id); @@ -1362,9 +1359,9 @@ MountRenewResult MountLeaseRenewer::renew(const MountRenewOperationEnvironment & result.writer_epoch = writer_epoch; result.seq = next_seq; result.write_attempt_id = write_attempt_id; - const auto finished = [&boot_clock, attempt_start_boot_ms](MountRenewResult done) + const auto finished = [this, attempt_start_boot_ms](MountRenewResult done) { - done.elapsed_ms = elapsedSince(attempt_start_boot_ms, boot_clock()); + done.elapsed_ms = elapsedSince(attempt_start_boot_ms, boot_ms_fn()); return done; }; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h index 4554e76c03b7..d8160af69f20 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h @@ -105,7 +105,6 @@ struct MountRenewRequestEvent struct MountRenewOperationEnvironment { - std::function boot_ms; /// Facts the mount fence cannot see (a remount request, a pool no longer live, shutdown). FALSE ends /// the renewal exactly as a lost fence does; the engine does not need to know which refused. std::function live; diff --git a/src/Disks/tests/gtest_cas_event_log.cpp b/src/Disks/tests/gtest_cas_event_log.cpp index 153839908b85..9f9564be285f 100644 --- a/src/Disks/tests/gtest_cas_event_log.cpp +++ b/src/Disks/tests/gtest_cas_event_log.cpp @@ -516,7 +516,6 @@ TEST(CASEvent, TheReportIsBuiltFromTheResult) ASSERT_EQ(quiet.attempts_sent, 1u); reportMountRenewCompletion(quiet, server_root_id, sink, quiet.attempt_start_boot_ms + 1000, std::nullopt); const MountRenewResult skipped = renewer.renew(MountRenewOperationEnvironment{ - .boot_ms = {}, .live = [] { return false; }, .cancelled = [] { return true; }, .on_request = {}, diff --git a/src/Disks/tests/gtest_cas_heartbeat.cpp b/src/Disks/tests/gtest_cas_heartbeat.cpp index 1db269e2476c..3efba9218854 100644 --- a/src/Disks/tests/gtest_cas_heartbeat.cpp +++ b/src/Disks/tests/gtest_cas_heartbeat.cpp @@ -232,12 +232,10 @@ class RenewalScriptBackend final : public InMemoryBackend }; MountRenewOperationEnvironment renewalEnvironment( - uint64_t & boot_ms, const std::function & live = {}, const std::function & cancelled = {}) { return MountRenewOperationEnvironment{ - .boot_ms = [&boot_ms] { return boot_ms; }, .live = live, .cancelled = cancelled, .on_request = {}, @@ -996,7 +994,7 @@ TEST(CASHeartbeat, RenewerStateAllowsOnlyActiveReleaseOrTerminal) [&] { return wall_ms; }, [] { return uint64_t{7}; }, {}, std::chrono::milliseconds(20), [&] { return boot_ms; }); EXPECT_EQ(renewer.state(), MountLeaseRenewerState::New); - EXPECT_RENEWER_STATE_REJECTION(renewer.renew(renewalEnvironment(boot_ms))); + EXPECT_RENEWER_STATE_REJECTION(renewer.renew(renewalEnvironment())); EXPECT_RENEWER_STATE_REJECTION(renewer.release(renewer.lastCommittedAttemptStartBootMs() + 1000)); EXPECT_EQ(renewer.start(), 100u); EXPECT_RENEWER_STATE_REJECTION(renewer.start()); @@ -1004,7 +1002,7 @@ TEST(CASHeartbeat, RenewerStateAllowsOnlyActiveReleaseOrTerminal) renewer.release(renewer.lastCommittedAttemptStartBootMs() + 1000); EXPECT_EQ(renewer.state(), MountLeaseRenewerState::Released); EXPECT_RENEWER_STATE_REJECTION(renewer.start()); - EXPECT_RENEWER_STATE_REJECTION(renewer.renew(renewalEnvironment(boot_ms))); + EXPECT_RENEWER_STATE_REJECTION(renewer.renew(renewalEnvironment())); EXPECT_RENEWER_STATE_REJECTION(renewer.release(renewer.lastCommittedAttemptStartBootMs() + 1000)); } @@ -1024,12 +1022,12 @@ TEST(CASHeartbeat, RenewerStateAllowsOnlyActiveReleaseOrTerminal) /// Live until the ambiguous attempt's resolving read has run, so that attempt is the only one /// sent and the renewal ends terminal with it unsettled. const MountRenewResult result = renewer.renew( - renewalEnvironment(boot_ms, /*live=*/[&] { return backend->read_calls == 0; })); + renewalEnvironment(/*live=*/[&] { return backend->read_calls == 0; })); EXPECT_EQ(result.outcome, MountRenewOutcome::Terminal); EXPECT_NE(result.failure, nullptr); EXPECT_EQ(renewer.state(), MountLeaseRenewerState::RenewalTerminal); EXPECT_RENEWER_STATE_REJECTION(renewer.start()); - EXPECT_RENEWER_STATE_REJECTION(renewer.renew(renewalEnvironment(boot_ms))); + EXPECT_RENEWER_STATE_REJECTION(renewer.renew(renewalEnvironment())); EXPECT_RENEWER_STATE_REJECTION(renewer.release(renewer.lastCommittedAttemptStartBootMs() + 1000)); } @@ -1054,7 +1052,7 @@ TEST(CASHeartbeat, RenewalRetriesOneImmutableBodyAndAdoptsLostResponse) backend->attempts.clear(); backend->actions = {RenewalScriptBackend::Action::ThrowBefore, RenewalScriptBackend::Action::Delegate}; - MountRenewResult retried = renewer.renew(renewalEnvironment(boot_ms)); + MountRenewResult retried = renewer.renew(renewalEnvironment()); ASSERT_EQ(retried.outcome, MountRenewOutcome::Committed); ASSERT_EQ(backend->attempts.size(), 2u); EXPECT_EQ(backend->attempts[0].key, backend->attempts[1].key); @@ -1065,7 +1063,7 @@ TEST(CASHeartbeat, RenewalRetriesOneImmutableBodyAndAdoptsLostResponse) backend->attempts.clear(); backend->actions = {RenewalScriptBackend::Action::LandThenThrow}; - MountRenewResult adopted = renewer.renew(renewalEnvironment(boot_ms)); + MountRenewResult adopted = renewer.renew(renewalEnvironment()); EXPECT_EQ(adopted.outcome, MountRenewOutcome::Committed); EXPECT_TRUE(adopted.resolved_by_read); EXPECT_EQ(adopted.attempts_sent, 1u); @@ -1096,7 +1094,7 @@ TEST(CASHeartbeat, RenewalOverConnectFailuresRecoversWithoutASettleRead) for (int i = 0; i < 60; ++i) backend->actions.push_back(RenewalScriptBackend::Action::ThrowConnectHint); backend->actions.push_back(RenewalScriptBackend::Action::Delegate); - const MountRenewResult renewed = renewer.renew(renewalEnvironment(boot_ms)); + const MountRenewResult renewed = renewer.renew(renewalEnvironment()); ASSERT_EQ(renewed.outcome, MountRenewOutcome::Committed); EXPECT_GT(renewed.attempts_sent, 1u); EXPECT_FALSE(renewed.resolved_by_read); /// classification `committed_after_retry` @@ -1122,8 +1120,7 @@ TEST(CASHeartbeat, CancellationBeforeSendIsNotAttemptedAndAllowsRelease) renewer.start(); backend->attempts.clear(); backend->read_calls = 0; - const MountRenewResult result = renewer.renew(renewalEnvironment( - boot_ms, /*live=*/[] { return false; }, /*cancelled=*/[] { return true; })); + const MountRenewResult result = renewer.renew(renewalEnvironment(/*live=*/[] { return false; }, /*cancelled=*/[] { return true; })); EXPECT_EQ(result.outcome, MountRenewOutcome::NotAttempted); EXPECT_EQ(result.failure, nullptr); EXPECT_EQ(renewer.state(), MountLeaseRenewerState::Active); @@ -1151,7 +1148,7 @@ TEST(CASHeartbeat, CancellationAfterSendIsTerminalAndForbidsRelease) backend->cancel_after_write = [&] { cancelled = true; }; backend->actions = {RenewalScriptBackend::Action::ReturnThenCancel}; const MountRenewResult result = renewer.renew( - renewalEnvironment(boot_ms, /*live=*/[&] { return !cancelled; }, /*cancelled=*/[&] { return cancelled; })); + renewalEnvironment(/*live=*/[&] { return !cancelled; }, /*cancelled=*/[&] { return cancelled; })); const DB::Exception failure = terminalException(result); EXPECT_EQ(failure.code(), DB::ErrorCodes::NETWORK_ERROR); EXPECT_TRUE(result.sent_any); @@ -1178,7 +1175,7 @@ TEST(CASHeartbeat, SlowResolvedSuccessKeepsAttemptStartAnchor) boot_ms = 150; backend->cancel_after_write = [&] { boot_ms = 400; }; backend->actions = {RenewalScriptBackend::Action::LandThenThrow}; - const MountRenewResult result = renewer.renew(renewalEnvironment(boot_ms)); + const MountRenewResult result = renewer.renew(renewalEnvironment()); EXPECT_EQ(result.outcome, MountRenewOutcome::Committed); EXPECT_EQ(result.attempt_start_boot_ms, 150u); EXPECT_EQ(renewer.lastCommittedAttemptStartBootMs(), 150u); @@ -1209,7 +1206,7 @@ TEST(CASHeartbeat, SamePairTwinAndForeignOrSuccessorStayTerminal) mustCommit(ops.op.replace(layout.mountKey("test"), encodeMountLease(current), got->etag, Retry::standard()), "competing slot"); backend->read_calls = 0; - const MountRenewResult result = renewer.renew(renewalEnvironment(boot_ms)); + const MountRenewResult result = renewer.renew(renewalEnvironment()); const DB::Exception failure = terminalException(result); EXPECT_NE(failure.code(), DB::ErrorCodes::LOGICAL_ERROR); EXPECT_EQ(renewer.state(), MountLeaseRenewerState::RenewalTerminal); @@ -1239,7 +1236,7 @@ TEST(CASHeartbeat, ExpectedPredecessorThenLateLandingIsAdoptedExactly) RenewalScriptBackend::Action::ThrowBeforeThenLandAfterResolve, RenewalScriptBackend::Action::Delegate, }; - const MountRenewResult result = renewer.renew(renewalEnvironment(boot_ms)); + const MountRenewResult result = renewer.renew(renewalEnvironment()); EXPECT_EQ(result.outcome, MountRenewOutcome::Committed); EXPECT_TRUE(result.resolved_by_read); ASSERT_EQ(backend->attempts.size(), 2u); @@ -1303,7 +1300,7 @@ TEST(CASHeartbeat, GcFenceAndVanishedMountStayTerminal) mustCommit(ops.op.replace(key, encodeMountLease(fenced), got->etag, Retry::standard()), "fence-out"); } - const DB::Exception failure = terminalException(renewer.renew(renewalEnvironment(boot_ms))); + const DB::Exception failure = terminalException(renewer.renew(renewalEnvironment())); EXPECT_NE(failure.code(), DB::ErrorCodes::LOGICAL_ERROR); EXPECT_EQ(renewer.state(), MountLeaseRenewerState::RenewalTerminal); }; @@ -1318,7 +1315,7 @@ TEST(CASHeartbeat, RenewReturnsWhatItsReportNeeds) { ReportFieldsCase c("commit"); c.backend->on_attempt = [&boot_ms = c.boot_ms] { boot_ms += 250; }; - const MountRenewResult result = c.renewer.renew(renewalEnvironment(c.boot_ms)); + const MountRenewResult result = c.renewer.renew(renewalEnvironment()); ASSERT_EQ(result.outcome, MountRenewOutcome::Committed); ASSERT_EQ(c.backend->attempts.size(), 1u); const MountLease sent = decodeMountLease(c.backend->attempts.back().bytes); @@ -1344,7 +1341,7 @@ TEST(CASHeartbeat, RenewReturnsWhatItsReportNeeds) mustCommit(c.ops.op.replace(key, encodeMountLease(foreign), got->etag, Retry::standard()), "foreign slot"); c.backend->attempts.clear(); c.backend->on_attempt = [&boot_ms = c.boot_ms] { boot_ms += 250; }; - const MountRenewResult result = c.renewer.renew(renewalEnvironment(c.boot_ms)); + const MountRenewResult result = c.renewer.renew(renewalEnvironment()); ASSERT_EQ(result.outcome, MountRenewOutcome::Terminal); ASSERT_EQ(c.backend->attempts.size(), 1u); const MountLease sent = decodeMountLease(c.backend->attempts.back().bytes); @@ -1361,7 +1358,7 @@ TEST(CASHeartbeat, RenewReturnsWhatItsReportNeeds) const auto got = c.ops.op.read(key, Retry::standard()); ASSERT_TRUE(got.has_value()); ASSERT_EQ(c.ops.op.remove(key, got->etag, Retry::standard()), Removal::Removed); - const MountRenewResult result = c.renewer.renew(renewalEnvironment(c.boot_ms)); + const MountRenewResult result = c.renewer.renew(renewalEnvironment()); ASSERT_EQ(result.outcome, MountRenewOutcome::Terminal); EXPECT_EQ(result.classification, MountRenewTerminalClassification::Vanished); EXPECT_EQ(result.seq, 2u); @@ -1371,8 +1368,7 @@ TEST(CASHeartbeat, RenewReturnsWhatItsReportNeeds) { /// Ended before its first request by a liveness that refuses with no stop requested. ReportFieldsCase c("refused"); - const MountRenewResult result = c.renewer.renew(renewalEnvironment( - c.boot_ms, /*live=*/[] { return false; }, /*cancelled=*/[] { return false; })); + const MountRenewResult result = c.renewer.renew(renewalEnvironment(/*live=*/[] { return false; }, /*cancelled=*/[] { return false; })); ASSERT_EQ(result.outcome, MountRenewOutcome::Terminal); EXPECT_TRUE(c.backend->attempts.empty()); EXPECT_EQ(result.classification, MountRenewTerminalClassification::FenceOrLifecycleLost); @@ -1385,8 +1381,7 @@ TEST(CASHeartbeat, RenewReturnsWhatItsReportNeeds) { /// Ended before its first request by a stop. ReportFieldsCase c("stopped"); - const MountRenewResult result = c.renewer.renew(renewalEnvironment( - c.boot_ms, /*live=*/[] { return false; }, /*cancelled=*/[] { return true; })); + const MountRenewResult result = c.renewer.renew(renewalEnvironment(/*live=*/[] { return false; }, /*cancelled=*/[] { return true; })); ASSERT_EQ(result.outcome, MountRenewOutcome::NotAttempted); EXPECT_EQ(result.classification, MountRenewTerminalClassification::Cancelled); EXPECT_EQ(result.seq, 2u); @@ -1412,7 +1407,7 @@ TEST(CASHeartbeat, LateDeliveryAfterTerminalCannotRearmOrOverwriteSuccessor) /// Live until the resolving read has run: the delayed write lands during that read, and the /// renewal ends terminal without a second attempt. const MountRenewResult result = renewer.renew( - renewalEnvironment(boot_ms, /*live=*/[&] { return backend->read_calls == 0; })); + renewalEnvironment(/*live=*/[&] { return backend->read_calls == 0; })); EXPECT_EQ(result.outcome, MountRenewOutcome::Terminal); /// The delayed write landed during the resolving read. It carries this renewer's own epoch, and @@ -1442,7 +1437,7 @@ TEST(CASHeartbeat, LateDeliveryAfterTerminalCannotRearmOrOverwriteSuccessor) backend->actions = {RenewalScriptBackend::Action::ThrowBefore}; backend->read_calls = 0; const MountRenewResult result = renewer.renew( - renewalEnvironment(boot_ms, /*live=*/[&] { return backend->read_calls == 0; })); + renewalEnvironment(/*live=*/[&] { return backend->read_calls == 0; })); ASSERT_EQ(result.outcome, MountRenewOutcome::Terminal); ASSERT_FALSE(backend->attempts.empty()); const auto delayed = backend->attempts.back(); @@ -1485,13 +1480,13 @@ TEST(CASHeartbeat, WallClockStepsAndBootSuspendCannotExtendAuthority) renewer.start(); wall_ms = 9'000'000; - EXPECT_EQ(renewer.renew(renewalEnvironment(boot_ms)).outcome, MountRenewOutcome::Committed); + EXPECT_EQ(renewer.renew(renewalEnvironment()).outcome, MountRenewOutcome::Committed); wall_ms = 1; - EXPECT_EQ(renewer.renew(renewalEnvironment(boot_ms)).outcome, MountRenewOutcome::Committed); + EXPECT_EQ(renewer.renew(renewalEnvironment()).outcome, MountRenewOutcome::Committed); backend->attempts.clear(); boot_ms += 10'000; - const MountRenewResult suspended = renewer.renew(renewalEnvironment(boot_ms)); + const MountRenewResult suspended = renewer.renew(renewalEnvironment()); ASSERT_EQ(suspended.outcome, MountRenewOutcome::Committed); EXPECT_EQ(suspended.attempt_start_boot_ms, boot_ms) << "a renewal after a suspend anchors at its own start, never at the deadline it last confirmed"; @@ -1549,7 +1544,7 @@ class UnboundedRenewalFixture } return !live || live(); }; - MountRenewOperationEnvironment environment = renewalEnvironment(boot_ms, bounded_live, cancelled); + MountRenewOperationEnvironment environment = renewalEnvironment(bounded_live, cancelled); environment.on_request = std::move(on_request); MountRenewResult result = renewer->renew(environment); EXPECT_FALSE(request_bound_hit) << "the renewal sent " << max_requests diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp index 70be8e9bc275..d4599dfd6168 100644 --- a/src/Disks/tests/gtest_cas_pool.cpp +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -3365,14 +3365,14 @@ TEST(CASPoolRemount, DirectRenewIsRefusedForBackgroundConfiguredRuntimeAfterStop CasEventSink sink; RuntimeUnderTest runtime_holder( backend, layout, - MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::hours(1), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }}, "test", sink, runtimeRenewBudget(), [] { return false; }); CasMountRuntime & runtime = *runtime_holder; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); - runtime.startBackgroundWorkers(std::chrono::hours(1)); + runtime.startBackgroundWorkers(); runtime.stopBackgroundWorkers(); EXPECT_RUNTIME_STATE_REJECTION(runtime.renewWatermarkOnce()); runtime.finishTeardown(true); @@ -3396,7 +3396,7 @@ TEST(CASPoolRemount, ReclaimWaitsForTheRenewalToEnd) CasEventSink sink; RuntimeUnderTest runtime_holder( backend, layout, - MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::milliseconds(0), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }}, "test", sink, runtimeRenewBudget(), [&] { @@ -3413,7 +3413,7 @@ TEST(CASPoolRemount, ReclaimWaitsForTheRenewalToEnd) backend->fault = RuntimeRenewBackend::Fault::BlockThenDelegate; /// Set after the setup's own writes: only the held renewal request sets it. backend->after_commit = [&] { renewal_request_returned = true; }; - runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + runtime.startBackgroundWorkers(); renewal_barrier.waitUntilArrived(); runtime.tripMountLost(); runtime.scheduleRemount(); @@ -3448,14 +3448,14 @@ TEST(CASPoolRemount, TeardownJoinsTheLeaseThreadBeforeRelease) CasEventSink sink; RuntimeUnderTest runtime_holder( backend, layout, - MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::hours(1), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }, .worker_factory = factory}, "test", sink, runtimeRenewBudget(), [] { return false; }); CasMountRuntime & runtime = *runtime_holder; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); - runtime.startBackgroundWorkers(std::chrono::hours(1)); + runtime.startBackgroundWorkers(); runtime.stopBackgroundWorkers(); EXPECT_EQ(worker_exits.load(), 1u); runtime.finishTeardown(true); @@ -3548,7 +3548,7 @@ TEST(CASMountRuntime, MemoryLimitDoesNotEndTheLeaseThread) RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ - .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::milliseconds(0), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }, .renewal_admitted_hook_for_test = [&] @@ -3573,7 +3573,7 @@ TEST(CASMountRuntime, MemoryLimitDoesNotEndTheLeaseThread) const uint64_t leases_lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); backend->fault = RuntimeRenewBackend::Fault::ThrowMemoryLimitExceeded; - runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + runtime.startBackgroundWorkers(); reported.waitUntilArrived(); const String reported_outcome = outcome; @@ -3627,7 +3627,7 @@ TEST(CASPoolRemount, NaturalTerminalTransitionMakesTheLeaseThreadExit) RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ - .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::hours(1), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }, .worker_factory = factory}, @@ -3645,7 +3645,7 @@ TEST(CASPoolRemount, NaturalTerminalTransitionMakesTheLeaseThreadExit) runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); - runtime.startBackgroundWorkers(std::chrono::hours(1)); + runtime.startBackgroundWorkers(); runtime.tripMountLost(); runtime.scheduleRemount(); transitioned.waitUntilArrived(); @@ -3699,7 +3699,7 @@ TEST(CASPoolRemount, ALeaseWaitCannotMissNaturalTerminalPublication) RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ - .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::hours(1), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }, .worker_factory = factory, @@ -3743,7 +3743,7 @@ TEST(CASPoolRemount, ALeaseWaitCannotMissNaturalTerminalPublication) runtime.tripMountLost(); runtime.scheduleRemount(); } - runtime.startBackgroundWorkers(std::chrono::hours(1)); + runtime.startBackgroundWorkers(); waiter.waitUntilArrived(); if (terminal == PoolLifecycle::IdentityLost) @@ -3792,7 +3792,7 @@ TEST(CASPoolRemount, VanishedReasonPreparationFailureLeavesTerminalTransitionRet RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ - .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::hours(1), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }, .worker_factory = factory, @@ -3806,7 +3806,7 @@ TEST(CASPoolRemount, VanishedReasonPreparationFailureLeavesTerminalTransitionRet runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); - runtime.startBackgroundWorkers(std::chrono::hours(1)); + runtime.startBackgroundWorkers(); runtime.tripMountLost(); expectThrowsCode(DB::ErrorCodes::NETWORK_ERROR, [&] @@ -3846,14 +3846,14 @@ TEST(CASPoolRemount, WorkerConstructionRollbackFailsOpenClosed) CasEventSink sink; RuntimeUnderTest runtime_holder( backend, layout, - MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::milliseconds(10), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }, .worker_factory = factory}, "test", sink, runtimeRenewBudget(), [] { return false; }); CasMountRuntime & runtime = *runtime_holder; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); - EXPECT_THROW(runtime.startBackgroundWorkers(std::chrono::milliseconds(10)), DB::Exception); + EXPECT_THROW(runtime.startBackgroundWorkers(), DB::Exception); EXPECT_FALSE(runtime.mayMutate()); EXPECT_FALSE(runtime.workersRunningForTest()); runtime.finishTeardown(false); @@ -3877,7 +3877,7 @@ TEST(CASPoolRemount, ExternalLossDuringRenewalUsesOneRecoveryGeneration) CasMountRuntime * runtime_ptr = nullptr; RuntimeUnderTest runtime_holder( backend, layout, - MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::milliseconds(0), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }}, "test", sink, runtimeRenewBudget(), [&] { @@ -3904,7 +3904,7 @@ TEST(CASPoolRemount, ExternalLossDuringRenewalUsesOneRecoveryGeneration) backend->barrier = &renewal_barrier; backend->fault = RuntimeRenewBackend::Fault::BlockThenDelegate; const auto lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); - runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + runtime.startBackgroundWorkers(); renewal_barrier.waitUntilArrived(); runtime.tripMountLost(); runtime.scheduleRemount(); @@ -3933,7 +3933,7 @@ TEST(CASPoolRemount, ConcurrentRemountRequestIsProcessedAfterActiveGeneration) CasEventSink sink; RuntimeUnderTest runtime_holder( backend, layout, - MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::hours(1), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }}, "test", sink, runtimeRenewBudget(), [&] { @@ -3945,7 +3945,7 @@ TEST(CASPoolRemount, ConcurrentRemountRequestIsProcessedAfterActiveGeneration) runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); - runtime.startBackgroundWorkers(std::chrono::hours(1)); + runtime.startBackgroundWorkers(); runtime.tripMountLost(); runtime.scheduleRemount(); first.waitUntilArrived(); @@ -3976,7 +3976,7 @@ TEST(CASPoolRemount, ImmediatePostRemountRenewalFailureIsNotDropped) CasMountRuntime * runtime_ptr = nullptr; RuntimeUnderTest runtime_holder( backend, layout, - MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(10'000), .background_watermark = true, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(10'000), .mount_renew_period = std::chrono::milliseconds(1000), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }}, "test", sink, runtimeRenewBudget(), [&] { @@ -4007,7 +4007,7 @@ TEST(CASPoolRemount, ImmediatePostRemountRenewalFailureIsNotDropped) runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 10'000); - runtime.startBackgroundWorkers(std::chrono::milliseconds(1000)); + runtime.startBackgroundWorkers(); runtime.tripMountLost(); runtime.scheduleRemount(); first.waitUntilArrived(); @@ -4065,7 +4065,8 @@ TEST(CASPoolShutdown, PreSendCancellationAllowsFarewellButAmbiguityDoesNot) CasEventSink sink; RuntimeUnderTest runtime_holder( backend, layout, - MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .mount_renew_period = std::chrono::milliseconds(ambiguous ? 0 : 3'600'000), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }}, "test", sink, runtimeRenewBudget(), [] { return false; }); CasMountRuntime & runtime = *runtime_holder; @@ -4076,7 +4077,7 @@ TEST(CASPoolShutdown, PreSendCancellationAllowsFarewellButAmbiguityDoesNot) { backend->barrier = &barrier; backend->fault = RuntimeRenewBackend::Fault::BlockThenThrow; - runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + runtime.startBackgroundWorkers(); barrier.waitUntilArrived(); auto stop = std::async(std::launch::async, [&] { runtime.stopBackgroundWorkers(); }); barrier.release(); @@ -4084,7 +4085,7 @@ TEST(CASPoolShutdown, PreSendCancellationAllowsFarewellButAmbiguityDoesNot) } else { - runtime.startBackgroundWorkers(std::chrono::hours(1)); + runtime.startBackgroundWorkers(); runtime.stopBackgroundWorkers(); } runtime.finishTeardown(true); @@ -4205,7 +4206,7 @@ TEST(CASPool, DeterministicWorkerFailureFencesWithoutWaitingForCadence) CasEventSink sink; RuntimeUnderTest runtime_holder( backend, layout, - MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::milliseconds(0), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }}, "test", sink, runtimeRenewBudget(), [&] { @@ -4218,7 +4219,7 @@ TEST(CASPool, DeterministicWorkerFailureFencesWithoutWaitingForCadence) runtime.armMountFence(uuid, 1, anchor + 1000); /// A definitive answer ends the lease thread's renewal; a transient fault would only be retried. fenceOutMount(*backend, layout.mountKey("test")); - runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + runtime.startBackgroundWorkers(); remount_entered.waitUntilArrived(); EXPECT_FALSE(runtime.mayMutate()); EXPECT_EQ(runtime.lifecycle(), PoolLifecycle::TransientNotLive); @@ -4255,7 +4256,7 @@ TEST(CASMountRuntime, ForgetEndsAnUnboundedRenewal) CasEventSink sink; RuntimeUnderTest runtime_holder( backend, layout, - MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::milliseconds(0), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms.load(); }, .worker_factory = factory}, "test", sink, runtimeRenewBudget(), [] { return false; }); CasMountRuntime & runtime = *runtime_holder; @@ -4271,7 +4272,7 @@ TEST(CASMountRuntime, ForgetEndsAnUnboundedRenewal) holding.arriveAndWait(); }); backend->outage = [] { return true; }; - runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + runtime.startBackgroundWorkers(); holding.waitUntilArrived(); const uint64_t writes_at_forget = backend->outage_writes.load(); @@ -4310,7 +4311,7 @@ TEST(CASMountRuntime, StopWakesTheRetryWaitOfAnUnboundedRenewal) CasEventSink sink; RuntimeUnderTest runtime_holder( backend, layout, - MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::milliseconds(0), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }}, "test", sink, runtimeRenewBudget(), [] { return false; }); CasMountRuntime & runtime = *runtime_holder; @@ -4330,7 +4331,7 @@ TEST(CASMountRuntime, StopWakesTheRetryWaitOfAnUnboundedRenewal) }); backend->outage = [] { return true; }; const uint64_t lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); - runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + runtime.startBackgroundWorkers(); ASSERT_EQ(wait_requested.wait_for(std::chrono::seconds(20)), std::future_status::ready); const uint64_t requested_ms = wait_requested.get(); @@ -4372,7 +4373,7 @@ TEST(CASMountRuntime, AStaleSuccessIsFollowedAtOnceByTheNextRenewal) CasEventSink sink; RuntimeUnderTest runtime_holder( backend, layout, - MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::milliseconds(500), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms.load(); }, /// The test clock moves only through the sleep seam, so a regression that zeroes the /// pauses would retry forever; the request count ends the renewal instead. @@ -4401,7 +4402,7 @@ TEST(CASMountRuntime, AStaleSuccessIsFollowedAtOnceByTheNextRenewal) if (commit_boot_ms.size() == 2) second_commit.arriveAndWait(); }; - runtime.startBackgroundWorkers(std::chrono::milliseconds(500)); + runtime.startBackgroundWorkers(); bool arrived = true; try @@ -4459,7 +4460,7 @@ TEST(CASMountRuntime, NothingArmsWhileARemountRequestIsPending) CasEventSink sink; RuntimeUnderTest runtime_holder( backend, layout, - MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::hours(1), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }}, "test", sink, runtimeRenewBudget(), [&] { @@ -4495,7 +4496,7 @@ TEST(CASMountRuntime, NothingArmsWhileARemountRequestIsPending) const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); const uint64_t lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); - runtime.startBackgroundWorkers(std::chrono::hours(1)); + runtime.startBackgroundWorkers(); /// A request without a trip: the fence is still armed when the reclaim starts. runtime.scheduleRemount(); @@ -4545,7 +4546,7 @@ TEST(CASMountRuntime, AnInterferenceReportDuringAReclaimIsServedByTheNextReclaim CasEventSink sink; RuntimeUnderTest runtime_holder( backend, layout, - MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::hours(1), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }}, "test", sink, runtimeRenewBudget(), [&] { @@ -4601,7 +4602,7 @@ TEST(CASMountRuntime, AnInterferenceReportDuringAReclaimIsServedByTheNextReclaim const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); const uint64_t lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); - runtime.startBackgroundWorkers(std::chrono::hours(1)); + runtime.startBackgroundWorkers(); runtime.scheduleRemount(); second_done.waitUntilArrived(); @@ -4646,7 +4647,7 @@ TEST(CASMountRuntime, AReclaimAcknowledgesOnlyTheGenerationItServed) CasEventSink sink; RuntimeUnderTest runtime_holder( backend, layout, - MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::hours(1), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }}, "test", sink, runtimeRenewBudget(), [&] { @@ -4675,7 +4676,7 @@ TEST(CASMountRuntime, AReclaimAcknowledgesOnlyTheGenerationItServed) const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); const uint64_t lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); - runtime.startBackgroundWorkers(std::chrono::hours(1)); + runtime.startBackgroundWorkers(); /// One interference report. runtime.tripMountLost(); runtime.scheduleRemount(); @@ -4722,7 +4723,7 @@ TEST(CASMountRuntime, AReclaimFinishedAfterTheForgetIntentArmsNothing) CasEventSink sink; RuntimeUnderTest runtime_holder( backend, layout, - MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::hours(1), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }, .worker_factory = factory}, "test", sink, runtimeRenewBudget(), [&] { @@ -4739,7 +4740,7 @@ TEST(CASMountRuntime, AReclaimFinishedAfterTheForgetIntentArmsNothing) runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); - runtime.startBackgroundWorkers(std::chrono::hours(1)); + runtime.startBackgroundWorkers(); runtime.tripMountLost(); runtime.scheduleRemount(); @@ -4789,7 +4790,7 @@ TEST(CASMountRuntime, StopDuringAReclaimJoinsTheThread) CasEventSink sink; RuntimeUnderTest runtime_holder( backend, layout, - MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::milliseconds(0), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }, .worker_factory = factory}, "test", sink, runtimeRenewBudget(), [&] { @@ -4814,7 +4815,7 @@ TEST(CASMountRuntime, StopDuringAReclaimJoinsTheThread) /// A definitive answer ends the first renewal and requests the reclaim; nobody claims the slot back. fenceOutMount(*backend, key); const uint64_t lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); - runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + runtime.startBackgroundWorkers(); reclaim_entered.waitUntilArrived(); stop_requested_by_test = true; @@ -4890,7 +4891,7 @@ TEST(CASMountRuntime, ARemountRequestEndsTheRenewalAndTheSameThreadReclaims) CasEventSink sink; RuntimeUnderTest runtime_holder( backend, layout, - MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::milliseconds(0), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms.load(); }, .worker_factory = factory}, "test", sink, runtimeRenewBudget(), [&] { @@ -4942,7 +4943,7 @@ TEST(CASMountRuntime, ARemountRequestEndsTheRenewalAndTheSameThreadReclaims) renewed.arriveAndWait(); } }; - runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + runtime.startBackgroundWorkers(); holding.waitUntilArrived(); const uint64_t writes_at_request = backend->outage_writes.load(); @@ -5023,7 +5024,7 @@ TEST(CASMountRuntime, ACommitConsumedAfterARemountRequestCannotOverwriteTheRecla RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ - .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::milliseconds(0), .background_watermark = true, .boot_ms_fn = [&] { @@ -5068,7 +5069,7 @@ TEST(CASMountRuntime, ACommitConsumedAfterARemountRequestCannotOverwriteTheRecla runtime.armMountFence(uuid, 1, anchor + 1000); /// Set after the setup's own writes: only the loop's renewal may raise the request. backend->after_commit = [&] { committed = true; }; - runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + runtime.startBackgroundWorkers(); second_admission.waitUntilArrived(); EXPECT_FALSE(hold_timed_out.load()); @@ -5102,7 +5103,7 @@ TEST(CASMountRuntimeDeathTest, OnlyTheLeaseThreadReplacesTheRenewerWhileItRuns) CasEventSink sink; RuntimeUnderTest runtime_holder( backend, layout, - MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .background_watermark = true, + MountConfig{.mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::milliseconds(0), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }, .renewal_admitted_hook_for_test = [&] { admitted.arriveAndWait(); }}, "test", sink, runtimeRenewBudget(), [] { return false; }); @@ -5111,7 +5112,7 @@ TEST(CASMountRuntimeDeathTest, OnlyTheLeaseThreadReplacesTheRenewerWhileItRuns) runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + 1000); - runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + runtime.startBackgroundWorkers(); /// The lease thread is held between its decision to renew and the renewal, with no lock held. admitted.waitUntilArrived(); @@ -5488,7 +5489,7 @@ void runExpiryScenario(const String & layout_prefix, ExpiryObservation & seen) RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ - .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), + .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), .mount_renew_period = std::chrono::milliseconds(kExpiryPeriodMs), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms.load(); }, .renewal_before_driver_lock_hook_for_test = [&] @@ -5550,7 +5551,7 @@ void runExpiryScenario(const String & layout_prefix, ExpiryObservation & seen) return true; }; - runtime.startBackgroundWorkers(std::chrono::milliseconds(kExpiryPeriodMs)); + runtime.startBackgroundWorkers(); third_pass.waitUntilArrived(); seen.writer_epoch_on_store = decodeMountLease(readObj(*backend, layout.mountKey("test"))->bytes).writer_epoch; @@ -5601,7 +5602,7 @@ void runReadOutageScenario(const String & layout_prefix, ReadOutageObservation & RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ - .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), + .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), .mount_renew_period = std::chrono::milliseconds(kExpiryPeriodMs), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms.load(); }, .renewal_before_driver_lock_hook_for_test = [&] @@ -5643,7 +5644,7 @@ void runReadOutageScenario(const String & layout_prefix, ReadOutageObservation & return true; }; - runtime.startBackgroundWorkers(std::chrono::milliseconds(kExpiryPeriodMs)); + runtime.startBackgroundWorkers(); second_pass.waitUntilArrived(); second_pass.release(); runtime.stopBackgroundWorkers(); @@ -5802,7 +5803,7 @@ void runExpiryLogScenario(const String & layout_prefix, ExpiryLogRig & rig, uint RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ - .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), + .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), .mount_renew_period = std::chrono::milliseconds(kExpiryPeriodMs), .background_watermark = true, .boot_ms_fn = [&] { return rig.boot_ms.load(); }, .renewal_before_driver_lock_hook_for_test = [&] @@ -5839,7 +5840,7 @@ void runExpiryLogScenario(const String & layout_prefix, ExpiryLogRig & rig, uint return true; }; - runtime.startBackgroundWorkers(std::chrono::milliseconds(kExpiryPeriodMs)); + runtime.startBackgroundWorkers(); stop.waitUntilArrived(); rig.final_log = rig.log.captured(); @@ -5963,7 +5964,7 @@ TEST(CASMountRuntime, ASuccessEndsTheFailureTextOfItsRun) RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ - .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), + .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), .mount_renew_period = std::chrono::milliseconds(kExpiryPeriodMs), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms.load(); }, .renewal_before_driver_lock_hook_for_test = [&] @@ -6004,7 +6005,7 @@ TEST(CASMountRuntime, ASuccessEndsTheFailureTextOfItsRun) return false; }; - runtime.startBackgroundWorkers(std::chrono::milliseconds(kExpiryPeriodMs)); + runtime.startBackgroundWorkers(); third_pass.waitUntilArrived(); third_pass.release(); runtime.stopBackgroundWorkers(); @@ -6042,7 +6043,7 @@ TEST(CASMountRuntime, EachRestoredExpiryIsCountedOnce) RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ - .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), + .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), .mount_renew_period = std::chrono::milliseconds(kExpiryPeriodMs), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms.load(); }, .renewal_before_driver_lock_hook_for_test = [&] @@ -6097,7 +6098,7 @@ TEST(CASMountRuntime, EachRestoredExpiryIsCountedOnce) return failing; }; - runtime.startBackgroundWorkers(std::chrono::milliseconds(kExpiryPeriodMs)); + runtime.startBackgroundWorkers(); last_pass.waitUntilArrived(); last_pass.release(); runtime.stopBackgroundWorkers(); @@ -6129,7 +6130,7 @@ TEST(CASMountRuntime, AnExpiryEndedByAFenceIsNotCountedAsARestore) RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ - .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), + .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), .mount_renew_period = std::chrono::milliseconds(kExpiryPeriodMs), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms.load(); }, .renewal_before_driver_lock_hook_for_test = [&] @@ -6159,7 +6160,7 @@ TEST(CASMountRuntime, AnExpiryEndedByAFenceIsNotCountedAsARestore) return failing; }; - runtime.startBackgroundWorkers(std::chrono::milliseconds(kExpiryPeriodMs)); + runtime.startBackgroundWorkers(); remount_entered.waitUntilArrived(); EXPECT_EQ(runtime.lifecycle(), PoolLifecycle::TransientNotLive); @@ -6241,7 +6242,7 @@ TEST(CASMountRuntime, AReadinessRenewalArmsOnlyWithRoomForARefAppend) RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ - .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), + .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), .mount_renew_period = std::chrono::milliseconds(kExpiryPeriodMs), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms.load(); }, .renewal_before_driver_lock_hook_for_test = [&] @@ -6303,7 +6304,7 @@ TEST(CASMountRuntime, AReadinessRenewalArmsOnlyWithRoomForARefAppend) boot_ms.store(kReadinessShortCommitBootMs); return false; }; - runtime.startBackgroundWorkers(std::chrono::milliseconds(kExpiryPeriodMs)); + runtime.startBackgroundWorkers(); bool arrived = true; try @@ -6386,7 +6387,7 @@ TEST(CASMountRuntime, ARenewalIsSentUnderALatchedFence) RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ - .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), + .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), .mount_renew_period = std::chrono::milliseconds(kExpiryPeriodMs), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms.load(); }, .renewal_before_driver_lock_hook_for_test = [&] @@ -6429,7 +6430,7 @@ TEST(CASMountRuntime, ARenewalIsSentUnderALatchedFence) } return false; }; - runtime.startBackgroundWorkers(std::chrono::milliseconds(kExpiryPeriodMs)); + runtime.startBackgroundWorkers(); third_put.waitUntilArrived(); runtime.tripMountLost(); @@ -6494,7 +6495,7 @@ TEST(CASMountRuntime, TheForgetIntentAloneEndsARenewal) RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ - .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), + .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), .mount_renew_period = std::chrono::milliseconds(kExpiryPeriodMs), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms.load(); }, .worker_factory = factory, @@ -6529,7 +6530,7 @@ TEST(CASMountRuntime, TheForgetIntentAloneEndsARenewal) boot_ms.fetch_add(kExpiryFailedPutMs); return true; }; - runtime.startBackgroundWorkers(std::chrono::milliseconds(kExpiryPeriodMs)); + runtime.startBackgroundWorkers(); third_put.waitUntilArrived(); const uint64_t generation_at_intent = runtime.remountRequestedGenerationForTest(); @@ -6573,7 +6574,7 @@ TEST(CASMountRuntime, ATripAloneIsRearmedByTheNextRenewal) RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ - .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), + .mount_lease_ttl_ms = std::chrono::milliseconds(kExpiryTtlMs), .mount_renew_period = std::chrono::milliseconds(kExpiryPeriodMs), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms.load(); }, .renewal_before_driver_lock_hook_for_test = [&] @@ -6601,7 +6602,7 @@ TEST(CASMountRuntime, ATripAloneIsRearmedByTheNextRenewal) runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); runtime.armMountFence(uuid, 1, anchor + kExpiryTtlMs); - runtime.startBackgroundWorkers(std::chrono::milliseconds(kExpiryPeriodMs)); + runtime.startBackgroundWorkers(); second_pass.waitUntilArrived(); EXPECT_STREQ(after_renewal.admit, "Ok") << "the renewal at 110 s re-armed the fence"; @@ -6640,7 +6641,7 @@ TEST(CASMountRuntime, ARenewalArmsNothingWhileARequestIsPending) RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ - .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::milliseconds(0), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms; }, .renewal_live_for_test = [&] @@ -6670,7 +6671,7 @@ TEST(CASMountRuntime, ARenewalArmsNothingWhileARequestIsPending) const uint64_t writes_before = backend->putOverwriteCount(key); const uint64_t requests_before = runtime.scheduleRemountCallCountForTest(); backend->after_commit = [&] { renewal_landed = true; }; - runtime.startBackgroundWorkers(std::chrono::milliseconds(0)); + runtime.startBackgroundWorkers(); reclaim_entered.waitUntilArrived(); EXPECT_TRUE(request_raised); @@ -7042,7 +7043,7 @@ void runDefinitiveAnswerDuringReadiness(bool reclaim_arms, DefinitiveReadinessSe RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ - .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::milliseconds(300), .background_watermark = true, .boot_ms_fn = [&] { return boot_ms.load(); }}, "test", sink, runtimeRenewBudget(), [&] @@ -7091,7 +7092,7 @@ void runDefinitiveAnswerDuringReadiness(bool reclaim_arms, DefinitiveReadinessSe } return false; }; - runtime.startBackgroundWorkers(std::chrono::milliseconds(300)); + runtime.startBackgroundWorkers(); seen.boot_before_wait = boot_ms.load(); seen.armed = runtime.waitUntilArmed(1000); @@ -7167,7 +7168,7 @@ void runLeaseThreadEndsOnItsOwn(LeaseThreadEndSeen & seen) RuntimeUnderTest runtime_holder( backend, layout, MountConfig{ - .mount_lease_ttl_ms = std::chrono::milliseconds(1000), + .mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::milliseconds(300), .background_watermark = true, .boot_ms_fn = [&] { return thread_gone.load() ? boot_ms.fetch_add(600) + 600 : boot_ms.load(); }, .worker_factory = factory}, @@ -7178,7 +7179,7 @@ void runLeaseThreadEndsOnItsOwn(LeaseThreadEndSeen & seen) boot_ms = 1070; ASSERT_FALSE(runtime.armIfAdmissible(anchor + 1000)); backend->failing = true; - runtime.startBackgroundWorkers(std::chrono::milliseconds(300)); + runtime.startBackgroundWorkers(); ASSERT_TRUE(exits.waitForAtLeast(1)) << "the loop must leave through its own error path"; thread_gone = true; From a9a02fc44be5587382815a1c0b235db029d54608 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 10:01:50 +0200 Subject: [PATCH 28/37] Derive allow_steal from the round trigger in CasGcScheduler `runRoundLogged` took both a trigger and `allow_steal`, and the two always agreed: scheduled rounds steal, manual rounds never do. `runOneRoundNow` always ran a manual round, so it loses its parameter. Co-Authored-By: Claude Sonnet 5.5 --- .../ContentAddressedMetadataStorage.cpp | 2 +- .../ContentAddressed/Gc/CasGcScheduler.cpp | 14 ++++---- .../ContentAddressed/Gc/CasGcScheduler.h | 12 ++++--- .../tests/gtest_cas_gc_arithmetic_intake.cpp | 2 +- src/Disks/tests/gtest_cas_gc_log.cpp | 34 +++++++++---------- .../tests/gtest_cas_gc_teardown_stop.cpp | 2 +- 6 files changed, 33 insertions(+), 33 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.cpp index 24b7c95f0bc5..98b7bdb98f9f 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedMetadataStorage.cpp @@ -657,7 +657,7 @@ Cas::RoundReport ContentAddressedMetadataStorage::runGarbageCollectionRoundNow() } snapshot = gc_scheduler; } - return snapshot->runOneRoundNow(Cas::GcRoundLogRecord::Trigger::Manual); + return snapshot->runOneRoundNow(); } Cas::RebuildReport ContentAddressedMetadataStorage::runGcRebuildNow(bool force) const diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.cpp index 46c9b8ba1f3b..518c52bddbe7 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.cpp @@ -190,9 +190,10 @@ void CasGcScheduler::onLeaseAcquired() } Cas::RoundReport CasGcScheduler::runRoundLogged(Cas::Gc & round_gc, GcRoundLogRecord::Trigger trigger, - std::function on_lease_acquired, bool allow_steal) + std::function on_lease_acquired) { using Rec = GcRoundLogRecord; + const bool allow_steal = trigger == Rec::Trigger::Scheduled; /// Mark a round in flight for the whole body (success AND exception paths). `isQuiescent` reads this; /// the FORGET / `GC STOP` tests use it to prove the scheduler's workers were joined (no round can be @@ -327,13 +328,10 @@ Cas::RoundReport CasGcScheduler::runRoundLogged(Cas::Gc & round_gc, GcRoundLogRe } } -Cas::RoundReport CasGcScheduler::runOneRoundNow(GcRoundLogRecord::Trigger trigger) +Cas::RoundReport CasGcScheduler::runOneRoundNow() { std::lock_guard round_lock(gc_round_mutex); - /// allow_steal=false: a manual round may acquire a FREE lease or renew ITS OWN, but must never - /// steal a live incumbent — see Cas::Gc::runRegularRound's doc comment. Dead-incumbent recovery - /// stays the loop's job (bounded ~2*interval; loop() below passes the default allow_steal=true). - const Cas::RoundReport report = runRoundLogged(gc, trigger, [this] { onLeaseAcquired(); }, /*allow_steal=*/false); + const Cas::RoundReport report = runRoundLogged(gc, GcRoundLogRecord::Trigger::Manual, [this] { onLeaseAcquired(); }); i_am_leader.store(report.acquired_lease, std::memory_order_relaxed); return report; } @@ -398,8 +396,8 @@ void CasGcScheduler::loop() /// lease is (re)acquired, before the fold runs - a new leader's first round is otherwise /// unprotected (i_am_leader would only flip below, AFTER the whole round returns), so a /// follower observing the frozen (owner, seq) across two of its own ticks would steal - /// deterministically once that first round outlasts them. allow_steal defaults to true here - /// (the loop is the ONLY caller allowed to execute the steal CAS). + /// deterministically once that first round outlasts them. A `Scheduled` round is the only + /// one that may execute the steal CAS. const Cas::RoundReport report = runRoundLogged(gc, GcRoundLogRecord::Trigger::Scheduled, [this] { onLeaseAcquired(); }); i_am_leader.store(report.acquired_lease, std::memory_order_relaxed); if (report.acquired_lease) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.h index e4178f42965a..d7501468bdb8 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.h @@ -134,9 +134,10 @@ class CasGcScheduler /// scheduled-round authority. Requests coalesce into one boolean while a round is pending. void requestRoundSoon(); - /// Test/diagnostics hook: run ONE round synchronously on the caller's thread. Returns the round - /// report so the SYSTEM command / tests can inspect it. Emits a Start + Finish record. - Cas::RoundReport runOneRoundNow(GcRoundLogRecord::Trigger trigger = GcRoundLogRecord::Trigger::Manual); + /// Run ONE manual round synchronously on the caller's thread. Returns the round report; emits a + /// Start + Finish record. A manual round acquires a free lease or renews its own but never steals a + /// live incumbent's: dead-incumbent recovery stays the loop's job. + Cas::RoundReport runOneRoundNow(); /// Returns per-disk GC health for `system.cas_mounts`. The fields describing /// rounds snapshot this scheduler's state, while `wedged_namespace_count` is read live from the @@ -199,9 +200,10 @@ class CasGcScheduler /// Run one round through the full logging path (Start record, ProfileEventsScope, Finish /// record). Used by BOTH loop() and runOneRoundNow. Logging is best-effort - the logger sink /// never throws into the round. Rethrows a round exception (after emitting an Aborted Finish). - /// `allow_steal` is forwarded to `Cas::Gc::runRegularRound` verbatim (see its doc comment). + /// A `Scheduled` round may steal an incumbent's lease; a `Manual` one never does (see + /// `Cas::Gc::runRegularRound`). Cas::RoundReport runRoundLogged(Cas::Gc & round_gc, GcRoundLogRecord::Trigger trigger, - std::function on_lease_acquired = {}, bool allow_steal = true); + std::function on_lease_acquired = {}); const Cas::PoolPtr store; const std::chrono::seconds interval; diff --git a/src/Disks/tests/gtest_cas_gc_arithmetic_intake.cpp b/src/Disks/tests/gtest_cas_gc_arithmetic_intake.cpp index d583b31df256..c4f2c5aeae22 100644 --- a/src/Disks/tests/gtest_cas_gc_arithmetic_intake.cpp +++ b/src/Disks/tests/gtest_cas_gc_arithmetic_intake.cpp @@ -91,7 +91,7 @@ std::map runRoundAndReadIntakeMetrics(const PoolPtr & store) std::vector rows; CasGcScheduler sched(store, std::chrono::seconds(1), "test::gc", "ca", [&](const GcRoundLogRecord & r) { rows.push_back(r); }); - EXPECT_TRUE(sched.runOneRoundNow(GcRoundLogRecord::Trigger::Manual).acquired_lease); + EXPECT_TRUE(sched.runOneRoundNow().acquired_lease); for (const GcRoundLogRecord & r : rows) if (r.event_type == GcRoundLogRecord::EventType::Phase && r.phase == "fold_ref_intake") return r.phase_metrics; diff --git a/src/Disks/tests/gtest_cas_gc_log.cpp b/src/Disks/tests/gtest_cas_gc_log.cpp index d70cc9aa78d8..d0c384fff88a 100644 --- a/src/Disks/tests/gtest_cas_gc_log.cpp +++ b/src/Disks/tests/gtest_cas_gc_log.cpp @@ -124,7 +124,7 @@ TEST(CASGCLog, EmitsStartFinishWithCounts) for (size_t round = 0; round < max_rounds; ++round) { const size_t before = rows.size(); - sched.runOneRoundNow(Rec::Trigger::Manual); + sched.runOneRoundNow(); store->renewWatermarkOnce(); /// Each call emits exactly one Start (first) and one Finish (last), with the round's phase rows @@ -240,11 +240,11 @@ TEST(CASGCSchedulerSteal, ManualRoundNeverStealsEvenADeadIncumbent) DB::Cas::CasGcScheduler sched(store, std::chrono::seconds(1), "test::gc", "ca"); /// obs #1: records the incumbent's (owner, seq, hb=absent). - EXPECT_FALSE(sched.runOneRoundNow(Rec::Trigger::Manual).acquired_lease); + EXPECT_FALSE(sched.runOneRoundNow().acquired_lease); /// obs #2 and #3: the same frozen (owner, seq, hb) observed repeatedly would be steal-eligible on /// the loop path (see the Core-level test this mirrors), but the manual path keeps backing off. - EXPECT_FALSE(sched.runOneRoundNow(Rec::Trigger::Manual).acquired_lease); - EXPECT_FALSE(sched.runOneRoundNow(Rec::Trigger::Manual).acquired_lease); + EXPECT_FALSE(sched.runOneRoundNow().acquired_lease); + EXPECT_FALSE(sched.runOneRoundNow().acquired_lease); } /// Negative-control companion to the test above (reviewer-requested): with the incumbent visibly alive @@ -265,12 +265,12 @@ TEST(CASGCSchedulerSteal, ManualRoundNeverStealsALiveHeartbeatingIncumbent) DB::Cas::CasGcScheduler sched(store, std::chrono::seconds(1), "test::gc", "ca"); /// obs #1: records (owner=incumbent, seq, hb=absent). - EXPECT_FALSE(sched.runOneRoundNow(Rec::Trigger::Manual).acquired_lease); + EXPECT_FALSE(sched.runOneRoundNow().acquired_lease); Gc::pulseHeartbeat(*store, kIncumbent); /// the incumbent is alive and pulsing (hb 0->1) /// obs #2: hb advanced since obs #1 => alive => no steal (never reaches the observe-only branch). - EXPECT_FALSE(sched.runOneRoundNow(Rec::Trigger::Manual).acquired_lease); + EXPECT_FALSE(sched.runOneRoundNow().acquired_lease); Gc::pulseHeartbeat(*store, kIncumbent); /// hb 1->2 - EXPECT_FALSE(sched.runOneRoundNow(Rec::Trigger::Manual).acquired_lease); + EXPECT_FALSE(sched.runOneRoundNow().acquired_lease); } /// A round whose backend throws must produce a Finish with `outcome == Aborted` and a non-empty @@ -288,7 +288,7 @@ TEST(CASGCLog, AbortedFinishOnThrowingRound) backend->arm.store(true); - EXPECT_THROW(sched.runOneRoundNow(Rec::Trigger::Manual), DB::Exception); + EXPECT_THROW(sched.runOneRoundNow(), DB::Exception); /// A throwing round still emits a Start and an Aborted Finish. It also emits the phase row of the /// phase it died in -- the timer is RAII, so it fires during unwinding, which is exactly the forensic @@ -349,7 +349,7 @@ TEST(CASGCLog, TransientThrowIsClassifiedAborted) [&](const Rec & r) { rows.push_back(r); }); backend->arm.store(true); - EXPECT_THROW(sched.runOneRoundNow(Rec::Trigger::Manual), DB::Exception); + EXPECT_THROW(sched.runOneRoundNow(), DB::Exception); const std::vector round_rows = roundRowsOnly(rows); ASSERT_EQ(round_rows.size(), 2u); @@ -426,7 +426,7 @@ TEST(CASGCLog, AbortedFinishCarriesProgressiveCounters) [&](const Rec & r) { rows.push_back(r); }); backend->arm.store(true); - EXPECT_THROW(sched.runOneRoundNow(Rec::Trigger::Manual), DB::Exception); + EXPECT_THROW(sched.runOneRoundNow(), DB::Exception); const std::vector round_rows = roundRowsOnly(rows); ASSERT_EQ(round_rows.size(), 2u); @@ -576,7 +576,7 @@ TEST(CASGCLog, EveryRowOfARoundSharesOneRoundId) store, std::chrono::seconds(1), "test::gc", "ca", [&](const Rec & r) { rows.push_back(r); }); - sched.runOneRoundNow(Rec::Trigger::Manual); + sched.runOneRoundNow(); const size_t after_first = rows.size(); ASSERT_GE(after_first, 2u); const String first_id = rows.front().round_id; @@ -585,7 +585,7 @@ TEST(CASGCLog, EveryRowOfARoundSharesOneRoundId) EXPECT_EQ(rows[i].round_id, first_id) << "row " << i << " of the first round has a different round_id"; store->renewWatermarkOnce(); - sched.runOneRoundNow(Rec::Trigger::Manual); + sched.runOneRoundNow(); ASSERT_GT(rows.size(), after_first); const String second_id = rows[after_first].round_id; EXPECT_FALSE(second_id.empty()); @@ -639,7 +639,7 @@ TEST(CASGCLog, FoldingRoundEmitsEveryPhaseInOrder) store, std::chrono::seconds(1), "test::gc", "ca", [&](const Rec & r) { rows.push_back(r); }); - ASSERT_TRUE(sched.runOneRoundNow(Rec::Trigger::Manual).acquired_lease); + ASSERT_TRUE(sched.runOneRoundNow().acquired_lease); const std::vector expected = { "lease", "pre_fold_ref_drain", "heartbeat_floor", "defer_decision", "parent_seal_read", @@ -703,7 +703,7 @@ TEST(CASGCLog, NotALeaderRoundEmitsOnlyTheLeasePhase) DB::Cas::CasGcScheduler sched( store, std::chrono::seconds(1), "test::gc", "ca", [&](const Rec & r) { rows.push_back(r); }); - EXPECT_FALSE(sched.runOneRoundNow(Rec::Trigger::Manual).acquired_lease); + EXPECT_FALSE(sched.runOneRoundNow().acquired_lease); EXPECT_EQ(phaseNames(rows, 0), (std::vector{"lease"})); EXPECT_EQ(metricsOf(rows, 0, "lease").at("acquired"), 0u); @@ -734,7 +734,7 @@ TEST(CASGCHealth, ReflectsLeadershipAndPendingReclaim) EXPECT_EQ(h0.pending_reclaim, 0); EXPECT_EQ(h0.wedged_namespace_count, 0u); - const RoundReport rep = sched.runOneRoundNow(Rec::Trigger::Manual); + const RoundReport rep = sched.runOneRoundNow(); ASSERT_TRUE(rep.acquired_lease); const auto h1 = sched.gcHealth(); @@ -785,7 +785,7 @@ TEST(CASGCLog, TransientFailureAfterTeardownBeganIsStopped) backend->arm_key = store->layout().gcStateKey(); backend->on_read = [&store] { store->beginTeardown(); }; - EXPECT_THROW(sched.runOneRoundNow(Rec::Trigger::Manual), DB::Exception); + EXPECT_THROW(sched.runOneRoundNow(), DB::Exception); const std::vector round_rows = roundRowsOnly(rows); ASSERT_EQ(round_rows.size(), 2u); @@ -820,7 +820,7 @@ TEST(CASGCLog, NonTransientFailureCoincidingWithTeardownStaysFailed) } backend->arm_key = store->layout().gcStateKey(); backend->on_read = [&store] { store->beginTeardown(); }; - EXPECT_THROW(sched.runOneRoundNow(Rec::Trigger::Manual), DB::Exception); + EXPECT_THROW(sched.runOneRoundNow(), DB::Exception); const std::vector round_rows = roundRowsOnly(rows); ASSERT_EQ(round_rows.size(), 2u); diff --git a/src/Disks/tests/gtest_cas_gc_teardown_stop.cpp b/src/Disks/tests/gtest_cas_gc_teardown_stop.cpp index 8b5e31dafff8..792e053498ab 100644 --- a/src/Disks/tests/gtest_cas_gc_teardown_stop.cpp +++ b/src/Disks/tests/gtest_cas_gc_teardown_stop.cpp @@ -417,7 +417,7 @@ TEST(CASGCTeardownStop, AQueuedScheduledTickEmitsNoStartAfterTeardownBegan) GateOpenedOnExit opener(*release); backend->armParkFirstRead(store->layout().gcStateKey(), entered, release); auto manual = std::async(std::launch::async, - [&sched] { return sched.runOneRoundNow(GcRoundLogRecord::Trigger::Manual); }); + [&sched] { return sched.runOneRoundNow(); }); entered->wait("entered"); /// The loop wakes and queues on the round mutex behind the parked manual round. From e6cabaf09379020aef9575227b68ceb818cb6dea Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 10:04:46 +0200 Subject: [PATCH 29/37] Use remountTerminal in the GC scheduler loops The scheduler repeated the three conditions of `CasMountRuntime::remountTerminal`. A `Vanished` lifecycle always has the intent published, so the `isVanished` term was redundant. `Pool::vanishedIntentPublished` had no other caller and goes. Co-Authored-By: Claude Sonnet 5.5 --- .../ContentAddressed/Gc/CasGcScheduler.cpp | 12 +++++------- .../MetadataStorages/ContentAddressed/Pool/CasPool.h | 8 +++----- 2 files changed, 8 insertions(+), 12 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.cpp index 518c52bddbe7..1e3e1db792b6 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.cpp @@ -357,15 +357,14 @@ void CasGcScheduler::loop() /// FOREVER — `acquireOrRenewLease` throws `CORRUPTED_DATA` against the vanished `gc/state` every /// interval (the G2 zombie: an error-log line + a Failed round row each tick), and worse, after /// `VanishedReplaced` the `allow_steal=true` rounds could STEAL the FOREIGN pool's `gc/state` lease - /// and fold/condemn/delete its objects. We also exit on a published FORGET intent - /// (`vanishedIntentPublished`, still pre-terminal) — earliest-signal discipline — and on `IdentityLost` - /// (rev.8: a fail-loud terminal state; the last G2-zombie case — eternal `CORRUPTED_DATA` retries + /// and fold/condemn/delete its objects. `remountTerminal` is true from the moment FORGET + /// publishes its intent (still pre-terminal) and on `IdentityLost` + /// (a fail-loud terminal state; the last G2-zombie case — eternal `CORRUPTED_DATA` retries /// against a half-erased pool — closes with it). Clearing `i_am_leader` before returning keeps /// `gcHealth` honest (a terminal, self-exited scheduler reports it no longer leads). The thread exits /// its OWN loop here — no join from this context (C6-safe); `stop()`/`~CasGcScheduler` still join the /// finished thread cleanly. - if (store->isVanished() || store->vanishedIntentPublished() - || store->lifecycle() == Cas::PoolLifecycle::IdentityLost) + if (store->remountTerminal()) { i_am_leader.store(false, std::memory_order_relaxed); { @@ -480,8 +479,7 @@ void CasGcScheduler::heartbeatLoop() /// rev.7 §3 [C1] + rev.8 §9 item 8: self-exit on ANY terminal (or FORGET-intent) pool, same as /// `loop()`. A terminal pool's advisory pulses would target a deleted `gc/hb` key (`IdentityLost`) or /// a FOREIGN pool's key (`VanishedReplaced`) — stop pulsing the moment the pool goes terminal. - if (store->isVanished() || store->vanishedIntentPublished() - || store->lifecycle() == Cas::PoolLifecycle::IdentityLost) + if (store->remountTerminal()) { { std::lock_guard exit_lock(terminal_exit_mutex); diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h index fb3e9641167a..4581f38a29fd 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h @@ -511,11 +511,9 @@ class Pool : public std::enable_shared_from_this /// Whether the pool has reached one of the two fully-terminal `Vanished` values /// (`VanishedReplaced` / `VanishedForgotten`). bool isVanished() const { return mount_runtime.isVanished(); } - /// Whether the terminal-intent latch is published — a natural `enterVanished`, OR FORGET's early - /// (spec §5 step 1) `publishVanishedIntent`, and NEVER `IdentityLost` ([C1]). See - /// `CasMountRuntime::vanishedIntentPublished`. The GC scheduler consults this ALONGSIDE `isVanished()` - /// to self-exit its loops the instant the pool is (being driven) terminal, at the earliest signal. - bool vanishedIntentPublished() const { return mount_runtime.vanishedIntentPublished(); } + /// Whether background work must stop: a terminal intent is published (a natural `enterVanished` or + /// FORGET's early publication) or the pool is `IdentityLost`. See `CasMountRuntime::remountTerminal`. + bool remountTerminal() const { return mount_runtime.remountTerminal(); } /// The store()-class lifecycle gate: throws the typed `INVALID_STATE` error, whose message names the /// terminal sub-state, when the pool has entered `IdentityLost` or any `Vanished` state; returns /// silently while `Live`/`TransientNotLive`. This is the minimal "nothing silently proceeds" hook the From 1152bc81a198eda80c6a24f9a2fc0eaef075d516 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 10:12:47 +0200 Subject: [PATCH 30/37] Remove the write-only fence, runtime and renewal result fields `MountFence::server_uuid` and `writer_epoch` were stored on every arm and never read; `backend_ptr`, `Pool::farewellRequests` and `MountRenewResult::sent_any` had no readers either. `armMountFence` takes the deadline only. The test that read `sent_any` reads `attempts_sent`. Co-Authored-By: Claude Sonnet 5.5 --- .../ContentAddressed/Pool/CasMountRuntime.cpp | 8 +- .../ContentAddressed/Pool/CasMountRuntime.h | 8 +- .../ContentAddressed/Pool/CasPool.cpp | 6 +- .../ContentAddressed/Pool/CasPool.h | 11 +-- .../ContentAddressed/Pool/CasServerRoot.cpp | 4 - .../ContentAddressed/Pool/CasServerRoot.h | 1 - src/Disks/tests/cas_test_helpers.h | 2 +- .../tests/gtest_cas_fence_generation.cpp | 2 +- src/Disks/tests/gtest_cas_heartbeat.cpp | 2 +- src/Disks/tests/gtest_cas_mount_runtime.cpp | 26 +++---- src/Disks/tests/gtest_cas_pool.cpp | 74 +++++++++---------- .../gtest_cas_ref_wedge_every_attempt.cpp | 2 +- 12 files changed, 63 insertions(+), 83 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index ded2a2ea9941..f57cfdf29635 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -47,7 +47,6 @@ int64_t wallClockNowSeconds() } CasMountRuntime::CasMountRuntime( - BackendPtr backend_ptr_, CasRequests & farewell_requests_, CasRequests & lease_requests_, const Layout & layout_, @@ -56,8 +55,7 @@ CasMountRuntime::CasMountRuntime( const CasEventSink & event_sink_, CasRequestBudget cas_request_budget_, std::function remount_attempt_) - : backend_ptr(std::move(backend_ptr_)) - , farewell_requests(farewell_requests_) + : farewell_requests(farewell_requests_) , lease_requests(lease_requests_) , layout(layout_) , config(std::move(config_)) @@ -272,10 +270,8 @@ void CasMountRuntime::setMountDeadline(uint64_t deadline_boot_ms) mount_fence.deadline_boot_ms.store(deadline_boot_ms, std::memory_order_release); } -void CasMountRuntime::armMountFence(UInt128 server_uuid, uint64_t writer_epoch, uint64_t deadline_boot_ms) +void CasMountRuntime::armMountFence(uint64_t deadline_boot_ms) { - mount_fence.server_uuid = server_uuid; - mount_fence.writer_epoch = writer_epoch; armFence(deadline_boot_ms, /*report_live=*/false); } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index 91487e93ecf9..fa5b4f06c248 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -107,8 +107,6 @@ struct MountConfig /// Container pause is already safe under either clock (the process is frozen, so no local check runs). struct MountFence { - UInt128 server_uuid{}; - uint64_t writer_epoch = 0; /// Until something arms a real lease deadline, the permissive default allows mutations. UINT64_MAX = /// unarmed (never expires); otherwise a CLOCK_BOOTTIME-milliseconds instant. std::atomic deadline_boot_ms{std::numeric_limits::max()}; @@ -119,14 +117,13 @@ struct MountFence /// the `MountLeaseRenewer`, local `MountFence`, build watermark and in-flight build registry, /// `live_writer_epoch`, and the lease thread. `Pool` retains the higher-level /// claim/recovery sequence and its `remount_mutex`; in particular, the runtime does not acquire or own -/// the ref-ledger locks. The runtime receives its backend, layout, configuration, event sink, request +/// the ref-ledger locks. The runtime receives its request planes, layout, configuration, event sink, request /// budget, and a callback that performs one pool-level remount attempt, so it has no `Pool` back-reference. /// `Pool` delegates preserve the existing callers and test seams. class CasMountRuntime { public: CasMountRuntime( - BackendPtr backend_ptr_, /// The planes the `MountLeaseRenewer` runs on: the claim and the farewell on an open-fence one, /// the renewal on `lease_requests_`, which has no lease budget and whose sleep a stop wakes. /// Owned by `Pool` and outliving this runtime. @@ -164,7 +161,7 @@ class CasMountRuntime void setMountDeadline(uint64_t deadline_boot_ms); /// Arm a new lease incarnation and clear any loss latched for the prior incarnation. Unconditional /// and reports no `Live`; a reclaim arms through `armIfAdmissible`. - void armMountFence(UInt128 server_uuid, uint64_t writer_epoch, uint64_t deadline_boot_ms); + void armMountFence(uint64_t deadline_boot_ms); /// Step 0 of a reclaim. One step under `driver_mutex`: record the requested generation this attempt /// serves and latch the fence. void beginReclaim(); @@ -461,7 +458,6 @@ class CasMountRuntime std::unique_lock lockTerminalPublication(); /// ---- injected environment (no `Pool` back-reference); initialized first, in this order ---- - BackendPtr backend_ptr; CasRequests & farewell_requests; CasRequests & lease_requests; const Layout & layout; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp index 861ab3e426eb..c5e7b6e8bd8c 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp @@ -258,7 +258,7 @@ Pool::Pool(BackendPtr backend_, PoolConfig config_, PoolMeta meta_) /// (post-construction). Declared/constructed AFTER `ref_ledger`, preserving the original member order /// verbatim (mount destroyed first, ledger last; both orders proven safe -- see the header note). , mount_runtime( - pool_backend, farewell_requests, lease_requests, + farewell_requests, lease_requests, pool_layout, config.mountConfig(), config.server_root_id, event_sink_, config.cas_request_budget, [this] { return tryRemountOnce(); }) @@ -328,9 +328,9 @@ void Pool::setMountDeadline(uint64_t deadline_boot_ms) mount_runtime.setMountDeadline(deadline_boot_ms); } -void Pool::armMountFence(UInt128 server_uuid, uint64_t writer_epoch, uint64_t deadline_boot_ms) +void Pool::armMountFence(uint64_t deadline_boot_ms) { - mount_runtime.armMountFence(server_uuid, writer_epoch, deadline_boot_ms); + mount_runtime.armMountFence(deadline_boot_ms); } String Pool::lifecycleReasonDetail(PoolLifecycle lc) const diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h index 4581f38a29fd..dd5f2630dad2 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h @@ -476,8 +476,8 @@ class Pool : public std::enable_shared_from_this /// Refresh the write-fence deadline (a CLOCK_BOOTTIME-milliseconds instant; release). /// renewer renew calls this on success. void setMountDeadline(uint64_t deadline_boot_ms); - /// Arm the fence at startup: set (uuid, epoch, deadline), clear `lost`. - void armMountFence(UInt128 server_uuid, uint64_t writer_epoch, uint64_t deadline_boot_ms); + /// Arm the fence at startup: set the deadline, clear `lost`. + void armMountFence(uint64_t deadline_boot_ms); void setArmMountFenceInterpositionHookForTest(std::function hook) { mount_runtime.setArmMountFenceInterpositionHookForTest(std::move(hook)); @@ -753,11 +753,6 @@ class Pool : public std::enable_shared_from_this /// operation admitted here is refused the moment the fence trips, is re-armed under a fresh lease /// incarnation, or runs out of room before the lease expires. CasRequests & mountRequests() { return mount_requests; } - /// The farewell plane, on an open fence -- shared with the mount-lease renewer's own claim/adopt, - /// not only its release: a self-remount claims with the fence already latched lost, so gating the - /// claim on the fence could never reclaim, and refusing the farewell because the mount fence has - /// already run down would leave the slot looking live until GC fences it out. - CasRequests & farewellRequests() { return farewell_requests; } /// The open-fence plane: GC, the offline tools, this pool's own reads, and the bootstrap-control /// claims. None of them hold a mount lease -- the claims are what ESTABLISHES one, so gating them /// on the fence would make a self-remount, which runs with the fence latched lost, unable ever to @@ -1252,7 +1247,7 @@ class Pool : public std::enable_shared_from_this /// from Pool. Owns the `MountLeaseRenewer`, the local `MountFence`, the per-server /// build watermark (the `builds_mutex`-guarded seq/registry) and its in-flight-build /// map, the live-incarnation `live_writer_epoch`, and the - /// lease thread (with one driver mutex/condition pair). Injected with backend/layout + /// lease thread (with one driver mutex/condition pair). Injected with the layout + /// the `MountConfig` slice + `server_root_id` + the event-sink reference + the pool `cas_request_budget` /// + a `remount_attempt` callback (== `Pool::tryRemountOnce`, which STAYS on Pool: the claim/recovery /// ORCHESTRATION drives these owned primitives). diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp index ff88b3f3ae27..8fd0334a1b77 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp @@ -1397,13 +1397,11 @@ MountRenewResult MountLeaseRenewer::renew(const MountRenewOperationEnvironment & result.outcome = MountRenewOutcome::Committed; result.attempts_sent = committed->attempts_sent; result.resolved_by_read = committed->resolved_by_read; - result.sent_any = committed->attempts_sent != 0; return finished(std::move(result)); } if (const Conflict * conflict = std::get_if(&*written)) { - result.sent_any = true; result.attempts_sent = conflict->attempts_sent; try { @@ -1418,7 +1416,6 @@ MountRenewResult MountLeaseRenewer::renew(const MountRenewOperationEnvironment & if (const Refused * refused = std::get_if(&*written)) { - result.sent_any = true; result.attempts_sent = refused->attempts_sent; result.classification = MountRenewTerminalClassification::DeterministicFailure; result.failure = std::make_exception_ptr(Exception( @@ -1429,7 +1426,6 @@ MountRenewResult MountLeaseRenewer::renew(const MountRenewOperationEnvironment & if (const GaveUp * gave_up = std::get_if(&*written)) { - result.sent_any = gave_up->sent_any; result.attempts_sent = gave_up->attempts_sent; /// Nothing was sent and the node was already stopping: the lease is exactly as it was, so this diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h index d8160af69f20..7fff6c2e7774 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h @@ -69,7 +69,6 @@ struct MountRenewResult /// a give-up, a conflict, or a store refusal. uint32_t attempts_sent = 0; bool resolved_by_read = false; - bool sent_any = false; std::exception_ptr failure; /// The body this renewal wrote or tried to write. uint64_t writer_epoch = 0; diff --git a/src/Disks/tests/cas_test_helpers.h b/src/Disks/tests/cas_test_helpers.h index ae87f129fce3..e82dfddc348d 100644 --- a/src/Disks/tests/cas_test_helpers.h +++ b/src/Disks/tests/cas_test_helpers.h @@ -2217,7 +2217,7 @@ using HintHoleBackend = HintHoleBackendOn; /// AFTER the reaction. It bumps the fence GENERATION, exactly as a real re-arm does. inline void rearmMountFenceAfterAnomalyForTest(const DB::Cas::PoolPtr & store) { - store->armMountFence(DB::UInt128{0, 1}, store->liveWriterEpoch(), store->bootMsNow() + 600000); + store->armMountFence(store->bootMsNow() + 600000); } /// Delegates the FIRST matching create-shaped write to `CountingBackend` -- so the write actually diff --git a/src/Disks/tests/gtest_cas_fence_generation.cpp b/src/Disks/tests/gtest_cas_fence_generation.cpp index ac726ec89e12..6adfefa125af 100644 --- a/src/Disks/tests/gtest_cas_fence_generation.cpp +++ b/src/Disks/tests/gtest_cas_fence_generation.cpp @@ -216,7 +216,7 @@ TEST(CASFenceGeneration, RearmPublishesTheNewGenerationBeforeOpeningTheFence) << "no runtime may be published in the re-arm interposition"; }); - store->armMountFence(DB::UInt128{0, 1}, store->liveWriterEpoch(), store->bootMsNow() + 600000); + store->armMountFence(store->bootMsNow() + 600000); store->setArmMountFenceInterpositionHookForTest(nullptr); EXPECT_FALSE(admitted_in_interposition); diff --git a/src/Disks/tests/gtest_cas_heartbeat.cpp b/src/Disks/tests/gtest_cas_heartbeat.cpp index 3efba9218854..bbd71b16ce55 100644 --- a/src/Disks/tests/gtest_cas_heartbeat.cpp +++ b/src/Disks/tests/gtest_cas_heartbeat.cpp @@ -1151,7 +1151,7 @@ TEST(CASHeartbeat, CancellationAfterSendIsTerminalAndForbidsRelease) renewalEnvironment(/*live=*/[&] { return !cancelled; }, /*cancelled=*/[&] { return cancelled; })); const DB::Exception failure = terminalException(result); EXPECT_EQ(failure.code(), DB::ErrorCodes::NETWORK_ERROR); - EXPECT_TRUE(result.sent_any); + EXPECT_GE(result.attempts_sent, 1u); EXPECT_EQ(backend->read_calls, 0u) << "post-write cancellation must not start a diagnostic read"; EXPECT_EQ(renewer.state(), MountLeaseRenewerState::RenewalTerminal); const String bytes_before = ops.op.read(layout.mountKey("test"), Retry::standard())->bytes; diff --git a/src/Disks/tests/gtest_cas_mount_runtime.cpp b/src/Disks/tests/gtest_cas_mount_runtime.cpp index f0804bb39701..04ac34041383 100644 --- a/src/Disks/tests/gtest_cas_mount_runtime.cpp +++ b/src/Disks/tests/gtest_cas_mount_runtime.cpp @@ -31,7 +31,7 @@ class RuntimeFixture , farewell(backend, Fence::open()) , lease(backend, Fence::open()) , runtime( - backend, farewell, lease, layout, + farewell, lease, layout, MountConfig{.boot_ms_fn = [this] { return boot_ms; }}, "test", sink, CasRequestBudget{.attempt_timeout_ms = attempt_timeout_ms, @@ -82,8 +82,6 @@ String refusalText(const std::function & refuse) return {}; } -constexpr DB::UInt128 kUuid{7}; - } /// The boundary is STRICT on both terms: a request that would only just finish as the lease runs out @@ -92,7 +90,7 @@ TEST(CASMountRuntime, AdmitRefusesAtTheExactBudgetBoundary) { RuntimeFixture f(/*lease_safety_margin_ms=*/20); f.boot_ms = 1'000; - f->armMountFence(kUuid, 1, /*deadline_boot_ms=*/1'100); /// 100 ms of lease left + f->armMountFence(/*deadline_boot_ms=*/1'100); /// 100 ms of lease left const uint64_t generation = f->fenceGeneration(); EXPECT_STREQ(admitName(f->admit(generation, 80)), "NoBudget") << "needed + margin == remaining must refuse"; @@ -105,7 +103,7 @@ TEST(CASMountRuntime, AdmitDoesNotWrapOnAnAbsurdNeed) { RuntimeFixture f(/*lease_safety_margin_ms=*/20); f.boot_ms = 1'000; - f->armMountFence(kUuid, 1, /*deadline_boot_ms=*/1'100); + f->armMountFence(/*deadline_boot_ms=*/1'100); EXPECT_STREQ(admitName(f->admit(f->fenceGeneration(), std::numeric_limits::max())), "NoBudget"); } @@ -114,7 +112,7 @@ TEST(CASMountRuntime, AdmitRefusesAnExpiredLease) { RuntimeFixture f(/*lease_safety_margin_ms=*/0); f.boot_ms = 1'000; - f->armMountFence(kUuid, 1, /*deadline_boot_ms=*/1'100); + f->armMountFence(/*deadline_boot_ms=*/1'100); const uint64_t generation = f->fenceGeneration(); f.boot_ms = 1'099; @@ -134,9 +132,9 @@ TEST(CASMountRuntime, AdmitRefusesAGenerationTheFenceMovedPast) { RuntimeFixture f(/*lease_safety_margin_ms=*/0); f.boot_ms = 1'000; - f->armMountFence(kUuid, 1, /*deadline_boot_ms=*/100'000); + f->armMountFence(/*deadline_boot_ms=*/100'000); const uint64_t stale = f->fenceGeneration(); - f->armMountFence(kUuid, 2, /*deadline_boot_ms=*/100'000); + f->armMountFence(/*deadline_boot_ms=*/100'000); EXPECT_STREQ(admitName(f->admit(stale, 0)), "LostOrRearmed"); EXPECT_STREQ(admitName(f->admit(f->fenceGeneration(), 0)), "Ok"); @@ -148,7 +146,7 @@ TEST(CASMountRuntime, AdmitRefusesALostFenceWhateverTheBudget) { RuntimeFixture f(/*lease_safety_margin_ms=*/0); f.boot_ms = 1'000; - f->armMountFence(kUuid, 1, /*deadline_boot_ms=*/100'000); + f->armMountFence(/*deadline_boot_ms=*/100'000); f->tripMountLost(); EXPECT_STREQ(admitName(f->admit(f->fenceGeneration(), 0)), "LostOrRearmed"); @@ -171,7 +169,7 @@ TEST(CASMountRuntime, RefAppendFenceOkIsAdmitAtTwoEnvelopes) /// timeout (10 ms); refAppendFenceOk asks for TWO of them (a write and its settlement read). RuntimeFixture f(/*lease_safety_margin_ms=*/20, /*attempt_timeout_ms=*/10); f.boot_ms = 1'000; - f->armMountFence(kUuid, 1, /*deadline_boot_ms=*/1'041); /// 41 ms left: one more than 2*10 + 20 + f->armMountFence(/*deadline_boot_ms=*/1'041); /// 41 ms left: one more than 2*10 + 20 EXPECT_TRUE(f->refAppendFenceOk()); EXPECT_STREQ(admitName(f->admit(f->fenceGeneration(), 20)), "Ok"); @@ -186,7 +184,7 @@ TEST(CASMountRuntime, RefAppendFenceOkIsAdmitAtTwoEnvelopesWithANonzeroCap) { RuntimeFixture f(/*lease_safety_margin_ms=*/20, /*attempt_timeout_ms=*/100, /*connect_timeout_cap_ms=*/50); f.boot_ms = 1'000; - f->armMountFence(kUuid, 1, /*deadline_boot_ms=*/1'421); /// 421 ms left: one more than 2*200 + 20 + f->armMountFence(/*deadline_boot_ms=*/1'421); /// 421 ms left: one more than 2*200 + 20 EXPECT_TRUE(f->refAppendFenceOk()); EXPECT_STREQ(admitName(f->admit(f->fenceGeneration(), 400)), "Ok"); @@ -203,7 +201,7 @@ TEST(CASMountRuntime, LeaseExpiredOnlyWhileLiveAndNotLost) f.boot_ms = 1'000; EXPECT_FALSE(f->leaseExpiredSinceBootMs().has_value()) << "an unarmed fence has no deadline to pass"; - f->armMountFence(kUuid, 1, /*deadline_boot_ms=*/1'100); + f->armMountFence(/*deadline_boot_ms=*/1'100); f.boot_ms = 1'099; EXPECT_FALSE(f->leaseExpiredSinceBootMs().has_value()); f.boot_ms = 1'100; @@ -226,7 +224,7 @@ TEST(CASMountRuntime, ExpiredLeaseRefusalSaysWritesResume) { RuntimeFixture f(/*lease_safety_margin_ms=*/0); f.boot_ms = 1'000; - f->armMountFence(kUuid, 1, /*deadline_boot_ms=*/1'100); + f->armMountFence(/*deadline_boot_ms=*/1'100); const uint64_t generation = f->fenceGeneration(); f.boot_ms = 1'100; @@ -235,7 +233,7 @@ TEST(CASMountRuntime, ExpiredLeaseRefusalSaysWritesResume) EXPECT_NE(expired.find("writes resume when a renewal restores it"), String::npos) << expired; /// A re-arm moves the generation while the lease stays expired: the caller's incarnation is gone. - f->armMountFence(kUuid, 1, /*deadline_boot_ms=*/1'100); + f->armMountFence(/*deadline_boot_ms=*/1'100); const String moved = refusalText([&] { f->checkFenceOrThrow(generation); }); EXPECT_EQ(moved.find("lease expired"), String::npos) << moved; EXPECT_NE(moved.find("mount fence tripped"), String::npos) << moved; diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp index d4599dfd6168..3e54b9d56a24 100644 --- a/src/Disks/tests/gtest_cas_pool.cpp +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -2072,7 +2072,7 @@ class RuntimeUnderTest RuntimeUnderTest(const std::shared_ptr & backend, Args &&... args) : farewell(backend, DB::Cas::Fence::open()) , lease(backend, DB::Cas::Fence::open()) - , runtime(backend, farewell, lease, std::forward(args)...) + , runtime(farewell, lease, std::forward(args)...) { /// What the request engine reserves per attempt is the BACKEND's attempt timeout, not the /// budget field alone; every construction of this holder pairs the two via `runtimeRenewBudget`, @@ -2178,7 +2178,7 @@ void verifyForeignConflictSinkIsNonInterfering(ForeignConflictSinkBehavior behav runtime_ptr = &runtime; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); DB::Cas::tests::OperationForTest successor_op(*backend); auto ours = (*successor_op).read(key, Retry::standard()); @@ -3371,7 +3371,7 @@ TEST(CASPoolRemount, DirectRenewIsRefusedForBackgroundConfiguredRuntimeAfterStop CasMountRuntime & runtime = *runtime_holder; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); runtime.startBackgroundWorkers(); runtime.stopBackgroundWorkers(); EXPECT_RUNTIME_STATE_REJECTION(runtime.renewWatermarkOnce()); @@ -3408,7 +3408,7 @@ TEST(CASPoolRemount, ReclaimWaitsForTheRenewalToEnd) CasMountRuntime & runtime = *runtime_holder; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); backend->barrier = &renewal_barrier; backend->fault = RuntimeRenewBackend::Fault::BlockThenDelegate; /// Set after the setup's own writes: only the held renewal request sets it. @@ -3454,7 +3454,7 @@ TEST(CASPoolRemount, TeardownJoinsTheLeaseThreadBeforeRelease) CasMountRuntime & runtime = *runtime_holder; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); runtime.startBackgroundWorkers(); runtime.stopBackgroundWorkers(); EXPECT_EQ(worker_exits.load(), 1u); @@ -3569,7 +3569,7 @@ TEST(CASMountRuntime, MemoryLimitDoesNotEndTheLeaseThread) CasMountRuntime & runtime = *runtime_holder; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); const uint64_t leases_lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); backend->fault = RuntimeRenewBackend::Fault::ThrowMemoryLimitExceeded; @@ -3644,7 +3644,7 @@ TEST(CASPoolRemount, NaturalTerminalTransitionMakesTheLeaseThreadExit) runtime_ptr = &runtime; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); runtime.startBackgroundWorkers(); runtime.tripMountLost(); runtime.scheduleRemount(); @@ -3736,7 +3736,7 @@ TEST(CASPoolRemount, ALeaseWaitCannotMissNaturalTerminalPublication) CasMountRuntime & runtime = *runtime_holder; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); if (wait == Wait::ReclaimBackoff) { /// A request before the start makes the first pass a reclaim, which fails and backs off. @@ -3805,7 +3805,7 @@ TEST(CASPoolRemount, VanishedReasonPreparationFailureLeavesTerminalTransitionRet CasMountRuntime & runtime = *runtime_holder; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); runtime.startBackgroundWorkers(); runtime.tripMountLost(); @@ -3852,7 +3852,7 @@ TEST(CASPoolRemount, WorkerConstructionRollbackFailsOpenClosed) CasMountRuntime & runtime = *runtime_holder; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); EXPECT_THROW(runtime.startBackgroundWorkers(), DB::Exception); EXPECT_FALSE(runtime.mayMutate()); EXPECT_FALSE(runtime.workersRunningForTest()); @@ -3900,7 +3900,7 @@ TEST(CASPoolRemount, ExternalLossDuringRenewalUsesOneRecoveryGeneration) runtime_ptr = &runtime; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); backend->barrier = &renewal_barrier; backend->fault = RuntimeRenewBackend::Fault::BlockThenDelegate; const auto lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); @@ -3944,7 +3944,7 @@ TEST(CASPoolRemount, ConcurrentRemountRequestIsProcessedAfterActiveGeneration) CasMountRuntime & runtime = *runtime_holder; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); runtime.startBackgroundWorkers(); runtime.tripMountLost(); runtime.scheduleRemount(); @@ -4006,7 +4006,7 @@ TEST(CASPoolRemount, ImmediatePostRemountRenewalFailureIsNotDropped) runtime_ptr = &runtime; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 10'000); + runtime.armMountFence(anchor + 10'000); runtime.startBackgroundWorkers(); runtime.tripMountLost(); runtime.scheduleRemount(); @@ -4072,7 +4072,7 @@ TEST(CASPoolShutdown, PreSendCancellationAllowsFarewellButAmbiguityDoesNot) CasMountRuntime & runtime = *runtime_holder; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); if (ambiguous) { backend->barrier = &barrier; @@ -4117,7 +4117,7 @@ TEST(CASPool, DirectTerminalFailureRethrowsTypedException) CasMountRuntime & runtime = *runtime_holder; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); /// The renewal lands, then its liveness ends before the commit is confirmed. backend->after_commit = [&] { renewal_live.store(false, std::memory_order_release); }; try @@ -4216,7 +4216,7 @@ TEST(CASPool, DeterministicWorkerFailureFencesWithoutWaitingForCadence) CasMountRuntime & runtime = *runtime_holder; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); /// A definitive answer ends the lease thread's renewal; a transient fault would only be retried. fenceOutMount(*backend, layout.mountKey("test")); runtime.startBackgroundWorkers(); @@ -4262,7 +4262,7 @@ TEST(CASMountRuntime, ForgetEndsAnUnboundedRenewal) CasMountRuntime & runtime = *runtime_holder; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); past_the_lease = anchor + 5'000; runtime_holder.setRetrySleepForTest([&](uint64_t ms) { @@ -4317,7 +4317,7 @@ TEST(CASMountRuntime, StopWakesTheRetryWaitOfAnUnboundedRenewal) CasMountRuntime & runtime = *runtime_holder; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); runtime_holder.setRetrySleepForTest([&](uint64_t ms) { const bool first = first_wait.exchange(false); @@ -4389,7 +4389,7 @@ TEST(CASMountRuntime, AStaleSuccessIsFollowedAtOnceByTheNextRenewal) runtime_ptr = &runtime; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); runtime_holder.setRetrySleepForTest([&](uint64_t ms) { boot_ms += ms; }); /// The first renewal starts one period after the anchor and fails for three lease lengths. boot_ms = anchor + 500; @@ -4494,7 +4494,7 @@ TEST(CASMountRuntime, NothingArmsWhileARemountRequestIsPending) runtime_ptr = &runtime; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); const uint64_t lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); runtime.startBackgroundWorkers(); /// A request without a trip: the fence is still armed when the reclaim starts. @@ -4600,7 +4600,7 @@ TEST(CASMountRuntime, AnInterferenceReportDuringAReclaimIsServedByTheNextReclaim runtime_ptr = &runtime; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); const uint64_t lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); runtime.startBackgroundWorkers(); runtime.scheduleRemount(); @@ -4674,7 +4674,7 @@ TEST(CASMountRuntime, AReclaimAcknowledgesOnlyTheGenerationItServed) runtime_ptr = &runtime; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); const uint64_t lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); runtime.startBackgroundWorkers(); /// One interference report. @@ -4739,7 +4739,7 @@ TEST(CASMountRuntime, AReclaimFinishedAfterTheForgetIntentArmsNothing) runtime_ptr = &runtime; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); runtime.startBackgroundWorkers(); runtime.tripMountLost(); runtime.scheduleRemount(); @@ -4810,7 +4810,7 @@ TEST(CASMountRuntime, StopDuringAReclaimJoinsTheThread) runtime_ptr = &runtime; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); const String key = layout.mountKey("test"); /// A definitive answer ends the first renewal and requests the reclaim; nobody claims the slot back. fenceOutMount(*backend, key); @@ -4922,7 +4922,7 @@ TEST(CASMountRuntime, ARemountRequestEndsTheRenewalAndTheSameThreadReclaims) runtime_ptr = &runtime; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); /// Five lease lengths on: a renewal bounded by its lease ended long before. past_the_lease = anchor + 5'000; runtime_holder.setRetrySleepForTest([&](uint64_t ms) @@ -5066,7 +5066,7 @@ TEST(CASMountRuntime, ACommitConsumedAfterARemountRequestCannotOverwriteTheRecla runtime_ptr = &runtime; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); /// Set after the setup's own writes: only the loop's renewal may raise the request. backend->after_commit = [&] { committed = true; }; runtime.startBackgroundWorkers(); @@ -5111,7 +5111,7 @@ TEST(CASMountRuntimeDeathTest, OnlyTheLeaseThreadReplacesTheRenewerWhileItRuns) CasMountRuntime & runtime = *runtime_holder; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); runtime.startBackgroundWorkers(); /// The lease thread is held between its decision to renew and the renewal, with no lock held. admitted.waitUntilArrived(); @@ -5524,7 +5524,7 @@ void runExpiryScenario(const String & layout_prefix, ExpiryObservation & seen) runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); ASSERT_EQ(anchor, kExpiryClaimBootMs); - runtime.armMountFence(uuid, 1, anchor + kExpiryTtlMs); + runtime.armMountFence(anchor + kExpiryTtlMs); seen.generation_before = runtime.fenceGeneration(); seen.attempts_before = eventCount(ProfileEvents::CASMountRenewalAttempts); @@ -5624,7 +5624,7 @@ void runReadOutageScenario(const String & layout_prefix, ReadOutageObservation & runtime_ptr = &runtime; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + kExpiryTtlMs); + runtime.armMountFence(anchor + kExpiryTtlMs); seen.attempts_before = eventCount(ProfileEvents::CASMountRenewalAttempts); seen.retries_before = eventCount(ProfileEvents::CASMountRenewalRetries); @@ -5827,7 +5827,7 @@ void runExpiryLogScenario(const String & layout_prefix, ExpiryLogRig & rig, uint rig.runtime = &runtime; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + kExpiryTtlMs); + runtime.armMountFence(anchor + kExpiryTtlMs); rig.generation = runtime.fenceGeneration(); backend->on_put = [&](uint32_t put_no) @@ -5991,7 +5991,7 @@ TEST(CASMountRuntime, ASuccessEndsTheFailureTextOfItsRun) runtime_ptr = &runtime; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + kExpiryTtlMs); + runtime.armMountFence(anchor + kExpiryTtlMs); backend->on_put = [&](uint32_t put_no) { @@ -6065,7 +6065,7 @@ TEST(CASMountRuntime, EachRestoredExpiryIsCountedOnce) runtime_ptr = &runtime; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + kExpiryTtlMs); + runtime.armMountFence(anchor + kExpiryTtlMs); generation = runtime.fenceGeneration(); /// Renewal 1 fails 24 times and then lands stale (put 25); renewal 2 restores (put 26). Renewal 3 @@ -6149,7 +6149,7 @@ TEST(CASMountRuntime, AnExpiryEndedByAFenceIsNotCountedAsARestore) CasMountRuntime & runtime = *runtime_holder; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + kExpiryTtlMs); + runtime.armMountFence(anchor + kExpiryTtlMs); /// The first renewal lands stale, so the lease is still expired when GC fences the slot out; the /// next renewal meets the fence. backend->on_put = [&](uint32_t put_no) @@ -6414,7 +6414,7 @@ TEST(CASMountRuntime, ARenewalIsSentUnderALatchedFence) runtime_ptr = &runtime; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + kExpiryTtlMs); + runtime.armMountFence(anchor + kExpiryTtlMs); /// Puts 1 to 4 fail, 1.3 s each; put 5 commits at 115.2 s, inside the lease of the 100 s claim. backend->on_put = [&](uint32_t put_no) @@ -6519,7 +6519,7 @@ TEST(CASMountRuntime, TheForgetIntentAloneEndsARenewal) CasMountRuntime & runtime = *runtime_holder; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + kExpiryTtlMs); + runtime.armMountFence(anchor + kExpiryTtlMs); /// Every put fails; put 3 is held until the intent is published. backend->on_put = [&](uint32_t put_no) @@ -6601,7 +6601,7 @@ TEST(CASMountRuntime, ATripAloneIsRearmedByTheNextRenewal) runtime_ptr = &runtime; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + kExpiryTtlMs); + runtime.armMountFence(anchor + kExpiryTtlMs); runtime.startBackgroundWorkers(); second_pass.waitUntilArrived(); @@ -6666,7 +6666,7 @@ TEST(CASMountRuntime, ARenewalArmsNothingWhileARequestIsPending) runtime_ptr = &runtime; runtime.installRenewer(uuid, 1, [&] { return wall_ms; }); const uint64_t anchor = runtime.startRenewer(); - runtime.armMountFence(uuid, 1, anchor + 1000); + runtime.armMountFence(anchor + 1000); const String key = layout.mountKey("test"); const uint64_t writes_before = backend->putOverwriteCount(key); const uint64_t requests_before = runtime.scheduleRemountCallCountForTest(); diff --git a/src/Disks/tests/gtest_cas_ref_wedge_every_attempt.cpp b/src/Disks/tests/gtest_cas_ref_wedge_every_attempt.cpp index 9fcd010643a9..71e7a223a215 100644 --- a/src/Disks/tests/gtest_cas_ref_wedge_every_attempt.cpp +++ b/src/Disks/tests/gtest_cas_ref_wedge_every_attempt.cpp @@ -444,7 +444,7 @@ String epochSealBytes(const RootNamespace & ns, const RefTxnId & id, std::option void bumpFenceGeneration(const PoolPtr & store, uint64_t writer_epoch) { store->tripMountLost(); - store->armMountFence(DB::UInt128{0, 1}, writer_epoch, store->bootMsNow() + 600000); + store->armMountFence(store->bootMsNow() + 600000); /// The fence re-arm alone moves the GENERATION; the live incarnation's writer epoch is a separate /// publication (`tryRemountOnce` does both), and the append lane derives its ids from that one. store->setLiveWriterEpochForTest(writer_epoch); From 8ffad69d274db997a703b30a09cde4cc1bc0e8b5 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 10:13:51 +0200 Subject: [PATCH 31/37] Log the stale-mount observation once per window start `claimMountAwaitingExpiry` logged the threshold right after the opening and remount callers had logged it with the holder's identity. The integration test counts the caller's line. Co-Authored-By: Claude Sonnet 5.5 --- .../MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp | 4 ---- tests/integration/test_cas_mount_renewal_retry/test.py | 2 +- 2 files changed, 1 insertion(+), 5 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp index 8fd0334a1b77..1ef95a61e35b 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp @@ -771,10 +771,6 @@ MountClaimResult claimMountAwaitingExpiry( watch = TokenWatch::sighted(*current_etag, mono_ms_fn()); if (on_wait_start && r.body) on_wait_start(*r.body, threshold_ms); - LOG_INFO(getLogger("CasMountLease"), - "Attempting to mount content-addressed server root {} after node change or hard " - "restart; waiting ~{} ms (token-stability observation) to confirm the previous " - "incarnation's operations are all finalized", srid, threshold_ms); } sleep_ms_fn(poll); diff --git a/tests/integration/test_cas_mount_renewal_retry/test.py b/tests/integration/test_cas_mount_renewal_retry/test.py index 72b5b0063665..392fd3992207 100644 --- a/tests/integration/test_cas_mount_renewal_retry/test.py +++ b/tests/integration/test_cas_mount_renewal_retry/test.py @@ -704,7 +704,7 @@ def log_count_since_last_restart(pattern): # hard-coded, so this stays correct if that fixed budget ever changes. poll_ms = max(1, MOUNT_RENEW_PERIOD_MS // 2) threshold_ms = MOUNT_LEASE_TTL_MS + MOUNT_LEASE_TTL_MS // 20 + poll_ms - observation = "waiting ~{} ms (token-stability observation)".format(threshold_ms) + observation = "observing its write-token for up to ~{} ms before reclaiming".format(threshold_ms) epoch_before = int( node.query( "SELECT writer_epoch FROM system.cas_mounts WHERE disk = '{}' LIMIT 1".format(DISK) From 8ddd7990d7385824e584f4ba20b5844129eebc48 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 10:17:00 +0200 Subject: [PATCH 32/37] Assert GC quiescence over a parked real round `CASGCTeardownStop.BackgroundRoundIsCutAtItsNextRequest` parks a real round and now checks `isQuiescent` before the start and while parked. The test that forced `round_in_flight` through `setRoundInFlightForTest` and the seam are deleted. Co-Authored-By: Claude Sonnet 5.5 --- .../ContentAddressed/Gc/CasGcScheduler.h | 4 ---- src/Disks/tests/gtest_cas_gc_teardown_stop.cpp | 2 ++ src/Disks/tests/gtest_cas_operation_gate.cpp | 15 --------------- 3 files changed, 2 insertions(+), 19 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.h index d7501468bdb8..befad0ca8c21 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.h @@ -161,10 +161,6 @@ class CasGcScheduler /// prove the scheduler's worker threads were joined — no round can be mid-flight once `stop()` returned. bool isQuiescent() const { return !round_in_flight.load(std::memory_order_acquire); } - /// Test seam: force the in-flight-round flag `isQuiescent` reads, so a test can drive - /// "running round => not quiescent" without spinning up a real round against a live backend. - void setRoundInFlightForTest(bool v) { round_in_flight.store(v, std::memory_order_release); } - /// Test seam (rev.7 §3 [C1]): block up to `timeout` for BOTH the pacing and heartbeat loops to have /// SELF-EXITED via the terminal-lifecycle check — a `Vanished` pool or a published FORGET intent — as /// opposed to exiting through `stop()` transitioning `scheduler_state` to `Stopped`. Returns false on timeout. Predicate-based diff --git a/src/Disks/tests/gtest_cas_gc_teardown_stop.cpp b/src/Disks/tests/gtest_cas_gc_teardown_stop.cpp index 792e053498ab..bd5f62c367e3 100644 --- a/src/Disks/tests/gtest_cas_gc_teardown_stop.cpp +++ b/src/Disks/tests/gtest_cas_gc_teardown_stop.cpp @@ -384,8 +384,10 @@ TEST(CASGCTeardownStop, BackgroundRoundIsCutAtItsNextRequest) auto release = std::make_shared(); GateOpenedOnExit opener(*release); backend->armParkFirstRead(store->layout().gcStateKey(), entered, release); + EXPECT_TRUE(sched.isQuiescent()) << "no round has started yet"; sched.start(); entered->wait("entered"); + EXPECT_FALSE(sched.isQuiescent()) << "a round parked inside a request must not read as GC-quiescent"; store->beginTeardown(); const uint64_t requests_at_arm = backend->requestsTotal(); diff --git a/src/Disks/tests/gtest_cas_operation_gate.cpp b/src/Disks/tests/gtest_cas_operation_gate.cpp index 2dbddf6c5f54..fa74d83b58e7 100644 --- a/src/Disks/tests/gtest_cas_operation_gate.cpp +++ b/src/Disks/tests/gtest_cas_operation_gate.cpp @@ -374,21 +374,6 @@ TEST(CASOperationGate, GcEntryPointsRefuseOnNotLive) Cas::tests::expectThrowsCode(ErrorCodes::NETWORK_ERROR, [&] { storage->runOneGcRoundForTest(); }); } -/// (i) `CasGcScheduler::isQuiescent` reflects the round-in-flight flag: a round in flight => not quiescent. -/// (This is the join-completion signal the FORGET / GC-STOP tests rely on.) -TEST(CASOperationGate, GcSchedulerIsQuiescentReflectsRoundInFlight) -{ - auto backend = std::make_shared(); - auto pool = Cas::tests::openPoolForTest(backend); - auto scheduler = std::make_shared( - pool, std::chrono::seconds(3600), "op-gate-test-gc", "disk", Cas::GcRoundLogger{}); - EXPECT_TRUE(scheduler->isQuiescent()); - scheduler->setRoundInFlightForTest(true); - EXPECT_FALSE(scheduler->isQuiescent()) << "a round in flight must NOT read as GC-quiescent"; - scheduler->setRoundInFlightForTest(false); - EXPECT_TRUE(scheduler->isQuiescent()); -} - /// (j) (acceptance matrix — transient auto-recovery / DROP-drain round-trip) The full §4 recovery arc on ONE /// storage: a Remove-class op (the DROP shape) throws the typed transient refusal while the mount lease is /// lost, then SUCCEEDS and actually drains once the disk self-remounts back to Live — no operator action, From 04b4e91aeb851ef8f1d12172eaa48c07c8a6d73c Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 10:18:44 +0200 Subject: [PATCH 33/37] Name the renewer's claim and farewell plane for what it is `MountLeaseRenewer::open_requests` is the plane of the claim and the farewell. It is now `claim_farewell_requests`; the pool's open plane is the GC plane. Co-Authored-By: Claude Sonnet 5.5 --- .../ContentAddressed/Pool/CasServerRoot.cpp | 14 +++++++------- .../ContentAddressed/Pool/CasServerRoot.h | 12 ++++++------ 2 files changed, 13 insertions(+), 13 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp index 1ef95a61e35b..3d00598d1d52 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.cpp @@ -1081,7 +1081,7 @@ String describeRequestFailure(const std::exception & failure) } MountLeaseRenewer::MountLeaseRenewer( - CasRequests & open_requests_, CasRequests & lease_requests_, + CasRequests & claim_farewell_requests_, CasRequests & lease_requests_, const Layout & layout_, const String & srid_, UInt128 server_uuid_, uint64_t writer_epoch_, std::chrono::milliseconds ttl_, std::function now_ms_fn_, @@ -1089,7 +1089,7 @@ MountLeaseRenewer::MountLeaseRenewer( CasEventSink event_sink_, std::chrono::milliseconds lease_safety_margin_, std::function boot_ms_fn_) - : open_requests(open_requests_) + : claim_farewell_requests(claim_farewell_requests_) , lease_requests(lease_requests_) , key(layout_.mountKey(srid_)) , srid(srid_) @@ -1225,7 +1225,7 @@ uint64_t MountLeaseRenewer::start(Liveness liveness) /// Off the mount fence: a self-remount claims with the fence already latched lost, and a claim /// admitted under it would be refused on every request. What makes the claim safe is that every /// write below is conditional. - CasOperation op = open_requests.admit(std::move(liveness)); + CasOperation op = claim_farewell_requests.admit(std::move(liveness)); const Etag etag = claim(op, body); seq = 1; @@ -1479,16 +1479,16 @@ void MountLeaseRenewer::terminate(CasOperation & op, uint64_t lease_deadline_boo .min_active_build_sequence = std::numeric_limits::max(), .write_attempt_id = newMountWriteAttemptId(), }); - /// The farewell is admitted on `open_requests` (see `release`, which calls this via `open_requests.admit()`), + /// The farewell is admitted on `claim_farewell_requests` (see `release`), /// so its own reservation -- attempt plus the read that settles it, `reservedFor(0, 2)` in - /// `CasOperation::writeLoop` -- is exactly `2 * open_requests.attemptReservationMs()`. A window + /// `CasOperation::writeLoop` -- is exactly `2 * claim_farewell_requests.attemptReservationMs()`. A window /// below that value refuses the write before its first attempt, deterministically, on every call: /// `kFarewellBudgetMs` alone predates the attempt-envelope reservation and can no longer be trusted /// to admit it. Saturating, like every other deadline computation on this path: an /// operator-configured envelope is not bounds-checked against this doubling, and wrapping past /// `UINT64_MAX` would turn a too-long window into a too-SHORT one -- the exact failure mode this fix /// exists to remove. - const uint64_t reservation_ms = open_requests.attemptReservationMs(); + const uint64_t reservation_ms = claim_farewell_requests.attemptReservationMs(); const uint64_t doubled_reservation_ms = reservation_ms > std::numeric_limits::max() / 2 ? std::numeric_limits::max() : reservation_ms * 2; @@ -1542,7 +1542,7 @@ void MountLeaseRenewer::release(uint64_t lease_deadline_boot_ms) /// Off the mount fence, for the same reason the claim is: a departing mount whose lease has already /// run down still has to hand the slot back, and refusing the write there would leave the slot /// looking live until GC fences it out. - CasOperation op = open_requests.admit(); + CasOperation op = claim_farewell_requests.admit(); terminate(op, lease_deadline_boot_ms); } diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h index 7fff6c2e7774..eeee0e08a242 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h @@ -571,16 +571,16 @@ bool isCreatorFenceTerminal(CasOperation & op, const Layout & layout, const Stri /// /// PLANES. A renewal runs on `lease_requests`, which has no lease budget: it keeps trying after the /// lease expired, and a stop, a remount request or a terminal lifecycle reaches it through its -/// liveness. The claim and the farewell are admitted on `open_requests`, off the fence: a self-remount -/// claims with the fence already latched lost, so a claim gated on the fence could never reclaim, and a -/// farewell refused because the fence has run down would leave the slot looking live until GC fences it -/// out. Neither is unguarded: a claim's safety is its own conditional write, and a caller that has +/// liveness. The claim and the farewell are admitted on `claim_farewell_requests`, off the fence: a +/// self-remount claims with the fence already latched lost, so a claim gated on the fence could never +/// reclaim, and a farewell refused because the fence has run down would leave the slot looking live +/// until GC fences it out. Neither is unguarded: a claim's safety is its own conditional write, and a caller that has /// shutdown facts hands them over as a `Liveness`. class MountLeaseRenewer { public: MountLeaseRenewer( - CasRequests & open_requests_, CasRequests & lease_requests_, + CasRequests & claim_farewell_requests_, CasRequests & lease_requests_, const Layout & layout_, const String & srid_, UInt128 server_uuid_, uint64_t writer_epoch_, std::chrono::milliseconds ttl_, std::function now_ms_fn_, @@ -617,7 +617,7 @@ class MountLeaseRenewer MountRenewResult terminalResult(MountRenewResult result); void terminate(CasOperation & op, uint64_t lease_deadline_boot_ms); - CasRequests & open_requests; + CasRequests & claim_farewell_requests; /// The plane of a renewal: no lease budget, and a sleep a stop wakes. CasRequests & lease_requests; String key; From 6bdde5475bf9b7085e79df728ca5748f5d946e55 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 10:47:49 +0200 Subject: [PATCH 34/37] Name the lease keeper's two recurring conditions; drop dead code `remountRequestPending` and `stopOrTerminal` replace the spelled-out generation and stop/terminal tests at every site with that exact meaning. `scheduleRemount` had no production caller and is folded into `scheduleRemountForTest`. The tail of `consumeRenewResult` repeated checks that `MountLeaseRenewer::terminalResult` makes before `renew` returns, so it never ran. Co-Authored-By: Claude Opus 5.5 --- .../ContentAddressed/Pool/CasMountRuntime.cpp | 67 +++++++------------ .../ContentAddressed/Pool/CasMountRuntime.h | 14 ++-- src/Disks/tests/gtest_cas_pool.cpp | 28 ++++---- 3 files changed, 48 insertions(+), 61 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index f57cfdf29635..cca136e70530 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -334,7 +334,7 @@ bool CasMountRuntime::waitUntilArmed(uint64_t timeout_ms) const { if (!mount_fence.lost.load(std::memory_order_acquire)) return true; - if (workers_stop_requested || remountTerminal() || !workers_started) + if (stopOrTerminal() || !workers_started) return false; const uint64_t now = bootMsNow(); if (now >= give_up) @@ -348,12 +348,21 @@ bool CasMountRuntime::waitUntilArmed(uint64_t timeout_ms) const bool CasMountRuntime::canArm(uint64_t deadline_boot_ms) const { /// The clock is read last and only when every other term holds. - return !workers_stop_requested - && !remountTerminal() - && remount_requested_generation <= remount_handled_generation + return !stopOrTerminal() + && !remountRequestPending() && budgetAdmits(deadline_boot_ms, bootMsNow(), cas_request_budget.writeAndSettlementReadMs()) == Fence::Admit::Ok; } +bool CasMountRuntime::remountRequestPending() const +{ + return remount_requested_generation > remount_handled_generation; +} + +bool CasMountRuntime::stopOrTerminal() const +{ + return workers_stop_requested || remountTerminal(); +} + bool CasMountRuntime::lossNeedsNewRequest() const { /// A request up to `reclaim_generation` was snapshotted by a reclaim that latched before this loss, @@ -491,9 +500,7 @@ bool CasMountRuntime::renewalLive() const std::lock_guard lock(driver_mutex); /// A lost fence alone does not end it: the renewal that makes an open or a reclaim ready runs under /// one, and every trip that must end it comes with a request, a terminal lifecycle or a stop. - return !workers_stop_requested - && remount_requested_generation <= remount_handled_generation - && !remountTerminal(); + return !stopOrTerminal() && !remountRequestPending(); } bool CasMountRuntime::renewalCancelled() const @@ -549,23 +556,6 @@ std::optional CasMountRuntime::consumeRenewResul driver_cv.notify_all(); } - if (result.outcome != MountRenewOutcome::Terminal) - return restored; - - if (!result.failure) - throw Exception(ErrorCodes::LOGICAL_ERROR, "CAS mount runtime: terminal renewal has no failure"); - try - { - std::rethrow_exception(result.failure); - } - catch (const Exception & e) - { - if (e.code() == ErrorCodes::LOGICAL_ERROR) - throw; - } - catch (...) - { - } return restored; } @@ -663,9 +653,7 @@ void CasMountRuntime::renewalLoop() /// waits for the backoff too. const auto woken = [this](bool by_request) { - const bool wake = workers_stop_requested - || remountTerminal() - || (by_request && remount_requested_generation > remount_handled_generation); + const bool wake = stopOrTerminal() || (by_request && remountRequestPending()); if (!wake && config.lease_wait_predicate_false_hook_for_test) config.lease_wait_predicate_false_hook_for_test(); return wake; @@ -681,9 +669,9 @@ void CasMountRuntime::renewalLoop() uint64_t snapshot = 0; { std::unique_lock lock(driver_mutex); - if (workers_stop_requested || remountTerminal()) + if (stopOrTerminal()) return; - if (remount_requested_generation > remount_handled_generation) + if (remountRequestPending()) { reclaim = true; snapshot = remount_requested_generation; @@ -987,23 +975,13 @@ void CasMountRuntime::publishVanishedIntent() driver_cv.notify_all(); } -void CasMountRuntime::scheduleRemount() -{ - schedule_remount_calls_for_test.fetch_add(1, std::memory_order_relaxed); - std::lock_guard lock(driver_mutex); - if (workers_stop_requested || remountTerminal()) - return; - ++remount_requested_generation; - driver_cv.notify_all(); -} - void CasMountRuntime::tripAndRequestRemount() { schedule_remount_calls_for_test.fetch_add(1, std::memory_order_relaxed); std::lock_guard lock(driver_mutex); /// Atomics only. tripMountLost(); - if (workers_stop_requested || remountTerminal()) + if (stopOrTerminal()) return; if (lossNeedsNewRequest()) ++remount_requested_generation; @@ -1012,9 +990,14 @@ void CasMountRuntime::tripAndRequestRemount() bool CasMountRuntime::scheduleRemountForTest() { - scheduleRemount(); + schedule_remount_calls_for_test.fetch_add(1, std::memory_order_relaxed); std::lock_guard lock(driver_mutex); - return workers_started && remount_requested_generation > remount_handled_generation; + if (!stopOrTerminal()) + { + ++remount_requested_generation; + driver_cv.notify_all(); + } + return workers_started && remountRequestPending(); } void CasMountRuntime::beginShutdownForTest() diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index fa5b4f06c248..b02dc40318f7 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -373,16 +373,16 @@ class CasMountRuntime void renewerReset(); void startBackgroundWorkers(); void stopBackgroundWorkers(); - /// Latch a recovery generation for the lease thread. It never constructs a thread. - void scheduleRemount(); /// One interference report: trip the fence and request a remount in one `driver_mutex` step, so no /// reclaim can arm between the two. Raises no generation when a request that no reclaim has started /// serving is pending: the reclaim that serves it latches after this trip. void tripAndRequestRemount(); + /// Request a remount without a trip, unless a stop or a terminal lifecycle refuses it. Returns whether + /// a lease thread runs and a request is pending. bool scheduleRemountForTest(); void beginShutdownForTest(); - /// Return how many remount requests were attempted, refused ones included: `scheduleRemount`, - /// `tripAndRequestRemount` and terminal renewals. This is useful for testing the renewer's loss callback without starting a real recovery. + /// Return how many remount requests were attempted, refused ones included: `scheduleRemountForTest`, + /// `tripAndRequestRemount` and terminal renewals. uint64_t scheduleRemountCallCountForTest() const { return schedule_remount_calls_for_test.load(std::memory_order_relaxed); @@ -453,6 +453,10 @@ class CasMountRuntime /// request that no reclaim has snapshotted is pending: the reclaim that serves it latches after the /// loss. bool lossNeedsNewRequest() const; + /// A remount request no reclaim has acknowledged. Requires `driver_mutex`. + bool remountRequestPending() const; + /// A stop is requested or the lifecycle is terminal. Requires `driver_mutex`. + bool stopOrTerminal() const; /// Throws `LOGICAL_ERROR` when a lease thread runs and the caller is not it. Requires `driver_mutex`. void checkRenewerOwner() const; std::unique_lock lockTerminalPublication(); @@ -502,7 +506,7 @@ class CasMountRuntime /// The requested generation `beginReclaim` recorded; `armIfAdmissible` acknowledges it. uint64_t reclaim_generation = 0; ThreadFromGlobalPool renewal_worker; - /// Counted entries into `scheduleRemount`; retained as a test-only observability seam. + /// Read only by `scheduleRemountCallCountForTest`. std::atomic schedule_remount_calls_for_test{0}; /// Local write fence. The unarmed default (`deadline_boot_ms = UINT64_MAX`, `lost = false`) permits diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp index 3e54b9d56a24..8c05090d5e2c 100644 --- a/src/Disks/tests/gtest_cas_pool.cpp +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -3416,7 +3416,7 @@ TEST(CASPoolRemount, ReclaimWaitsForTheRenewalToEnd) runtime.startBackgroundWorkers(); renewal_barrier.waitUntilArrived(); runtime.tripMountLost(); - runtime.scheduleRemount(); + runtime.scheduleRemountForTest(); EXPECT_EQ(remount_calls.load(), 0u) << "the reclaim must wait until the renewal in flight ends"; renewal_barrier.release(); remount_barrier.waitUntilArrived(); @@ -3647,7 +3647,7 @@ TEST(CASPoolRemount, NaturalTerminalTransitionMakesTheLeaseThreadExit) runtime.armMountFence(anchor + 1000); runtime.startBackgroundWorkers(); runtime.tripMountLost(); - runtime.scheduleRemount(); + runtime.scheduleRemountForTest(); transitioned.waitUntilArrived(); transitioned.release(); const bool exited_without_stop = exits.waitForAtLeast(1); @@ -3741,7 +3741,7 @@ TEST(CASPoolRemount, ALeaseWaitCannotMissNaturalTerminalPublication) { /// A request before the start makes the first pass a reclaim, which fails and backs off. runtime.tripMountLost(); - runtime.scheduleRemount(); + runtime.scheduleRemountForTest(); } runtime.startBackgroundWorkers(); @@ -3907,7 +3907,7 @@ TEST(CASPoolRemount, ExternalLossDuringRenewalUsesOneRecoveryGeneration) runtime.startBackgroundWorkers(); renewal_barrier.waitUntilArrived(); runtime.tripMountLost(); - runtime.scheduleRemount(); + runtime.scheduleRemountForTest(); renewal_barrier.release(); remount_barrier.waitUntilArrived(); EXPECT_EQ(remount_calls.load(), 1u); @@ -3947,9 +3947,9 @@ TEST(CASPoolRemount, ConcurrentRemountRequestIsProcessedAfterActiveGeneration) runtime.armMountFence(anchor + 1000); runtime.startBackgroundWorkers(); runtime.tripMountLost(); - runtime.scheduleRemount(); + runtime.scheduleRemountForTest(); first.waitUntilArrived(); - runtime.scheduleRemount(); + runtime.scheduleRemountForTest(); first.release(); second.waitUntilArrived(); EXPECT_EQ(calls.load(), 2u); @@ -4009,7 +4009,7 @@ TEST(CASPoolRemount, ImmediatePostRemountRenewalFailureIsNotDropped) runtime.armMountFence(anchor + 10'000); runtime.startBackgroundWorkers(); runtime.tripMountLost(); - runtime.scheduleRemount(); + runtime.scheduleRemountForTest(); first.waitUntilArrived(); first.release(); second.waitUntilArrived(); @@ -4498,7 +4498,7 @@ TEST(CASMountRuntime, NothingArmsWhileARemountRequestIsPending) const uint64_t lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); runtime.startBackgroundWorkers(); /// A request without a trip: the fence is still armed when the reclaim starts. - runtime.scheduleRemount(); + runtime.scheduleRemountForTest(); second_done.waitUntilArrived(); EXPECT_TRUE(may_mutate_before_first_latch) << "the request alone must leave the fence armed"; @@ -4603,7 +4603,7 @@ TEST(CASMountRuntime, AnInterferenceReportDuringAReclaimIsServedByTheNextReclaim runtime.armMountFence(anchor + 1000); const uint64_t lost_before = ProfileEvents::global_counters[ProfileEvents::CASMountLeaseLost].load(); runtime.startBackgroundWorkers(); - runtime.scheduleRemount(); + runtime.scheduleRemountForTest(); second_done.waitUntilArrived(); EXPECT_FALSE(report_never_started.load()) << "the report did not start within the bound"; @@ -4679,7 +4679,7 @@ TEST(CASMountRuntime, AReclaimAcknowledgesOnlyTheGenerationItServed) runtime.startBackgroundWorkers(); /// One interference report. runtime.tripMountLost(); - runtime.scheduleRemount(); + runtime.scheduleRemountForTest(); reclaimed.waitUntilArrived(); EXPECT_EQ(calls.load(), 3u); @@ -4742,7 +4742,7 @@ TEST(CASMountRuntime, AReclaimFinishedAfterTheForgetIntentArmsNothing) runtime.armMountFence(anchor + 1000); runtime.startBackgroundWorkers(); runtime.tripMountLost(); - runtime.scheduleRemount(); + runtime.scheduleRemountForTest(); latched.waitUntilArrived(); runtime.publishVanishedIntent(); @@ -4950,7 +4950,7 @@ TEST(CASMountRuntime, ARemountRequestEndsTheRenewalAndTheSameThreadReclaims) const uint64_t waits_at_request = waits.load(); /// An interference report while the renewal retries past its lease. runtime.tripMountLost(); - runtime.scheduleRemount(); + runtime.scheduleRemountForTest(); holding.release(); renewed.waitUntilArrived(); @@ -5040,7 +5040,7 @@ TEST(CASMountRuntime, ACommitConsumedAfterARemountRequestCannotOverwriteTheRecla .renewal_live_for_test = [&] { if (committed.load() && !requested.exchange(true)) - runtime_ptr->scheduleRemount(); + runtime_ptr->scheduleRemountForTest(); return true; }}, "test", sink, runtimeRenewBudget(), [&] @@ -6650,7 +6650,7 @@ TEST(CASMountRuntime, ARenewalArmsNothingWhileARequestIsPending) { request_raised = true; runtime_ptr->tripMountLost(); - runtime_ptr->scheduleRemount(); + runtime_ptr->scheduleRemountForTest(); } return true; }}, From 8920c7619ba2fbcc2218a91e0c979189bc608879 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 11:08:21 +0200 Subject: [PATCH 35/37] Bound the lease-keeper tests that could hang; pin two untested paths - `ADefinitiveAnswerDuringReadinessIsServedByAReclaim` bounds the open's wait by its fence-clock reads, so a lost remount request fails the test instead of leaving the wait on a frozen clock. - `ForgetRacingActiveRemountThreadCompletesBounded` ends the binary on a FORGET deadlock instead of joining a thread that never returns. - `ScheduledRoundStealsADeadIncumbent`: a scheduled GC round takes over a dead leader's lease. - `AnOpenFailsWhenThePoolBecomesTerminalWhileItWaits`: the open's failure text for a pool that became terminal. - `SentinelsDeletedEntersIdentityLostTerminal` also checks the fence is latched; `CancellationAfterSendIsTerminalAndForbidsRelease` expects exactly one request. Co-Authored-By: Claude Opus 5.5 --- src/Disks/tests/gtest_cas_forget.cpp | 12 ++- src/Disks/tests/gtest_cas_gc_log.cpp | 45 ++++++++++ src/Disks/tests/gtest_cas_heartbeat.cpp | 2 +- .../tests/gtest_cas_lifecycle_condition.cpp | 1 + src/Disks/tests/gtest_cas_pool.cpp | 87 ++++++++++++++++++- 5 files changed, 143 insertions(+), 4 deletions(-) diff --git a/src/Disks/tests/gtest_cas_forget.cpp b/src/Disks/tests/gtest_cas_forget.cpp index f8e9f566f03c..c8af7dc950a9 100644 --- a/src/Disks/tests/gtest_cas_forget.cpp +++ b/src/Disks/tests/gtest_cas_forget.cpp @@ -13,6 +13,8 @@ #include #include #include +#include +#include #include #include #include @@ -385,8 +387,14 @@ TEST(CASForget, ForgetRacingActiveRemountThreadCompletesBounded) store->forgetDisk([] {}, kForgetReason); done.set_value(); }); - EXPECT_EQ(fut.wait_for(std::chrono::seconds(30)), std::future_status::ready) - << "FORGET must not deadlock against an in-flight self-remount"; + if (fut.wait_for(std::chrono::seconds(30)) != std::future_status::ready) + { + ADD_FAILURE() << "FORGET must not deadlock against an in-flight self-remount"; + /// Nothing can release the deadlocked threads, and the process does not exit while they are + /// parked, so the binary ends here instead of hanging. + std::fflush(stdout); + std::_Exit(1); + } forgetter.join(); /// Disarm before ~Pool so its residual teardown is not fighting the injected fault. diff --git a/src/Disks/tests/gtest_cas_gc_log.cpp b/src/Disks/tests/gtest_cas_gc_log.cpp index d0c384fff88a..84791c8b149c 100644 --- a/src/Disks/tests/gtest_cas_gc_log.cpp +++ b/src/Disks/tests/gtest_cas_gc_log.cpp @@ -10,6 +10,7 @@ #include #include +#include #include #include #include @@ -273,6 +274,50 @@ TEST(CASGCSchedulerSteal, ManualRoundNeverStealsALiveHeartbeatingIncumbent) EXPECT_FALSE(sched.runOneRoundNow().acquired_lease); } +/// The loop's side of `ManualRoundNeverStealsEvenADeadIncumbent`: its scheduled rounds see the dead +/// incumbent's lease tuple and heartbeat frozen across two of their own observations, and the second one +/// takes the lease over. +TEST(CASGCSchedulerSteal, ScheduledRoundStealsADeadIncumbent) +{ + constexpr size_t kRoundBound = 4; + auto backend = std::make_shared(); + auto store = Pool::open(backend, PoolConfig{.pool_prefix = "p", .server_root_id = "test"}); + + const UInt128 kIncumbent = hexToU128("00000000000000000000000000000abc"); + Gc incumbent(store, kIncumbent); + ASSERT_TRUE(incumbent.runRegularRound().acquired_lease); + + std::mutex rows_mutex; + std::condition_variable rows_cv; + std::vector finishes; + DB::Cas::CasGcScheduler sched( + store, std::chrono::seconds(1), "test::gc", "ca", + [&](const Rec & r) + { + if (r.event_type != Rec::EventType::Finish) + return; + std::lock_guard g(rows_mutex); + finishes.push_back(r); + rows_cv.notify_all(); + }); + + sched.start(); + bool stolen = false; + for (size_t round = 1; round <= kRoundBound && !stolen; ++round) + { + sched.requestRoundSoon(); + std::unique_lock lock(rows_mutex); + ASSERT_TRUE(rows_cv.wait_for(lock, std::chrono::seconds(30), [&] { return finishes.size() >= round; })) + << "timed out waiting for Finish row #" << round; + for (const Rec & r : finishes) + stolen = stolen || r.outcome == Rec::Outcome::Success || r.outcome == Rec::Outcome::Deferred; + } + sched.stop(); + for (const Rec & r : finishes) + EXPECT_EQ(r.trigger, Rec::Trigger::Scheduled); + EXPECT_TRUE(stolen) << "no scheduled round in " << kRoundBound << " took over the dead incumbent's lease"; +} + /// A round whose backend throws must produce a Finish with `outcome == Aborted` and a non-empty /// `error`, and `runOneRoundNow` must rethrow the exception (the round failure is observable, not /// swallowed — the logging sink itself is best-effort, but the round error propagates). diff --git a/src/Disks/tests/gtest_cas_heartbeat.cpp b/src/Disks/tests/gtest_cas_heartbeat.cpp index bbd71b16ce55..da33b3406b65 100644 --- a/src/Disks/tests/gtest_cas_heartbeat.cpp +++ b/src/Disks/tests/gtest_cas_heartbeat.cpp @@ -1151,7 +1151,7 @@ TEST(CASHeartbeat, CancellationAfterSendIsTerminalAndForbidsRelease) renewalEnvironment(/*live=*/[&] { return !cancelled; }, /*cancelled=*/[&] { return cancelled; })); const DB::Exception failure = terminalException(result); EXPECT_EQ(failure.code(), DB::ErrorCodes::NETWORK_ERROR); - EXPECT_GE(result.attempts_sent, 1u); + EXPECT_EQ(result.attempts_sent, 1u); EXPECT_EQ(backend->read_calls, 0u) << "post-write cancellation must not start a diagnostic read"; EXPECT_EQ(renewer.state(), MountLeaseRenewerState::RenewalTerminal); const String bytes_before = ops.op.read(layout.mountKey("test"), Retry::standard())->bytes; diff --git a/src/Disks/tests/gtest_cas_lifecycle_condition.cpp b/src/Disks/tests/gtest_cas_lifecycle_condition.cpp index 8d6133b75592..ba9970a5ef8f 100644 --- a/src/Disks/tests/gtest_cas_lifecycle_condition.cpp +++ b/src/Disks/tests/gtest_cas_lifecycle_condition.cpp @@ -137,6 +137,7 @@ TEST(CASLifecycleCondition, SentinelsDeletedEntersIdentityLostTerminal) EXPECT_FALSE(store->tryRemountOnce()); EXPECT_EQ(store->lifecycle(), PoolLifecycle::IdentityLost); EXPECT_FALSE(store->isVanished()) << "IdentityLost is a distinct terminal, not a Vanished state"; + EXPECT_FALSE(store->mayMutate()) << "the fence stays latched"; /// store()-class access now fails loud with the typed lifecycle error. DB::Cas::tests::expectThrowsCode(DB::ErrorCodes::INVALID_STATE, [&] { store->throwIfLifecycleTerminal(); }); diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp index 8c05090d5e2c..8ba0713b5693 100644 --- a/src/Disks/tests/gtest_cas_pool.cpp +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -58,6 +58,7 @@ extern const Event CASRemountFailed; extern const Event CASMountLeaseExpired; extern const Event CASMountRenewalAttempts; extern const Event CASMountRenewalRetries; +extern const Event CASIdentityLost; } using namespace DB::Cas; @@ -7003,6 +7004,71 @@ TEST(CASPool, AnOpenWithoutALeaseThreadFailsWhenItsClaimIsTooOld) EXPECT_EQ(eventCount(ProfileEvents::CASMountRenewalAttempts), attempts_before) << "nothing renewed the claim"; } +/// The pool becomes terminal while the open waits: the readiness renewal meets a fenced-out slot, and the +/// reclaim it requests finds both pool sentinels gone and enters `IdentityLost`. The open fails naming the +/// terminal pool. Every later write of the mount key moves the fence clock a lease, so a regression that +/// keeps the pool going ends the open's wait instead of hanging. +TEST(CASMountRuntime, AnOpenFailsWhenThePoolBecomesTerminalWhileItWaits) +{ + auto fake_boot = std::make_shared>(kReadyClaimBootMs); + auto backend = std::make_shared(); + seedFencedPredecessor(*backend, "ready-terminal"); + const Layout layout("ready-terminal"); + const String mount_key = layout.mountKey("s"); + const String meta_key = layout.poolMetaKey(); + const String owner_key = layout.ownerKey("s"); + backend->mount_key = mount_key; + PoolConfig config = readinessPoolConfig("ready-terminal", fake_boot); + const uint64_t ttl_ms = static_cast(config.mount_lease_ttl_ms.count()); + /// The hook is stored in the backend, so it holds a plain pointer to it. + DB::Cas::Backend * raw_backend = backend.get(); + /// 1, 2: the open's reclaim and adopt. 3: the readiness renewal. 4: the fence-out made inside it. + backend->on_mount_write = [fake_boot, raw_backend, mount_key, meta_key, owner_key, ttl_ms](uint32_t write_no) + { + if (write_no == 2) + { + *fake_boot += kReadyClaimAgeMs; + } + else if (write_no == 3) + { + fenceOutMount(*raw_backend, mount_key); + DB::Cas::tests::OperationForTest op(*raw_backend); + for (const String & key : {meta_key, owner_key}) + { + const auto got = (*op).read(key, Retry::once()); + EXPECT_TRUE(got.has_value()) << key; + if (got) + (*op).remove(key, got->etag, Retry::once()); + } + } + else if (write_no > 4) + { + *fake_boot += ttl_ms; + } + return false; + }; + const uint64_t identity_lost_before = eventCount(ProfileEvents::CASIdentityLost); + + int code = 0; + String message; + try + { + (void)Pool::open(backend, config); + ADD_FAILURE() << "an open whose pool became terminal must fail"; + } + catch (const DB::Exception & e) + { + code = e.code(); + message = e.message(); + } + + EXPECT_EQ(code, DB::ErrorCodes::ABORTED) << message; + EXPECT_NE(message.find("the pool became terminal while the open waited"), String::npos) << message; + EXPECT_EQ(message.find("no renewal armed the fence"), String::npos) << message; + EXPECT_EQ(backend->mount_writes.load(), 4u) << "no mount write after the fence-out"; + EXPECT_EQ(eventCount(ProfileEvents::CASIdentityLost), identity_lost_before + 1); +} + namespace { /// What the open's wait and the reclaim saw when the readiness renewal met a definitive answer. @@ -7019,8 +7085,13 @@ struct DefinitiveReadinessSeen uint64_t live_writer_epoch = 0; uint64_t requested_generation = 0; uint64_t lease_lost = 0; + uint32_t wait_clock_reads = 0; }; +/// The open's wait reads the fence clock once per wake-up. On a regression that leaves it unnotified the +/// injected clock stands still, so at this many reads the fixture moves the clock past the wait's bound. +constexpr uint32_t kReadinessWaitWakeUpBound = 20; + /// The open's shape at the runtime level, with a lease of 1000 ms: the claim starts at 100 ms and the open /// decides at 1070 ms, with 30 ms left and 40 ms needed. The slot is GC-fenced before the loop starts, so the /// readiness renewal gets a definitive answer. The reclaim either claims epoch 2 and arms, or fails each @@ -7033,6 +7104,9 @@ void runDefinitiveAnswerDuringReadiness(bool reclaim_arms, DefinitiveReadinessSe std::atomic boot_ms{100}; std::atomic puts{0}; std::atomic reclaims{0}; + const std::thread::id test_thread = std::this_thread::get_id(); + std::atomic waiting{false}; + std::atomic wait_clock_reads{0}; CasMountRuntime * runtime_ptr = nullptr; const uint64_t lost_before = eventCount(ProfileEvents::CASMountLeaseLost); CasEventSink sink; @@ -7045,7 +7119,13 @@ void runDefinitiveAnswerDuringReadiness(bool reclaim_arms, DefinitiveReadinessSe MountConfig{ .mount_lease_ttl_ms = std::chrono::milliseconds(1000), .mount_renew_period = std::chrono::milliseconds(300), .background_watermark = true, - .boot_ms_fn = [&] { return boot_ms.load(); }}, + .boot_ms_fn = [&] + { + if (waiting.load() && std::this_thread::get_id() == test_thread + && ++wait_clock_reads == kReadinessWaitWakeUpBound) + boot_ms.fetch_add(2000); + return boot_ms.load(); + }}, "test", sink, runtimeRenewBudget(), [&] { CasMountRuntime & reclaiming = *runtime_ptr; @@ -7095,8 +7175,11 @@ void runDefinitiveAnswerDuringReadiness(bool reclaim_arms, DefinitiveReadinessSe runtime.startBackgroundWorkers(); seen.boot_before_wait = boot_ms.load(); + waiting = true; seen.armed = runtime.waitUntilArmed(1000); + waiting = false; seen.boot_after_wait = boot_ms.load(); + seen.wait_clock_reads = wait_clock_reads.load(); /// The join orders every write of the lease thread before the reads below. runtime.stopBackgroundWorkers(); @@ -7202,6 +7285,7 @@ TEST(CASMountRuntime, ADefinitiveAnswerDuringReadinessIsServedByAReclaim) { DefinitiveReadinessSeen reclaimed; ASSERT_NO_FATAL_FAILURE(runDefinitiveAnswerDuringReadiness(/*reclaim_arms=*/true, reclaimed)); + EXPECT_LT(reclaimed.wait_clock_reads, kReadinessWaitWakeUpBound) << "nothing ended the wait: it ran out its wake-ups"; EXPECT_TRUE(reclaimed.armed) << "the reclaim armed the fence"; EXPECT_EQ(reclaimed.puts_at_first_reclaim, 1u) << "the loop reclaimed before any further renewal"; EXPECT_EQ(reclaimed.admit_at_first_reclaim, "LostOrRearmed") << "no write is admitted before the arm"; @@ -7216,6 +7300,7 @@ TEST(CASMountRuntime, ADefinitiveAnswerDuringReadinessIsServedByAReclaim) DefinitiveReadinessSeen failing; ASSERT_NO_FATAL_FAILURE(runDefinitiveAnswerDuringReadiness(/*reclaim_arms=*/false, failing)); + EXPECT_LT(failing.wait_clock_reads, kReadinessWaitWakeUpBound) << "nothing ended the wait: it ran out its wake-ups"; EXPECT_FALSE(failing.armed); EXPECT_GE(failing.boot_after_wait, failing.boot_before_wait + 1000) << "the wait gave up after one lease on the fence clock"; EXPECT_EQ(failing.puts_at_first_reclaim, 1u); From 98ca1ad3d5d7f829e1df2ed4db2684b814edaa40 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 11:20:41 +0200 Subject: [PATCH 36/37] Make lease-keeper comments, counters and docs say what the code does - `claimed_not_armed` means the claim's arming conditions did not hold; what follows is a renewal, another reclaim or the thread's exit. - `CASRemountSucceeded`, `CASRemountFailed` and `CASMountLeaseLost` describe what counts them. - `cas_unsafe_remount_no_delay`: a conditional write reserves the write and its settling read; a single-envelope request starts one envelope later. - Troubleshooting covers a lease thread that ended on its own error path. - Stale names and old tags removed; two tests renamed to what they test. Co-Authored-By: Claude Opus 5.5 --- .../cas/architecture/mounts-and-leases.md | 12 +++--- docs/en/antalya/cas/configuration.md | 2 +- docs/en/antalya/cas/operations/debugging.md | 6 ++- docs/en/antalya/cas/operations/monitoring.md | 8 ++-- .../antalya/cas/operations/troubleshooting.md | 6 +++ src/Common/ProfileEvents.cpp | 6 +-- .../ContentAddressedSettings.cpp | 2 +- .../ContentAddressed/Gc/CasGcScheduler.cpp | 12 +++--- .../ContentAddressed/Pool/CasMountRuntime.cpp | 6 +-- .../ContentAddressed/Pool/CasMountRuntime.h | 36 ++++++++-------- .../ContentAddressed/Pool/CasPool.cpp | 17 ++++---- .../ContentAddressed/Pool/CasPool.h | 43 ++++++++----------- .../ContentAddressed/Pool/CasRefLedger.h | 8 ++-- .../ContentAddressed/Pool/CasServerRoot.h | 4 +- .../tests/gtest_cas_lifecycle_condition.cpp | 17 +++----- src/Disks/tests/gtest_cas_pool.cpp | 14 +++--- src/Disks/tests/gtest_cas_ref_writer.cpp | 10 ++--- 17 files changed, 105 insertions(+), 104 deletions(-) diff --git a/docs/en/antalya/cas/architecture/mounts-and-leases.md b/docs/en/antalya/cas/architecture/mounts-and-leases.md index e6fc591864e4..cca84a80f031 100644 --- a/docs/en/antalya/cas/architecture/mounts-and-leases.md +++ b/docs/en/antalya/cas/architecture/mounts-and-leases.md @@ -271,14 +271,16 @@ TTL, still leaves room for a ref append: `2 × envelope + margin` (16 s with the fence stays latched, the lease thread renews at once, and the first renewal whose own deadline leaves that room arms the fence; then the open returns. A renewal that commits with less room arms nothing, and the next one follows at once. If no renewal arms the fence within one TTL, the open stops and joins the lease -thread and fails with `ABORTED`, naming the last failed renewal request. A reclaim follows the same rule: a -claim with too little lease left keeps the pool `TransientNotLive` with the fence latched, and the next -renewal arms the fence and reports `Live`. The `mount_remount` row of such a reclaim has `outcome = 'ok'` +thread and fails with `ABORTED`, naming the last failed renewal request. A reclaim whose arming conditions +(below) do not hold keeps the pool `TransientNotLive` with the fence latched. Then the next renewal arms +the fence and reports `Live` when no request is pending, a newer request gets another reclaim, and a stop +or a terminal lifecycle ends the thread. The `mount_remount` row of such a reclaim has `outcome = 'ok'` and `step = 'claimed_not_armed'`. One lease thread per writable mount renews the lease and runs the self-remount, one after the other. -`scheduleRemount` increments a requested-generation latch and wakes the thread; a pending request also -ends a renewal in progress. While the thread runs, only it replaces, starts or resets the renewer. +An interference report (`tripAndRequestRemount`) and a terminal renewal trip the fence, raise the +requested remount generation unless a request no reclaim has started serving is already pending or the +pool is stopping or terminal, and wake the thread; a pending request also ends a renewal in progress. While the thread runs, only it replaces, starts or resets the renewer. - A reclaim latches the fence first. - It arms the fence, and reports `Live`, only when no newer request is pending, no stop is requested diff --git a/docs/en/antalya/cas/configuration.md b/docs/en/antalya/cas/configuration.md index 1d37513a8f0a..cba267215e8d 100644 --- a/docs/en/antalya/cas/configuration.md +++ b/docs/en/antalya/cas/configuration.md @@ -109,7 +109,7 @@ entirely before release. Treat this table as a snapshot of the current build, no | `cas_gc_io_concurrency` | `16` | Bounded pool size for GC object-storage requests that run in parallel: the fold's read-ahead (checkpoints, ref logs, manifests, zero-candidate HEADs), the orphan-manifest sweep planning reads, the `SYSTEM CAS GC REBUILD` read-ahead, and the `pending_deletes` blob `HEAD` + conditional `DELETE` fan-out. Not covered: meta writes (`cas_gc_meta_pool_size`) and all other GC requests, which run on the round thread. `1` runs the covered requests sequentially. `cas_gc_read_concurrency` is rejected without an alias; use `cas_gc_io_concurrency` instead | | `cas_attempt_timeout_ms` | `5000` | Budget for one HTTP attempt of a writable Native mount's control-plane requests (read, head, list, remove, conditional write), at least 1. Together with the connect cap it forms the attempt envelope (`cas_attempt_timeout_ms + 2 × cap`; the cap is `cas_attempt_timeout_ms` itself when the disk's `connect_timeout_ms` is `0`, else `min(connect_timeout_ms, cas_attempt_timeout_ms)`) that the lease arithmetic reserves: one TCP connect and one TLS handshake under the cap each, send/receive bounded per socket operation by `cas_attempt_timeout_ms`. With background renewal the cadence check requires `cas_mount_renew_period_ms + 2 × envelope + cas_lease_safety_margin_ms < cas_mount_lease_ttl_ms`, which puts an effective ceiling on the frozen connect cap: under the defaults (TTL 30000, period 10000, margin 2000) the envelope must stay under 9000, so a disk `connect_timeout_ms` of 2000 ms or more refuses to open writable — lower the connect timeout or raise the TTL if you hit this | | `cas_lease_safety_margin_ms` | `2000` | Room kept between a request and the mount lease deadline: a write is admitted only while the requests it may send plus this margin fit in the remaining lease. Validated against the mount lease TTL when a writable mount opens: the attempt envelope + `cas_lease_safety_margin_ms` must be strictly less than the mount lease TTL, and `cas_mount_renew_period_ms` + 2 × envelope + `cas_lease_safety_margin_ms` too, or the disk refuses to open writable | -| `cas_unsafe_remount_no_delay` | `0` | Reclaim a mount slot that carries this server's own uuid at once after a hard restart, without observing the slot's token for the lease TTL. Unsafe whenever two processes can hold the same `server_uuid` (a copied uuid file, a stalled predecessor). After such a reclaim the predecessor can still start conditional writes until its own cutoff (`confirmed deadline − cas_lease_safety_margin_ms − 2 × envelope`) or until its next renewal meets the token guard, and a request it already sent may still materialize later. That is not a data hazard: ref-log keys carry `(writer_epoch, sequence)` and creates are conditional, so two writers can never commit different bodies to one key, and recovery's epoch seal settles any straggler (recovery fails closed after 64 successive seal-create attempts displaced by newly materializing old-epoch transactions). The exposure is availability, not data. Intended for test stands and deployments that guarantee one process per uuid | +| `cas_unsafe_remount_no_delay` | `0` | Reclaim a mount slot that carries this server's own uuid at once after a hard restart, without observing the slot's token for the lease TTL. Unsafe whenever two processes can hold the same `server_uuid` (a copied uuid file, a stalled predecessor). After such a reclaim, unless its next renewal meets the token guard first, the predecessor can still start a conditional write, a ref-log append included, until its own cutoff (`confirmed deadline − cas_lease_safety_margin_ms − 2 × envelope`: the write and the read that settles it), and a single-envelope request such as a conditional delete until one envelope later. A request it already sent may still materialize later. That is not a data hazard: ref-log keys carry `(writer_epoch, sequence)` and creates are conditional, so two writers can never commit different bodies to one key, and recovery's epoch seal settles any straggler (recovery fails closed after 64 successive seal-create attempts displaced by newly materializing old-epoch transactions). The exposure is availability, not data. Intended for test stands and deployments that guarantee one process per uuid | | `cas_staging_backend` | `local` | Blob staging backend (`local` \| `s3`); `s3` is opt-in and requires native same-store copy on writable mount | All servers sharing a pool must run the same `cas_mount_lease_ttl_ms` and `cas_mount_renew_period_ms`. diff --git a/docs/en/antalya/cas/operations/debugging.md b/docs/en/antalya/cas/operations/debugging.md index b1184ddae67b..bf890886b8db 100644 --- a/docs/en/antalya/cas/operations/debugging.md +++ b/docs/en/antalya/cas/operations/debugging.md @@ -147,8 +147,10 @@ follows: differs by classification. - A following `mount_remount` row names the whole-chain `attempt_no` and final `step`. An `ok` row with `step = 'publish_live'` restored `Live` under the reported fresh `writer_epoch`; with - `step = 'claimed_not_armed'` the claim succeeded with too little lease to admit a write, and the next - renewal arms the fence and reports `Live`. A `failed` row's `step` and optional `error` identify where + `step = 'claimed_not_armed'` the claim succeeded but its arming conditions did not hold (too little + lease left, a newer remount request, a stop, or a terminal lifecycle), and the fence stays latched. The + next renewal arms it and reports `Live` when no request is pending; a newer request gets another + reclaim; a stop or a terminal lifecycle ends the lease thread. A `failed` row's `step` and optional `error` identify where that whole-chain attempt stopped. Use deltas of the mount counters from diff --git a/docs/en/antalya/cas/operations/monitoring.md b/docs/en/antalya/cas/operations/monitoring.md index 6acf51a5b0cf..d8235e1bbf38 100644 --- a/docs/en/antalya/cas/operations/monitoring.md +++ b/docs/en/antalya/cas/operations/monitoring.md @@ -61,7 +61,7 @@ window and correlate them with the `server_root_id` in `system.cas_log`. | `CASMountRenewalRecovered` | One per logical renewal committed after a retry or exact resolving `GET` | Recovered object-store blips that retained the existing mount incarnation | | `CASMountLeaseExpired` | One per renewal that restored a lease that had expired | Moves at the restore, not when the lease expires. While it is expired, `system.cas_mounts` shows `lifecycle_reason = 'lease_expired'` and writes are refused; the `watermark_renew` row of the restoring renewal carries `expired_ms` | | `CASRemountAttempts` | One per invocation of the existing whole-chain remount attempt | Includes both successful and failed attempts | -| `CASRemountSucceeded` | One per whole-chain attempt that completed every step through the arm step, under a fresh writer epoch | Must be a subset of `CASRemountAttempts`. Includes a reclaim whose claim left too little lease to arm the fence (`step = 'claimed_not_armed'`); the next renewal arms it | +| `CASRemountSucceeded` | One per whole-chain attempt that completed every step through the arm step, under a fresh writer epoch | Must be a subset of `CASRemountAttempts`. Includes a reclaim that claimed but whose arming conditions did not hold (`step = 'claimed_not_armed'`), with the fence still latched | | `CASRemountFailed` | One per whole-chain attempt that stopped before the arm step | Includes a named step exception or a step that returned transiently, also after the mount claim succeeded (for example at `renewer_start` or `quiesce_ref_tables`) | `CASMountLeaseLost` complements those counters. It increments exactly once per operational @@ -90,8 +90,10 @@ Ordinary first-attempt success produces no row. Every `mount_remount` attempt pr `failed` and details `attempt_no`, `step`, `server_root_id`, optional `writer_epoch`, and optional `error`. An `ok` row has `step = 'publish_live'` when the reclaim armed the fence and reported `Live`, or -`step = 'claimed_not_armed'` when its claim left too little lease to admit a write; the pool then stays -`not_live` until the next renewal arms the fence. +`step = 'claimed_not_armed'` when it claimed but its arming conditions did not hold: too little lease left +for a write, a newer remount request, a stop, or a terminal lifecycle. The fence stays latched. What follows +is the lease thread's next renewal when no request is pending, another reclaim when a newer request is, or +the thread's exit on a stop or a terminal lifecycle. Default-level text logging is bounded per logical operation. A renewal logs nothing until it ends, and nothing at all when it succeeds on its first request. A renewal that needed a retry, was settled diff --git a/docs/en/antalya/cas/operations/troubleshooting.md b/docs/en/antalya/cas/operations/troubleshooting.md index 670c9454f121..baf31ce3386b 100644 --- a/docs/en/antalya/cas/operations/troubleshooting.md +++ b/docs/en/antalya/cas/operations/troubleshooting.md @@ -65,6 +65,12 @@ Start with the `watermark_renew` timeline described in and optional `error` identify the failed owner/catalog/epoch/claim/install/quiescence/fence step. The current protocol retries the whole chain with bounded backoff; it does not preserve per-step progress. Repeated failure at the same step is the actionable signal. +6. **Lease thread ended on its own error path.** The server log has an `ERROR` line from the `CasPool` + logger that starts with `CAS mount-lease renewal loop`, with the exception that ended the thread. + The disk shows `lifecycle = 'not_live'` in `system.cas_mounts`, every write is refused, and no + `watermark_renew` or `mount_remount` row follows: nothing renews, reclaims or reopens the mount. + Restart the server, and report the logged exception as a defect: the lease keeper exits this way + only when its own state machine broke its contract. The default-level log policy is intentionally bounded. The lease keeper logs nothing about a renewal until the renewal ends. A renewal that succeeds on its first request logs nothing. A renewal that diff --git a/src/Common/ProfileEvents.cpp b/src/Common/ProfileEvents.cpp index 71f905b1a297..94a6e851276a 100644 --- a/src/Common/ProfileEvents.cpp +++ b/src/Common/ProfileEvents.cpp @@ -956,11 +956,11 @@ The server successfully detected this situation and will download merged part fr M(CASMountRenewalRetries, "Number of physical conditional renewal PUTs sent after the first attempt of one logical CAS mount-lease renewal.", ValueType::Number) \ M(CASMountRenewalResolved, "Number of CAS mount-lease renewals whose committed outcome was proved by an exact resolving GET.", ValueType::Number) \ M(CASMountRenewalRecovered, "Number of logical CAS mount-lease renewals that committed after a physical retry or exact resolving GET.", ValueType::Number) \ - M(CASMountLeaseLost, "Number of Live-to-TransientNotLive transitions of a CAS mount lease, one per loss. A terminal renewal and an interference report count one, as does any other trip of the fence while no terminal intent is published. A renewal ended by a pending remount request, a stop or a FORGET does not count one.", ValueType::Number) \ + M(CASMountLeaseLost, "Number of Live-to-TransientNotLive transitions of a CAS mount lease. A fence trip counts one when it finds the pool Live and no terminal intent published: a terminal renewal, an interference report, the start of a reclaim, the lease thread's error exit. The open's latch, a stop and a FORGET count nothing, nor does a trip on a pool that is already not Live.", ValueType::Number) \ M(CASMountLeaseExpired, "Number of times a renewal restored a CAS mount lease that had expired. While the lease is expired this server refuses writes and system.cas_mounts shows lifecycle_reason = 'lease_expired'; the watermark_renew event of the restoring renewal carries expired_ms.", ValueType::Number) \ M(CASRemountAttempts, "Number of invocations of the CAS whole-chain remount attempt.", ValueType::Number) \ - M(CASRemountSucceeded, "Number of CAS whole-chain remount attempts that restored Live under a fresh writer epoch.", ValueType::Number) \ - M(CASRemountFailed, "Number of CAS whole-chain remount attempts that returned without restoring Live, including caught step exceptions.", ValueType::Number) \ + M(CASRemountSucceeded, "Number of CAS whole-chain remount attempts that claimed a fresh writer epoch and reached the arm decision. An attempt whose arming conditions did not hold (step `claimed_not_armed`) counts here too, with the fence still latched and the pool not Live.", ValueType::Number) \ + M(CASRemountFailed, "Number of CAS whole-chain remount attempts that returned before the arm decision: stopped by a terminal lifecycle or the pool identity probe, a mount that was not claimable, or a step that threw, before or after the claim (for example the renewer start or the ref-table quiesce).", ValueType::Number) \ M(CASMountReleaseSkippedForeignOccupant, "Counts conclusive CAS mount-renewal observations that found a FOREIGN successor in the slot. This is the EXPECTED end state of a failover: renewal fences the deposed runtime, terminal teardown skips the farewell without backend I/O, and the successor's slot is left byte-for-byte untouched. Steady non-zero values on a cluster that is not failing over are worth investigating; a value that tracks failovers is normal.", ValueType::Number) \ M(CASMountExclusivityViolation, "Counts CAS mount releases where a runtime that still BELIEVED IT OWNED the mount (no deposition ever observed) found a FOREIGN occupant in the slot. This is the single-writer guarantee being broken, not a failover: the release refuses, the slot is left untouched, and the write fence is latched so the runtime stops trusting itself. This must always be zero.", ValueType::Number) \ M(CASRemountHeldTransient, "Counts CAS remount attempts that could not decide the pool's identity and were held transient: the `_pool_meta` probe was inconclusive or its body could not be decoded (a partially written object, an unreadable store, or a pool whose format this build no longer reads). The mount stays fenced closed and the remount loop retries; a value that keeps growing means the pool will never remount without operator action -- read the accompanying warning for the probe's own reason.", ValueType::Number) \ diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.cpp index eed81bee1aca..c015676c8ac5 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/ContentAddressedSettings.cpp @@ -73,7 +73,7 @@ constexpr std::string_view CAS_KEY_PREFIX = "cas_"; DECLARE(UInt64, gc_round_outcome_entry_budget, 5000, "GcOutcomes per-round entry cap across the redelete/spared audit log (0 = unbounded)", 0) \ DECLARE(UInt64, mount_lease_ttl_ms, 30000, "Mount lease validity after a successful claim or renewal, in milliseconds", 0) \ DECLARE(UInt64, mount_renew_period_ms, 10000, "Interval between background mount lease renewals, in milliseconds", 0) \ - DECLARE(Bool, unsafe_remount_no_delay, false, "Reclaim a mount slot that carries this server's own uuid at once after a hard restart, without observing the slot's token for the lease TTL. Unsafe whenever two processes can hold the same server_uuid (a copied uuid file, a stalled predecessor): after such a reclaim the predecessor can still start conditional writes until its own cutoff (confirmed deadline − margin − 2 × envelope) or until its next renewal meets the token guard, and a request it already sent may materialize later. Ref-log keys carry (writer_epoch, sequence) and creates are conditional, so two writers can never commit different bodies to one key, and recovery's epoch seal settles stragglers -- the exposure is availability, not data: recovery fails closed after 64 successive seal-create attempts displaced by newly materializing old-epoch transactions. Intended for test stands and deployments that guarantee one process per uuid", 0) \ + DECLARE(Bool, unsafe_remount_no_delay, false, "Reclaim a mount slot that carries this server's own uuid at once after a hard restart, without observing the slot's token for the lease TTL. Unsafe whenever two processes can hold the same server_uuid (a copied uuid file, a stalled predecessor): after such a reclaim, unless its next renewal meets the token guard first, the predecessor can still start a conditional write, a ref-log append included, until its own cutoff (confirmed deadline − margin − 2 × envelope: the write and the read that settles it), and a single-envelope request such as a conditional delete until one envelope later; a request it already sent may materialize later. Ref-log keys carry (writer_epoch, sequence) and creates are conditional, so two writers can never commit different bodies to one key, and recovery's epoch seal settles stragglers -- the exposure is availability, not data: recovery fails closed after 64 successive seal-create attempts displaced by newly materializing old-epoch transactions. Intended for test stands and deployments that guarantee one process per uuid", 0) \ DECLARE(String, server_root_id, "", "REQUIRED explicit layout subtree identity; macros expand as in the s3 endpoint", 0) \ DECLARE(UInt64, part_folder_cache_bytes, 64ULL << 20, "Part-folder view cache byte budget (0 disables retention)", 0) \ DECLARE(UInt64, part_folder_cache_max_entries, 10000, "Part-folder view cache entry cap", 0) \ diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.cpp index 1e3e1db792b6..e35049671b2c 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Gc/CasGcScheduler.cpp @@ -350,19 +350,19 @@ void CasGcScheduler::loop() return; round_requested = false; } - /// rev.7 §3 [C1] + rev.8 §9 item 8: self-exit the pacing loop the moment the pool reaches — or is + /// Self-exit the pacing loop the moment the pool reaches — or is /// being driven toward — ANY terminal state. A NATURAL terminal transition (`VanishedReplaced` after /// a foreign pool took the prefix, or `IdentityLost` once the sentinels are gone) never calls /// `stop()` on this scheduler: only `~Pool`/FORGET join it. Without this check the loop would tick /// FOREVER — `acquireOrRenewLease` throws `CORRUPTED_DATA` against the vanished `gc/state` every - /// interval (the G2 zombie: an error-log line + a Failed round row each tick), and worse, after + /// interval (an error-log line + a Failed round row each tick), and worse, after /// `VanishedReplaced` the `allow_steal=true` rounds could STEAL the FOREIGN pool's `gc/state` lease /// and fold/condemn/delete its objects. `remountTerminal` is true from the moment FORGET /// publishes its intent (still pre-terminal) and on `IdentityLost` - /// (a fail-loud terminal state; the last G2-zombie case — eternal `CORRUPTED_DATA` retries - /// against a half-erased pool — closes with it). Clearing `i_am_leader` before returning keeps + /// (a fail-loud terminal state, which also ends eternal `CORRUPTED_DATA` retries against a + /// half-erased pool). Clearing `i_am_leader` before returning keeps /// `gcHealth` honest (a terminal, self-exited scheduler reports it no longer leads). The thread exits - /// its OWN loop here — no join from this context (C6-safe); `stop()`/`~CasGcScheduler` still join the + /// its OWN loop here — no join from this context; `stop()`/`~CasGcScheduler` still join the /// finished thread cleanly. if (store->remountTerminal()) { @@ -476,7 +476,7 @@ void CasGcScheduler::heartbeatLoop() { return scheduler_state == SchedulerState::Stopped; })) return; } - /// rev.7 §3 [C1] + rev.8 §9 item 8: self-exit on ANY terminal (or FORGET-intent) pool, same as + /// Self-exit on ANY terminal (or FORGET-intent) pool, same as /// `loop()`. A terminal pool's advisory pulses would target a deleted `gc/hb` key (`IdentityLost`) or /// a FOREIGN pool's key (`VanishedReplaced`) — stop pulsing the moment the pool goes terminal. if (store->remountTerminal()) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp index cca136e70530..5dfaf0c0c043 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.cpp @@ -885,7 +885,7 @@ void CasMountRuntime::enterIdentityLost() LOG_WARNING(getLogger("CasPool"), "Content-addressed pool '{}' entered IdentityLost: the pool sentinels (_pool_meta and the owner " "anchor) are authoritatively absent (both KeyAbsent). This is a fail-loud TERMINAL state: " - "store-class access now fails loud and this pool's remount + GC threads self-exit. " + "store-class access now fails loud and this pool's lease and GC threads self-exit. " "Recover by restart or SYSTEM CAS FORGET — a matching-sentinel restore does NOT " "auto-revive this disk.", server_root_id); @@ -927,7 +927,7 @@ void CasMountRuntime::enterVanished(PoolLifecycle which, const String & reason) { transitioned = true; - /// Publish the terminal-intent latch (spec §3). For a natural transition this is the FIRST + /// Publish the terminal-intent latch. For a natural transition this is the FIRST /// publish; for FORGET, `publishVanishedIntent` already set it at step 1. Either way it is /// published before the state store below and while holding the mutex used by every /// `driver_cv` terminal predicate. @@ -963,7 +963,7 @@ void CasMountRuntime::enterVanished(PoolLifecycle which, const String & reason) void CasMountRuntime::publishVanishedIntent() { - /// FORGET's first step: publish the terminal-intent latch WITHOUT settling the state. `scheduleRemount` + /// FORGET's first step: publish the terminal-intent latch WITHOUT settling the state. Remount requests /// and the lease loop consult `vanished_intent` at their step boundaries, so this stops new remount /// scheduling and makes the lease loop exit at its next step — bounding FORGET's join of the lease /// thread to one step + one backend timeout. The state store + WARN follow in `enterVanished`. diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h index b02dc40318f7..868c4fb80100 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasMountRuntime.h @@ -190,10 +190,11 @@ class CasMountRuntime /// The real boot clock: `CLOCK_BOOTTIME` in milliseconds. Static so tests can compose it. static uint64_t bootMs(); - /// ---- fence-generation admission (rev.7 [C2]/[D1]) ---- - /// Bumped by EVERY `tripMountLost` (a fence loss) and EVERY `armMountFence` (a re-arm -- a fresh - /// lease incarnation, e.g. after a self-remount). A durable-effect caller captures this value once - /// at admission and compares it again immediately before its durable backend call: a DIFFERENT + /// ---- fence-generation admission ---- + /// Bumped by every trip (`tripMountLost`, `tripFenceWithoutOperationalLoss`) and every arm + /// (`armFence`, through `armMountFence`, `armIfAdmissible` and the renewal consume step). A + /// durable-effect caller captures this value once at admission and compares it again immediately + /// before its durable backend call: a DIFFERENT /// value means the lease incarnation moved from under it since admission -- even when the fence /// happens to be live again under a brand-new incarnation, the caller's write is stale and must not /// land. See `checkFenceOrThrow`. @@ -207,18 +208,16 @@ class CasMountRuntime /// admission; the caller's write must never reach the backend in either case. void checkFenceOrThrow(uint64_t admitted_generation) const; - /// ---- pool lifecycle condition (rev.7 §1, spec §§1-3); enum at namespace scope below ---- + /// ---- pool lifecycle condition; enum at namespace scope below ---- /// Atomic read of the current lifecycle (acquire). PoolLifecycle lifecycle() const { return pool_lifecycle.load(std::memory_order_acquire); } /// Whether the pool has reached one of the two fully-terminal `Vanished` values /// (`VanishedReplaced` / `VanishedForgotten`). bool isVanished() const; /// Whether the terminal-intent latch (`vanished_intent`) is published — set by a natural - /// `enterVanished`, OR EARLY (spec §5 step 1) by FORGET's `publishVanishedIntent`, and NEVER by the - /// non-absorbing `IdentityLost` ([C1]). This is the EARLIEST terminal signal: it can already be true - /// while the state is still pre-terminal (mid-FORGET). Consulted alongside `isVanished()` wherever - /// background work must stop the moment the pool is (being driven) terminal: `scheduleRemount`, the - /// lease loop and the GC scheduler. + /// `enterVanished`, OR EARLY by FORGET's `publishVanishedIntent`, and NEVER by `IdentityLost`. This + /// is the EARLIEST terminal signal: it can already be true while the state is still pre-terminal + /// (mid-FORGET). `remountTerminal` folds it in for the lease thread and the GC scheduler. bool vanishedIntentPublished() const { return vanished_intent.load(std::memory_order_acquire); } /// Non-terminal lease-loss transition: `Live -> TransientNotLive`. Idempotent and lock-free; a @@ -256,14 +255,14 @@ class CasMountRuntime /// terminal transition does NOT call this — its `enterVanished` publishes the latch itself. void publishVanishedIntent(); - /// One-way transition to a fully-terminal `Vanished` value (spec §3). Publishes the terminal-intent + /// One-way transition to a fully-terminal `Vanished` value. Publishes the terminal-intent /// latch (so the runtime stops scheduling remount work and the lease loop exits at its next step /// boundary) if it is not already published, records `reason`, stores the state, then emits ONE WARN + /// one `CASDataRootVanished` ProfileEvent. Idempotent: the first terminal STATE transition wins (keyed /// on the `Vanished*` lifecycle value, not on `vanished_intent`, because FORGET publishes that intent /// early at step 1). `which` MUST be one of the two `Vanished*` values (`VanishedReplaced` or /// `VanishedForgotten`). `reason` is retained and - /// surfaced verbatim in the `VanishedForgotten` [D5] error message (see `vanishedReason`). Threads exit + /// surfaced verbatim in the `VanishedForgotten` error message (see `vanishedReason`). Threads exit /// their own loops; the joins happen in `~Pool` for a natural transition, or synchronously in /// `Pool::forgetDisk` for FORGET. Must be called under the caller's remount serialization /// (Pool::remount_mutex). @@ -323,9 +322,9 @@ class CasMountRuntime /// must stop: a published terminal `Vanished` intent (`vanished_intent` — set early by /// FORGET, or by a natural `enterVanished`, and already subsuming every settled `Vanished*` state since /// it is published before the state store) OR `IdentityLost` (a fail-loud TERMINAL state — no - /// demoted observer; recovery is restart or FORGET). Consulted by `scheduleRemount`, by the arming rule - /// and by the lease loop at every step boundary. (The GC scheduler applies the same three-way test through - /// `Pool`.) + /// demoted observer; recovery is restart or FORGET). Consulted by every remount request, by the arming + /// rule and by the lease loop at every step boundary. The GC scheduler calls it through + /// `Pool::remountTerminal`. bool remountTerminal() const { return vanished_intent.load(std::memory_order_acquire) @@ -382,7 +381,7 @@ class CasMountRuntime bool scheduleRemountForTest(); void beginShutdownForTest(); /// Return how many remount requests were attempted, refused ones included: `scheduleRemountForTest`, - /// `tripAndRequestRemount` and terminal renewals. + /// `tripAndRequestRemount` and terminal renewals not ended by a stop. uint64_t scheduleRemountCallCountForTest() const { return schedule_remount_calls_for_test.load(std::memory_order_relaxed); @@ -514,12 +513,11 @@ class CasMountRuntime /// is the gate at the ref-append mutation chokepoint. MountFence mount_fence; - /// Fence-generation token (rev.7 [C2]): bumped by `tripMountLost` and `armMountFence`. See - /// `fenceGeneration`/`checkFenceOrThrow`. + /// Fence-generation token: bumped by every trip and every arm. See `fenceGeneration`/`checkFenceOrThrow`. std::atomic fence_generation{0}; std::function arm_mount_fence_interposition_hook_for_test; /// The first expired deadline of an expiry that a renewal committed past did not end. `UINT64_MAX` - /// when there is none. Written by the renewal consumer and by `armMountFence`. + /// when there is none. Written by `publishRenewedDeadline` and `armFence`. std::atomic lease_expired_at_boot_ms{std::numeric_limits::max()}; mutable std::mutex renew_failure_mutex; String last_renew_failure; diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp index c5e7b6e8bd8c..358d8d807719 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp @@ -1128,9 +1128,9 @@ void Pool::forgetDisk(const std::function & stop_and_join_gc, const Stri /// it, so the reclaim `finishTeardown` is written to override could never happen. Server shutdown /// arms instead: it joins the same scheduler with no remount step to preserve. /// - /// (1) Publish the terminal-intent latch FIRST (spec §5). The runtime stops latching remounts and - /// the remount loop bails at its next step boundary, so every join below is bounded to one step + one - /// backend timeout. + /// (1) Publish the terminal-intent latch FIRST. The runtime stops latching remounts and the lease + /// thread exits at its next step boundary, so every join below is bounded to one step + one backend + /// timeout. mount_runtime.publishVanishedIntent(); /// (2) Trip the local fence — the deliberate decommission act (allowed on a live disk). No durable- @@ -1235,7 +1235,8 @@ bool Pool::tryRemountOnce() event.reason = !succeeded ? "whole-chain remount returned without restoring Live" : (armed ? "whole-chain remount restored Live under a fresh mount incarnation" - : "whole-chain remount claimed a fresh mount incarnation; the next renewal arms the fence"); + : "whole-chain remount claimed a fresh mount incarnation but its arming conditions did not hold; " + "the fence stays latched until a later renewal or reclaim arms it"); event.detail = { {"attempt_no", std::to_string(attempt_no)}, {"step", String{step}}, @@ -1474,8 +1475,8 @@ bool Pool::tryRemountOnce() /// where the gate is open while the epoch is still stale. step = "publish_writer_epoch"; mount_runtime.setLiveWriterEpoch(writer_epoch); - /// 2. CANCEL OR JOIN every in-flight ref-table recovery, and BLOCK here until none is left (spec - /// §3: "self-remount cancels or waits out recovery before rearming"). A recovery admitted under + /// 2. CANCEL OR JOIN every in-flight ref-table recovery, and BLOCK here until none is left. A + /// recovery admitted under /// the outgoing incarnation WRITES -- its seal CAS-walk mints epoch seals and advances the /// `_ckpt` -- so it must be stopped at this boundary rather than caught one site at a time /// after the incarnation has already changed underneath it. Strictly before the quiesce below @@ -1491,8 +1492,8 @@ bool Pool::tryRemountOnce() config.remount_quiesce_hook_for_test(); /// No-throw commit section. The fence and the lifecycle are published only after epoch, renewer, - /// recovery cancellation and ref-runtime quiescence are complete. A claim whose deadline no longer - /// admits a ref append leaves the fence latched; the lease thread's next renewal arms it. + /// recovery cancellation and ref-runtime quiescence are complete. A claim that fails the arming rule + /// leaves the fence latched for the lease thread's next renewal or a newer reclaim. step = "arm_fence"; armed = mount_runtime.armIfAdmissible( remount_anchor_boot_ms > std::numeric_limits::max() - ttl_ms diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h index dd5f2630dad2..cbdc72f5a398 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h @@ -470,8 +470,8 @@ class Pool : public std::enable_shared_from_this /// `lost` and the monotonic deadline has not passed. Permissive until armed: a Pool that has not /// armed the fence (the default deadline is steady_clock::time_point::max()) always allows mutations. bool mayMutate() const; - /// Latch the fence to lost (once lost, stays lost). Called by the renewer on a superseded - /// or foreign observation; the gated mutate chokepoints then fail closed. + /// Test seam: `CasMountRuntime::tripMountLost`, with no remount request. Production trips through + /// the runtime. void tripMountLost(); /// Refresh the write-fence deadline (a CLOCK_BOOTTIME-milliseconds instant; release). /// renewer renew calls this on success. @@ -494,11 +494,12 @@ class Pool : public std::enable_shared_from_this /// The real boot clock: CLOCK_BOOTTIME in milliseconds. Static so tests can compose it. static uint64_t bootMs(); - /// ---- fence-generation admission (rev.7 [C2]/[D1]; owned by `mount_runtime`) ---- - /// Bumped on every `tripMountLost`/`armMountFence`. Forwarders used directly by the S3-native - /// staging-buffer finalize (`ContentAddressedTransaction::writeFile`) -- the durable-effect site - /// outside `CasPlainObjects` that needs to capture-then-recheck a fence-generation token across an - /// async, potentially long-running upload. `CasPlainObjects` reaches the same primitives via + /// ---- fence-generation admission (owned by `mount_runtime`) ---- + /// Bumped by every trip and every arm of the fence (see `CasMountRuntime::fenceGeneration`). + /// Forwarders used directly by the S3-native staging-buffer finalize + /// (`ContentAddressedTransaction::writeFile`) -- the durable-effect site outside `CasPlainObjects` + /// that needs to capture-then-recheck a fence-generation token across an async, potentially + /// long-running upload. `CasPlainObjects` reaches the same primitives via /// injected callbacks (see its own constructor). uint64_t fenceGeneration() const { return mount_runtime.fenceGeneration(); } /// Throws the typed transient refusal (`throwCasTransientUnavailable`) unless the fence is currently @@ -809,26 +810,21 @@ class Pool : public std::enable_shared_from_this /// `Pool::open`. Orchestration stays here; the owned mount primitives it drives (renewer swap, /// epoch bump, fence re-arm) live on `mount_runtime`. Returns false (and changes nothing durable /// beyond the epoch bump) when the - /// mount cannot be claimed (foreign owner / a genuinely live twin) — the caller retries. Safe to - /// call concurrently (serialized internally); also the synchronous test seam. + /// mount cannot be claimed (foreign owner / a genuinely live twin) — the caller retries. Serialized + /// by `remount_mutex`. While a lease thread runs, a call from any other thread fails at the renewer's + /// owner check (a `LOGICAL_ERROR`). Also the synchronous test seam. bool tryRemountOnce(); - /// Test seam: latch the private self-remount path directly. In production the runtime terminal - /// consumer calls `scheduleRemount`, while external loss paths raise the same generation for the lease - /// thread. - /// Returns true iff an unhandled recovery generation exists after the call. + /// Test seam: request a remount without a trip. Production requests one through + /// `reportImpossibleInterference` and a terminal renewal. Returns true iff a lease thread runs and a + /// request is pending after the call. bool scheduleRemountForTest(); - /// Test seam: how many times `scheduleRemount` has been ENTERED, counted - /// unconditionally as its very first statement. This increments even under the default - /// `background_watermark = false` (no worker exists; a - /// test never pays for a real self-remount attempt racing this Pool's own still-live renewer, which - /// -- confirmed while building this seam -- reliably takes 30+ seconds per call and is not something - /// a fast unit test should be driving). Positively pins that a production call site (e.g. - /// `reportImpossibleInterference`) actually invoked `scheduleRemount`, as opposed to merely observing - /// `mayMutate() == false` (which `tripMountLost` alone already accounts for). + /// Test seam: see `CasMountRuntime::scheduleRemountCallCountForTest`. It counts with no lease thread + /// too, so a test can pin that a call site such as `reportImpossibleInterference` requested a + /// remount, which `mayMutate() == false` alone does not show. uint64_t scheduleRemountCallCountForTest() const { return mount_runtime.scheduleRemountCallCountForTest(); } /// Test seam: publish the same worker-stop request as `~Pool` without tearing the pool down, so a - /// test can assert `scheduleRemount` refuses to latch work once teardown has begun. + /// test can assert a remount request is refused once teardown has begun. void beginShutdownForTest(); @@ -921,8 +917,7 @@ class Pool : public std::enable_shared_from_this /// makes `key` exclusively ours: foreign bytes observed at our own wedge key, or the wedge hard /// contract itself violated at new-id-allocation time. LOG_ERROR with full context, emit a /// `ForeignInterference` CasEvent, then fence this mount closed and arm the SAME bounded - /// self-remount a foreign/superseded lease renewal already drives (`tripMountLost` followed by - /// `scheduleRemount`). Diagnosis is strictly off the + /// self-remount a foreign/superseded lease renewal already drives (`tripAndRequestRemount`). Diagnosis is strictly off the /// critical path: ONE background GET of `key` (best-effort, single attempt), decoded as far as its /// ref-log header parses, logged -- never blocking or throwing on the caller's thread. Does NOT /// itself throw: every call site raises its OWN `LOGICAL_ERROR` immediately after this returns, so diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefLedger.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefLedger.h index 2414a621216e..51d5699668f3 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefLedger.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasRefLedger.h @@ -770,10 +770,10 @@ class CasRefLedger /// Both this flag and the condition variable are guarded by `state_mutex`. bool recovery_in_progress = false; std::condition_variable recovery_cv; - /// The self-remount cancellation request (spec §3: "self-remount cancels or waits out recovery - /// before rearming"). Set by `cancelRecoveriesAndAwaitQuiescence` from the remount thread and - /// polled by the recovery walk at EVERY I/O boundary; a recovery that observes it abandons its - /// attempt having written nothing and installed nothing. + /// The self-remount cancellation request: a reclaim cancels or waits out recovery before it arms. + /// Set by `cancelRecoveriesAndAwaitQuiescence`, which `Pool::tryRemountOnce` calls, and polled by + /// the recovery walk at EVERY I/O boundary; a recovery that observes it abandons its attempt + /// having written nothing and installed nothing. /// /// ATOMIC, not `state_mutex`-guarded like its two neighbours, and that is the point: the /// canceller must be able to publish the request WITHOUT queueing behind the very recovery it is diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h index eeee0e08a242..64049332fac1 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasServerRoot.h @@ -574,8 +574,8 @@ bool isCreatorFenceTerminal(CasOperation & op, const Layout & layout, const Stri /// liveness. The claim and the farewell are admitted on `claim_farewell_requests`, off the fence: a /// self-remount claims with the fence already latched lost, so a claim gated on the fence could never /// reclaim, and a farewell refused because the fence has run down would leave the slot looking live -/// until GC fences it out. Neither is unguarded: a claim's safety is its own conditional write, and a caller that has -/// shutdown facts hands them over as a `Liveness`. +/// until GC fences it out. Neither is unguarded: a claim's safety is its own conditional write, and a +/// caller that has shutdown facts hands them over as a `Liveness`. class MountLeaseRenewer { public: diff --git a/src/Disks/tests/gtest_cas_lifecycle_condition.cpp b/src/Disks/tests/gtest_cas_lifecycle_condition.cpp index ba9970a5ef8f..51b40d4cac0b 100644 --- a/src/Disks/tests/gtest_cas_lifecycle_condition.cpp +++ b/src/Disks/tests/gtest_cas_lifecycle_condition.cpp @@ -118,7 +118,7 @@ class ToggleableTransportFaultBackend final : public InMemoryBackend /// `Vanished`) and store()-class access fails loud. rev.8: `IdentityLost` is a fail-loud TERMINAL state — /// `isVanished()` still reads false (it is a distinct terminal), but a direct gate re-probe refuses without /// ever claiming/allocating/writing (the thread-exit behavior of the background observer is covered by -/// `RemountThreadSelfExitsOnceIdentityLost` below). +/// `AnIdentityLostPoolRefusesARemountRequest` below). TEST(CASLifecycleCondition, SentinelsDeletedEntersIdentityLostTerminal) { auto backend = std::make_shared(); @@ -151,15 +151,12 @@ TEST(CASLifecycleCondition, SentinelsDeletedEntersIdentityLostTerminal) EXPECT_GE(backend->headCount(meta_key), 1u) << "the gate still probes _pool_meta authoritatively"; } -/// (a2) rev.8 worker-exit: `IdentityLost` is terminal, so the persistent self-remount worker must self-exit -/// — mirroring how a `Vanished` pool refuses to latch work. With `background_watermark = true`, `scheduleRemount` -/// must REFUSE to latch a recovery generation once the pool is `IdentityLost` (`remountTerminal` covers it), -/// exactly as it refuses on a published `Vanished` intent. -TEST(CASLifecycleCondition, RemountThreadSelfExitsOnceIdentityLost) +/// (a2) `IdentityLost` is terminal: a remount request is refused (`remountTerminal` covers it), as it is on +/// a published `Vanished` intent. +TEST(CASLifecycleCondition, AnIdentityLostPoolRefusesARemountRequest) { auto backend = std::make_shared(); - /// `background_watermark = true` so the persistent recovery worker exists in production mode - /// (mirrors gtest_cas_pool.cpp's ShutdownGuardRefusesToArmRemount setup). + /// `background_watermark = true` so the lease thread exists. auto store = DB::Cas::Pool::open(backend, DB::Cas::PoolConfig{.pool_prefix = "p", .server_root_id = "test", .background_watermark = true}); @@ -169,10 +166,8 @@ TEST(CASLifecycleCondition, RemountThreadSelfExitsOnceIdentityLost) EXPECT_FALSE(store->tryRemountOnce()); ASSERT_EQ(store->lifecycle(), PoolLifecycle::IdentityLost); - /// The runtime terminal consumer (or any direct `scheduleRemount`) must now refuse: no worker runs on a - /// terminal pool. EXPECT_FALSE(store->scheduleRemountForTest()) - << "an IdentityLost pool is terminal (rev.8) — scheduleRemount must not latch recovery work"; + << "an IdentityLost pool is terminal: a remount request must not latch recovery work"; } /// (b) `_pool_meta` present but its `pool_id` is foreign → `Vanished(replaced)` immediately. diff --git a/src/Disks/tests/gtest_cas_pool.cpp b/src/Disks/tests/gtest_cas_pool.cpp index 8ba0713b5693..70b19185b556 100644 --- a/src/Disks/tests/gtest_cas_pool.cpp +++ b/src/Disks/tests/gtest_cas_pool.cpp @@ -1737,18 +1737,18 @@ TEST(CASPoolRemount, ForeignOwnerIsNeverTakenOver) TEST(CASPoolRemount, ShutdownGuardRefusesToArmRemount) { auto backend = std::make_shared(); - /// `background_watermark = true` so `scheduleRemount` can latch a recovery generation for the - /// persistent worker in production mode (the same gate both runtime workers check). + /// `background_watermark = true` so a remount request can latch a recovery generation for the + /// lease thread. auto store = DB::Cas::Pool::open(backend, DB::Cas::PoolConfig{.pool_prefix = "p", .server_root_id = "test", .background_watermark = true}); /// Teardown has begun: `Pool` latches this before joining either persistent worker. store->beginShutdownForTest(); - /// A lease-renewal failure firing during teardown re-enters `scheduleRemount`. With the guard it - /// must refuse to latch another generation after the workers are stopping. + /// A remount request during teardown must not latch another generation after the workers are + /// stopping. EXPECT_FALSE(store->scheduleRemountForTest()) - << "scheduleRemount must not latch recovery work once teardown has begun"; + << "a remount request must not latch recovery work once teardown has begun"; } namespace @@ -2711,8 +2711,8 @@ TEST(CASMountOpenWaits, FencedPriorReclaimsWithoutAnyWait) /// A reclaim arms the fence only when its claim still admits a ref append: more than two envelopes plus /// the margin of lease, strictly, with no period term. With attempt 100 and cap 100 that is 650 ms of the /// 1000 ms lease. A quiescence that leaves less reports success at step `claimed_not_armed` with the pool -/// not `Live`; the lease thread's next renewal arms it. -TEST(CASPoolRemount, RemountRenewerRedoUsesTheEnvelope) +/// not `Live` and the fence latched. +TEST(CASPoolRemount, AReclaimArmsOnlyWithRoomForTwoEnvelopesAndTheMargin) { struct AtRemountEvent { diff --git a/src/Disks/tests/gtest_cas_ref_writer.cpp b/src/Disks/tests/gtest_cas_ref_writer.cpp index b5d38fe584d6..69058842e66b 100644 --- a/src/Disks/tests/gtest_cas_ref_writer.cpp +++ b/src/Disks/tests/gtest_cas_ref_writer.cpp @@ -2051,15 +2051,15 @@ TEST(CASAnomalyPolicy, ForeignBytesAtWedgeKeyTripFenceAndRemount) EXPECT_EQ(store->laneStateForTest(ns), RefLaneState::Faulted) << "foreign interference must fault the lane"; EXPECT_FALSE(store->mayMutate()) << "the local write fence must trip closed on the anomaly"; - /// Positively pins that `reportImpossibleInterference` called `scheduleRemount` (not just - /// `tripMountLost`, which alone already accounts for `mayMutate() == false` above). Counted at - /// `scheduleRemount`'s own entry regardless of `background_watermark` -- see that accessor's + /// Positively pins that `reportImpossibleInterference` requested a remount (not just a trip, + /// which alone already accounts for `mayMutate() == false` above). Counted at + /// `tripAndRequestRemount`'s entry regardless of `background_watermark` -- see that accessor's /// comment for why this test deliberately does NOT enable `background_watermark` to observe a real /// automatic recovery: doing so was tried and makes the store's self-remount attempt race its own /// still-live renewer for 30+ seconds per call (confirmed while building this test), which is not /// something a fast unit test should be driving. EXPECT_EQ(store->scheduleRemountCallCountForTest(), 1u) - << "reportImpossibleInterference must have called scheduleRemount exactly once"; + << "reportImpossibleInterference must have requested a remount exactly once"; const std::vector observed = seen->snapshot(); const auto has_event = std::any_of(observed.begin(), observed.end(), @@ -2132,7 +2132,7 @@ TEST(CASAnomalyPolicy, NonReadyAtNewIdAllocationFaultsAndFailsClosed) /// `background_watermark` plus automatic recovery -- that combination makes the store's self-remount /// race its own still-live renewer). EXPECT_EQ(store->scheduleRemountCallCountForTest(), 1u) - << "reportImpossibleInterference must have called scheduleRemount exactly once"; + << "reportImpossibleInterference must have requested a remount exactly once"; const std::vector observed = seen->snapshot(); const auto has_event = std::any_of(observed.begin(), observed.end(), From 56897c5ccbcd96637e2afbf853b4e160270dd851 Mon Sep 17 00:00:00 2001 From: Mikhail Filimonov Date: Fri, 2 Oct 2026 11:30:49 +0200 Subject: [PATCH 37/37] Fix four lease-keeper comments that named the wrong callers or order Co-Authored-By: Claude Opus 5.5 --- .../MetadataStorages/ContentAddressed/Pool/CasPool.cpp | 2 +- .../MetadataStorages/ContentAddressed/Pool/CasPool.h | 10 ++++++---- src/Disks/tests/gtest_cas_lifecycle_condition.cpp | 4 ++-- src/Disks/tests/gtest_cas_ref_writer.cpp | 5 ++--- 4 files changed, 11 insertions(+), 10 deletions(-) diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp index 358d8d807719..4ae25ded1407 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.cpp @@ -1471,7 +1471,7 @@ bool Pool::tryRemountOnce() /// Starting the renewer does NOT clear `lost`, so the fence stays closed here and no append/publish can race the /// swap. /// 1. Bump the live epoch so every subsequent `allocateRefTxnId` sorts strictly above any older - /// (dead-incarnation or twin) durable log. Do this BEFORE `armMountFence` so there is no window + /// (dead-incarnation or twin) durable log. Do this BEFORE `armIfAdmissible` so there is no window /// where the gate is open while the epoch is still stale. step = "publish_writer_epoch"; mount_runtime.setLiveWriterEpoch(writer_epoch); diff --git a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h index cbdc72f5a398..a469bdbd6de4 100644 --- a/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h +++ b/src/Disks/DiskObjectStorage/MetadataStorages/ContentAddressed/Pool/CasPool.h @@ -473,10 +473,11 @@ class Pool : public std::enable_shared_from_this /// Test seam: `CasMountRuntime::tripMountLost`, with no remount request. Production trips through /// the runtime. void tripMountLost(); - /// Refresh the write-fence deadline (a CLOCK_BOOTTIME-milliseconds instant; release). - /// renewer renew calls this on success. + /// Test seam: `CasMountRuntime::setMountDeadline`. Production publishes a renewed deadline through + /// `CasMountRuntime::publishRenewedDeadline`. void setMountDeadline(uint64_t deadline_boot_ms); - /// Arm the fence at startup: set the deadline, clear `lost`. + /// Test seam: `CasMountRuntime::armMountFence`. Production arms through `armIfAdmissible` and the + /// renewal's consume step. void armMountFence(uint64_t deadline_boot_ms); void setArmMountFenceInterpositionHookForTest(std::function hook) { @@ -812,7 +813,8 @@ class Pool : public std::enable_shared_from_this /// beyond the epoch bump) when the /// mount cannot be claimed (foreign owner / a genuinely live twin) — the caller retries. Serialized /// by `remount_mutex`. While a lease thread runs, a call from any other thread fails at the renewer's - /// owner check (a `LOGICAL_ERROR`). Also the synchronous test seam. + /// owner check (a `LOGICAL_ERROR`), after it has claimed the mount under a fresh epoch, so such a call + /// is never harmless. Also the synchronous test seam. bool tryRemountOnce(); /// Test seam: request a remount without a trip. Production requests one through diff --git a/src/Disks/tests/gtest_cas_lifecycle_condition.cpp b/src/Disks/tests/gtest_cas_lifecycle_condition.cpp index 51b40d4cac0b..9d5fead2376a 100644 --- a/src/Disks/tests/gtest_cas_lifecycle_condition.cpp +++ b/src/Disks/tests/gtest_cas_lifecycle_condition.cpp @@ -115,9 +115,9 @@ class ToggleableTransportFaultBackend final : public InMemoryBackend } /// (a) `_pool_meta` + the owner anchor authoritatively absent → the gate enters `IdentityLost` (never -/// `Vanished`) and store()-class access fails loud. rev.8: `IdentityLost` is a fail-loud TERMINAL state — +/// `Vanished`) and store()-class access fails loud. `IdentityLost` is a fail-loud TERMINAL state — /// `isVanished()` still reads false (it is a distinct terminal), but a direct gate re-probe refuses without -/// ever claiming/allocating/writing (the thread-exit behavior of the background observer is covered by +/// ever claiming/allocating/writing (a remount request on an `IdentityLost` pool is covered by /// `AnIdentityLostPoolRefusesARemountRequest` below). TEST(CASLifecycleCondition, SentinelsDeletedEntersIdentityLostTerminal) { diff --git a/src/Disks/tests/gtest_cas_ref_writer.cpp b/src/Disks/tests/gtest_cas_ref_writer.cpp index 69058842e66b..be99715c05cf 100644 --- a/src/Disks/tests/gtest_cas_ref_writer.cpp +++ b/src/Disks/tests/gtest_cas_ref_writer.cpp @@ -2053,9 +2053,8 @@ TEST(CASAnomalyPolicy, ForeignBytesAtWedgeKeyTripFenceAndRemount) EXPECT_FALSE(store->mayMutate()) << "the local write fence must trip closed on the anomaly"; /// Positively pins that `reportImpossibleInterference` requested a remount (not just a trip, /// which alone already accounts for `mayMutate() == false` above). Counted at - /// `tripAndRequestRemount`'s entry regardless of `background_watermark` -- see that accessor's - /// comment for why this test deliberately does NOT enable `background_watermark` to observe a real - /// automatic recovery: doing so was tried and makes the store's self-remount attempt race its own + /// `tripAndRequestRemount`'s entry regardless of `background_watermark`. This test does not enable + /// `background_watermark` to observe a real automatic recovery: doing so was tried and makes the store's self-remount attempt race its own /// still-live renewer for 30+ seconds per call (confirmed while building this test), which is not /// something a fast unit test should be driving. EXPECT_EQ(store->scheduleRemountCallCountForTest(), 1u)