From c39b9947c62b7c3e542f12aba4cd68c4a553edbe Mon Sep 17 00:00:00 2001 From: Aikins Laryea Date: Sat, 3 Oct 2026 15:39:41 +0000 Subject: [PATCH 01/17] fix(wal): tag records with their owning catalog and route replay per owner Graph table commits log to the single main WAL, but replay resolved table OIDs through the ambient catalog, so committed graph data resolved against the wrong catalog's OID space at recovery. Records now carry their owning catalog's name on the wire (checkpoint bundle format version 2), catalog entries back-point to their owning catalog, replay resolves each record through its owner, and local storage keys staged tables by owner so cross-graph transactions commit each table to its own graph. Main's catalog carries no storage manager, so owner-aware storage resolution falls back to the database storage manager for main. A tagged record whose graph was created in the same WAL (and is therefore not loaded yet during main-WAL replay) materializes it on demand, and materialized ANY graphs whose catalogs were never checkpointed get their infra tables recreated so OID spaces line up. Checkpoint recovery recognizes every persisted bundle-format version 1..CHECKPOINT_BUNDLE_FORMAT_VERSION as a bundle: a version-1 marker written by an earlier build must still apply graph shadows instead of taking the pre-bundle legacy path that discards them. --- src/catalog/catalog.cpp | 50 +-- src/catalog/catalog_entry/catalog_entry.cpp | 5 + src/catalog/catalog_set.cpp | 17 + src/extension/extension_manager.cpp | 2 +- src/graph/on_disk_graph.cpp | 4 +- src/include/catalog/catalog.h | 2 +- .../catalog/catalog_entry/catalog_entry.h | 7 + .../catalog_entry/index_catalog_entry.h | 7 + src/include/catalog/catalog_set.h | 9 + src/include/main/database_manager.h | 32 ++ .../storage/local_storage/local_storage.h | 21 +- .../storage/local_storage/local_table.h | 2 + .../storage/partition_storage_registry.h | 10 + src/include/storage/table/table.h | 2 + src/include/storage/wal/local_wal.h | 46 ++- .../storage/wal/record/wal_record_base.h | 2 + src/include/storage/wal/wal.h | 2 +- src/include/storage/wal/wal_replayer.h | 26 ++ src/include/transaction/transaction.h | 19 +- src/main/database_manager.cpp | 349 ++++++++++++------ src/processor/operator/ddl/create_index.cpp | 4 +- .../operator/scan/count_rel_table.cpp | 3 +- .../operator/scan/scan_node_table.cpp | 3 +- src/storage/local_storage/local_storage.cpp | 110 ++++-- src/storage/partition_storage_registry.cpp | 28 +- src/storage/storage_manager.cpp | 4 + src/storage/table/node_table.cpp | 57 +-- src/storage/table/rel_table.cpp | 39 +- src/storage/table/table.cpp | 5 +- src/storage/wal/local_wal.cpp | 54 ++- .../create_catalog_entry_record_replay.cpp | 47 ++- .../records/create_index_record_replay.cpp | 19 +- .../drop_catalog_entry_record_replay.cpp | 2 +- .../records/node_deletion_record_replay.cpp | 7 +- .../wal/records/node_update_record_replay.cpp | 7 +- .../records/rel_deletion_record_replay.cpp | 4 +- .../rel_detach_delete_record_replay.cpp | 4 +- .../wal/records/rel_update_record_replay.cpp | 4 +- .../records/table_insertion_record_replay.cpp | 7 +- src/storage/wal/wal_record.cpp | 6 + src/storage/wal/wal_replayer.cpp | 96 ++++- src/transaction/transaction.cpp | 87 +++-- 42 files changed, 897 insertions(+), 314 deletions(-) diff --git a/src/catalog/catalog.cpp b/src/catalog/catalog.cpp index 697b7175fb..4f9b7631e2 100644 --- a/src/catalog/catalog.cpp +++ b/src/catalog/catalog.cpp @@ -77,6 +77,10 @@ Catalog* Catalog::Get(const main::ClientContext& context) { return context.getAttachedDatabase()->getCatalog(); } auto dbManager = main::DatabaseManager::Get(context); + if (auto* replayOwnerCatalog = dbManager->getReplayOwnerCatalog(); + replayOwnerCatalog != nullptr) { + return replayOwnerCatalog; + } if (dbManager->hasDefaultGraph()) { auto graphCatalog = dbManager->getDefaultGraphCatalog(); if (graphCatalog != nullptr) { @@ -87,16 +91,16 @@ Catalog* Catalog::Get(const main::ClientContext& context) { } void Catalog::initCatalogSets() { - tables = std::make_unique(); - sequences = std::make_unique(); - functions = std::make_unique(); - types = std::make_unique(); - indexes = std::make_unique(); - macros = std::make_unique(); - internalTables = std::make_unique(true /* isInternal */); - internalSequences = std::make_unique(true /* isInternal */); - internalFunctions = std::make_unique(true /* isInternal */); - graphs = std::make_unique(); + tables = std::make_unique(this); + sequences = std::make_unique(this); + functions = std::make_unique(this); + types = std::make_unique(this); + indexes = std::make_unique(this); + macros = std::make_unique(this); + internalTables = std::make_unique(this, true /* isInternal */); + internalSequences = std::make_unique(this, true /* isInternal */); + internalFunctions = std::make_unique(this, true /* isInternal */); + graphs = std::make_unique(this); } bool Catalog::containsTable(const Transaction* transaction, const std::string& tableName, @@ -413,10 +417,10 @@ bool Catalog::containsType(const Transaction* transaction, const std::string& ty return types->containsEntry(transaction, typeName); } -void Catalog::createIndex(Transaction* transaction, std::unique_ptr indexCatalogEntry, - bool skipLoggingToWAL) { +oid_t Catalog::createIndex(Transaction* transaction, + std::unique_ptr indexCatalogEntry, bool skipLoggingToWAL) { DASSERT(indexCatalogEntry->getType() == CatalogEntryType::INDEX_ENTRY); - indexes->createEntry(transaction, std::move(indexCatalogEntry), skipLoggingToWAL); + return indexes->createEntry(transaction, std::move(indexCatalogEntry), skipLoggingToWAL); } IndexCatalogEntry* Catalog::getIndex(const Transaction* transaction, table_id_t tableID, @@ -820,17 +824,17 @@ void Catalog::serializeSnapshot(Serializer& ser, common::transaction_t snapshotT } void Catalog::deserialize(Deserializer& deSer) { - tables = CatalogSet::deserialize(deSer); - sequences = CatalogSet::deserialize(deSer); - functions = CatalogSet::deserialize(deSer); + tables = CatalogSet::deserialize(this, deSer); + sequences = CatalogSet::deserialize(this, deSer); + functions = CatalogSet::deserialize(this, deSer); registerBuiltInFunctions(); - types = CatalogSet::deserialize(deSer); - indexes = CatalogSet::deserialize(deSer); - macros = CatalogSet::deserialize(deSer); - internalTables = CatalogSet::deserialize(deSer); - internalSequences = CatalogSet::deserialize(deSer); - internalFunctions = CatalogSet::deserialize(deSer); - graphs = CatalogSet::deserialize(deSer); + types = CatalogSet::deserialize(this, deSer); + indexes = CatalogSet::deserialize(this, deSer); + macros = CatalogSet::deserialize(this, deSer); + internalTables = CatalogSet::deserialize(this, deSer); + internalSequences = CatalogSet::deserialize(this, deSer); + internalFunctions = CatalogSet::deserialize(this, deSer); + graphs = CatalogSet::deserialize(this, deSer); } } // namespace catalog diff --git a/src/catalog/catalog_entry/catalog_entry.cpp b/src/catalog/catalog_entry/catalog_entry.cpp index d4eb81d44c..073a509f27 100644 --- a/src/catalog/catalog_entry/catalog_entry.cpp +++ b/src/catalog/catalog_entry/catalog_entry.cpp @@ -1,5 +1,6 @@ #include "catalog/catalog_entry/catalog_entry.h" +#include "catalog/catalog.h" #include "catalog/catalog_entry/graph_catalog_entry.h" #include "catalog/catalog_entry/index_catalog_entry.h" #include "catalog/catalog_entry/scalar_macro_catalog_entry.h" @@ -78,5 +79,9 @@ void CatalogEntry::copyFrom(const CatalogEntry& other) { hasParent_ = other.hasParent_; } +std::string CatalogEntry::getOwningCatalogName() const { + return owningCatalog ? owningCatalog->getCatalogName() : ""; +} + } // namespace catalog } // namespace lbug diff --git a/src/catalog/catalog_set.cpp b/src/catalog/catalog_set.cpp index c092097b21..a6aa569a6b 100644 --- a/src/catalog/catalog_set.cpp +++ b/src/catalog/catalog_set.cpp @@ -3,6 +3,7 @@ #include #include "binder/ddl/bound_alter_info.h" +#include "catalog/catalog.h" #include "catalog/catalog_entry/dummy_catalog_entry.h" #include "catalog/catalog_entry/table_catalog_entry.h" #include "common/assert.h" @@ -23,6 +24,12 @@ CatalogSet::CatalogSet(bool isInternal) { } } +CatalogSet::CatalogSet(Catalog* catalog, bool isInternal) : catalog{catalog} { + if (isInternal) { + nextOID = INTERNAL_CATALOG_SET_START_OID; + } +} + static bool checkWWConflict(const Transaction* transaction, const CatalogEntry* entry) { return (entry->getTimestamp() >= Transaction::START_TRANSACTION_ID && entry->getTimestamp() != transaction->getID()) || @@ -104,6 +111,7 @@ CatalogEntry* CatalogSet::createEntryNoLock(const Transaction* transaction, } void CatalogSet::emplaceNoLock(std::unique_ptr entry) { + entry->setOwningCatalog(catalog); if (entries.contains(entry->getName())) { entry->setPrev(std::move(entries.at(entry->getName()))); entries.erase(entry->getName()); @@ -300,8 +308,13 @@ void CatalogSet::serializeSnapshot(Serializer serializer, const Transaction* sna } std::unique_ptr CatalogSet::deserialize(Deserializer& deserializer) { + return deserialize(nullptr, deserializer); +} + +std::unique_ptr CatalogSet::deserialize(Catalog* catalog, Deserializer& deserializer) { std::string debuggingInfo; auto catalogSet = std::make_unique(); + catalogSet->catalog = catalog; deserializer.validateDebuggingInfo(debuggingInfo, "nextOID"); deserializer.deserializeValue(catalogSet->nextOID); uint64_t numEntries = 0; @@ -316,6 +329,10 @@ std::unique_ptr CatalogSet::deserialize(Deserializer& deserializer) return catalogSet; } +std::string CatalogSet::getOwnerCatalogName() const { + return catalog ? catalog->getCatalogName() : ""; +} + // Ideally we should not trigger the following check. Instead, we should throw more informative // error message at catalog level. void CatalogSet::validateExistNoLock(const Transaction* transaction, diff --git a/src/extension/extension_manager.cpp b/src/extension/extension_manager.cpp index b12c043aba..9e0e099449 100644 --- a/src/extension/extension_manager.cpp +++ b/src/extension/extension_manager.cpp @@ -46,7 +46,7 @@ void ExtensionManager::loadExtension(const std::string& path, main::ClientContex isOfficial ? ExtensionSource::OFFICIAL : ExtensionSource::USER)); auto transaction = transaction::Transaction::Get(*context); if (transaction->shouldLogToWAL()) { - transaction->getLocalWAL().logLoadExtension(path); + transaction->getLocalWAL().logLoadExtension("" /* main catalog */, path); } } diff --git a/src/graph/on_disk_graph.cpp b/src/graph/on_disk_graph.cpp index 5ebf9acc21..34ae88fe6b 100644 --- a/src/graph/on_disk_graph.cpp +++ b/src/graph/on_disk_graph.cpp @@ -361,8 +361,8 @@ bool OnDiskGraphVertexScanState::next() { auto endOffset = std::min(endOffsetExclusive, tableScanState->source == TableScanSource::COMMITTED ? startOffsetOfNextGroup : - startOffsetOfNextGroup + transaction->getUncommittedOffset( - tableScanState->table->getTableID(), currentOffset)); + startOffsetOfNextGroup + + transaction->getUncommittedOffset(*tableScanState->table, currentOffset)); numNodesToScan = std::min(endOffset - currentOffset, DEFAULT_VECTOR_CAPACITY); auto result = tableScanState->scanNext(transaction, currentOffset, numNodesToScan); currentOffset += result.numRows; diff --git a/src/include/catalog/catalog.h b/src/include/catalog/catalog.h index 8b27e92125..563bcc3121 100644 --- a/src/include/catalog/catalog.h +++ b/src/include/catalog/catalog.h @@ -156,7 +156,7 @@ class LBUG_API Catalog { common::table_id_t tableID) const; // Create index entry. - void createIndex(transaction::Transaction* transaction, + common::oid_t createIndex(transaction::Transaction* transaction, std::unique_ptr indexCatalogEntry, bool skipLoggingToWAL = false); // Drop all index entries within a table. void dropAllIndexes(transaction::Transaction* transaction, common::table_id_t tableID); diff --git a/src/include/catalog/catalog_entry/catalog_entry.h b/src/include/catalog/catalog_entry/catalog_entry.h index dd4b14e38c..bbee21114f 100644 --- a/src/include/catalog/catalog_entry/catalog_entry.h +++ b/src/include/catalog/catalog_entry/catalog_entry.h @@ -15,6 +15,8 @@ class ClientContext; namespace catalog { +class Catalog; + struct LBUG_API ToCypherInfo { virtual ~ToCypherInfo() = default; @@ -40,6 +42,9 @@ class LBUG_API CatalogEntry { // getter & setter //===--------------------------------------------------------------------===// CatalogEntryType getType() const { return type; } + void setOwningCatalog(Catalog* catalog) { owningCatalog = catalog; } + Catalog* getOwningCatalog() const { return owningCatalog; } + std::string getOwningCatalogName() const; void rename(std::string name_) { this->name = std::move(name_); } std::string getName() const { return name; } common::transaction_t getTimestamp() const { return timestamp; } @@ -99,6 +104,8 @@ class LBUG_API CatalogEntry { protected: CatalogEntryType type; + // Never serialized; re-established from the owning CatalogSet on load. + Catalog* owningCatalog = nullptr; std::string name; common::oid_t oid; common::transaction_t timestamp; diff --git a/src/include/catalog/catalog_entry/index_catalog_entry.h b/src/include/catalog/catalog_entry/index_catalog_entry.h index 2f6f6c25d7..5d7e1958d2 100644 --- a/src/include/catalog/catalog_entry/index_catalog_entry.h +++ b/src/include/catalog/catalog_entry/index_catalog_entry.h @@ -73,6 +73,13 @@ class LBUG_API IndexCatalogEntry final : public CatalogEntry { common::table_id_t getTableID() const { return tableID; } + // The catalog-set key embeds the table ID (see getInternalIndexName), so the + // name must change with it. + void setTableID(common::table_id_t tableID_) { + tableID = tableID_; + rename(getInternalIndexName(tableID_, indexName)); + } + std::string getIndexName() const { return indexName; } std::vector getPropertyIDs() const { return propertyIDs; } diff --git a/src/include/catalog/catalog_set.h b/src/include/catalog/catalog_set.h index fdf05c5e3a..ef5997244f 100644 --- a/src/include/catalog/catalog_set.h +++ b/src/include/catalog/catalog_set.h @@ -22,12 +22,15 @@ class Transaction; using CatalogEntrySet = common::case_insensitive_map_t; namespace catalog { +class Catalog; + class LBUG_API CatalogSet { friend class storage::UndoBuffer; public: CatalogSet() = default; explicit CatalogSet(bool isInternal); + explicit CatalogSet(Catalog* catalog, bool isInternal = false); bool containsEntry(const transaction::Transaction* transaction, const std::string& name); CatalogEntry* getEntry(const transaction::Transaction* transaction, const std::string& name); common::oid_t createEntry(transaction::Transaction* transaction, @@ -45,6 +48,11 @@ class LBUG_API CatalogSet { void serializeSnapshot(common::Serializer serializer, const transaction::Transaction* snapshotTxn) const; static std::unique_ptr deserialize(common::Deserializer& deserializer); + static std::unique_ptr deserialize(Catalog* catalog, + common::Deserializer& deserializer); + + Catalog* getCatalog() const { return catalog; } + std::string getOwnerCatalogName() const; common::oid_t getNextOID() { std::unique_lock lck{mtx}; @@ -86,6 +94,7 @@ class LBUG_API CatalogSet { private: mutable std::shared_mutex mtx; + Catalog* catalog = nullptr; common::oid_t nextOID = 0; common::case_insensitive_map_t> entries; }; diff --git a/src/include/main/database_manager.h b/src/include/main/database_manager.h index 618d809273..40ced05737 100644 --- a/src/include/main/database_manager.h +++ b/src/include/main/database_manager.h @@ -1,6 +1,9 @@ #pragma once +#include #include +#include +#include #include "attached_database.h" #include "storage/partition_storage_registry.h" @@ -8,6 +11,7 @@ namespace lbug { namespace catalog { class Catalog; +class GraphCatalogEntry; } // namespace catalog namespace storage { @@ -19,9 +23,12 @@ class PartitionStorageRegistry; namespace main { +struct GraphWALReplayRequest; + class DatabaseManager { public: DatabaseManager(); + ~DatabaseManager(); void registerAttachedDatabase(std::unique_ptr attachedDatabase); bool hasAttachedDatabase(const std::string& name); @@ -40,14 +47,32 @@ class DatabaseManager { void loadGraphsFromCatalog(storage::MemoryManager* memoryManager, main::ClientContext* clientContext, bool mainCheckpointCommitted, bool checkpointBundle, std::optional legacyCheckpointDatabaseID = std::nullopt); + // Materializes one named graph on demand (see loadGraphsFromCatalog for the batch + // equivalent). Used during WAL replay when a record tagged with the graph's name + // arrives before the graph is loaded. Returns true when the graph is loaded afterwards. + bool loadGraphFromCatalog(storage::MemoryManager* memoryManager, + main::ClientContext* clientContext, const std::string& graphName); + // Replays graph WALs queued by loadGraphFromCatalog. Deferred while a recovery + // transaction is active and run at the next transaction-free point (a replayed commit + // or the end of a replay pass). + void replayPendingGraphWALs(main::ClientContext* clientContext); void setDefaultGraph(const std::string& graphName); void clearDefaultGraph(); bool hasGraph(const std::string& graphName); catalog::Catalog* getGraphCatalog(const std::string& graphName); + // Runs action with the graph registry's shared lock held, so a concurrent DROP GRAPH + // cannot destroy the catalog for the action's duration. Callers that dereference the + // catalog after a getGraphCatalog lookup must use this instead; getGraphCatalog only + // guards the lookup itself. + void withGraphCatalog(const std::string& graphName, + const std::function& action); catalog::Catalog* getDefaultGraphCatalog() const; + catalog::Catalog* getReplayOwnerCatalog() const { return replayOwnerCatalog; } + void setReplayOwnerCatalog(catalog::Catalog* catalog) { replayOwnerCatalog = catalog; } bool hasDefaultGraph() const { return defaultGraph != "" && defaultGraph != "main"; } std::string getDefaultGraphName() const { return defaultGraph; } std::vector getGraphs() const; + void bumpGraphCatalogVersions(const std::unordered_set& catalogs); storage::StorageManager* getDefaultGraphStorageManager() const; LBUG_API void invalidateCache(); @@ -62,12 +87,19 @@ class DatabaseManager { LBUG_API static DatabaseManager* Get(const ClientContext& context); private: + bool loadGraph(main::ClientContext* clientContext, storage::MemoryManager* memoryManager, + catalog::GraphCatalogEntry* graphEntry, bool mainCheckpointCommitted, bool checkpointBundle, + std::optional legacyCheckpointDatabaseID); + std::vector> attachedDatabases; std::string defaultDatabase; std::vector> graphs; + mutable std::shared_mutex graphsMutex; // Owns the per-partition data files of partitioned node tables (phase-B per-partition // storage; see docs/partitioning.md 6b). storage::PartitionStorageRegistry partitionStorageRegistry; + catalog::Catalog* replayOwnerCatalog = nullptr; + std::vector> pendingGraphWALReplays; public: storage::PartitionStorageRegistry* getPartitionStorageRegistry() { diff --git a/src/include/storage/local_storage/local_storage.h b/src/include/storage/local_storage/local_storage.h index a1a714bb86..0a445fca71 100644 --- a/src/include/storage/local_storage/local_storage.h +++ b/src/include/storage/local_storage/local_storage.h @@ -1,5 +1,6 @@ #pragma once +#include #include #include "common/copy_constructors.h" @@ -11,6 +12,22 @@ namespace main { class ClientContext; } // namespace main namespace storage { +// Local tables are keyed by their owning catalog as well as their table ID: table IDs are +// per-catalog, so a transaction touching several graphs can hold uncommitted state for +// same-ID tables in different catalogs. +struct LocalTableKey { + std::string ownerCatalogName; + common::table_id_t tableID; + bool operator==(const LocalTableKey&) const = default; +}; + +struct LocalTableKeyHash { + std::size_t operator()(const LocalTableKey& key) const { + return std::hash{}(key.ownerCatalogName) ^ + (std::hash{}(key.tableID) << 1); + } +}; + // Data structures in LocalStorage are not thread-safe. // For now, we only support single thread insertions and updates. Once we optimize them with // multiple threads, LocalStorage and its related data structures should be reworked to be @@ -23,7 +40,7 @@ class LocalStorage { // Do nothing if the table already exists, otherwise create a new local table. LocalTable* getOrCreateLocalTable(Table& table); // Return nullptr if no local table exists. - LocalTable* getLocalTable(common::table_id_t tableID) const; + LocalTable* getLocalTable(const Table& table) const; // Optimistic page allocation is scoped to one storage manager (each partition child has // its own data file and page manager). `sm == nullptr` selects the main database file. @@ -37,7 +54,7 @@ class LocalStorage { private: main::ClientContext& clientContext; - std::unordered_map> tables; + std::unordered_map, LocalTableKeyHash> tables; // The mutex is only needed when working with the optimistic allocators std::mutex mtx; diff --git a/src/include/storage/local_storage/local_table.h b/src/include/storage/local_storage/local_table.h index 6359cdbb26..96e36269e6 100644 --- a/src/include/storage/local_storage/local_table.h +++ b/src/include/storage/local_storage/local_table.h @@ -27,6 +27,8 @@ class LocalTable { virtual common::TableType getTableType() const = 0; virtual common::row_idx_t getNumTotalRows() = 0; + const Table& getTable() const { return table; } + template const TARGET& constCast() { return common::dynamic_cast_checked(*this); diff --git a/src/include/storage/partition_storage_registry.h b/src/include/storage/partition_storage_registry.h index 1b30b19de2..1c1d6c2fea 100644 --- a/src/include/storage/partition_storage_registry.h +++ b/src/include/storage/partition_storage_registry.h @@ -61,11 +61,21 @@ class PartitionStorageRegistry { // children resolve through the main StorageManager exactly as before. static storage::NodeTable* resolveNodeTable(main::ClientContext* context, catalog::TableCatalogEntry& entry); + static storage::NodeTable* resolveNodeTable(main::ClientContext* context, + catalog::TableCatalogEntry& entry, catalog::Catalog* ownerCatalog); + + // StorageManager of an owner catalog resolved from a WAL/local-storage owner tag. Graph + // catalogs own their storage managers; the main catalog never does — main's tables live + // in the main database's StorageManager, owned by main::Database. + static storage::StorageManager* resolveOwnerStorageManager(main::ClientContext* context, + catalog::Catalog* ownerCatalog); // By-ID variant for paths that only carry a table ID (local-storage commit, WAL replay). // Throws if the ID is unknown to the catalog. static storage::NodeTable* resolveNodeTableByID(main::ClientContext* context, common::table_id_t tableID); + static storage::NodeTable* resolveNodeTableByID(main::ClientContext* context, + common::table_id_t tableID, catalog::Catalog* ownerCatalog); // Closes each listed child's file handles and deletes its data + WAL files. Used by the // DROP-parent cascade and by rollback cleanup of dynamically created partitions. diff --git a/src/include/storage/table/table.h b/src/include/storage/table/table.h index b2a9a240ad..83f0fc1bf7 100644 --- a/src/include/storage/table/table.h +++ b/src/include/storage/table/table.h @@ -166,6 +166,7 @@ class LBUG_API Table { common::TableType getTableType() const { return tableType; } common::table_id_t getTableID() const { return tableID; } std::string getTableName() const { return tableName; } + const std::string& getOwnerCatalogName() const { return ownerCatalogName; } // The StorageManager that owns this table's data file. Partition children report their // own per-partition manager, not the main database's. StorageManager* getStorageManager() const { return storageManager; } @@ -227,6 +228,7 @@ class LBUG_API Table { common::TableType tableType; common::table_id_t tableID; std::string tableName; + std::string ownerCatalogName; bool enableCompression; MemoryManager* memoryManager; StorageManager* storageManager; diff --git a/src/include/storage/wal/local_wal.h b/src/include/storage/wal/local_wal.h index 64cff4b305..0039d799f9 100644 --- a/src/include/storage/wal/local_wal.h +++ b/src/include/storage/wal/local_wal.h @@ -24,28 +24,36 @@ class LocalWAL { public: explicit LocalWAL(MemoryManager& mm, bool enableChecksums); - void logCreateCatalogEntryRecord(catalog::CatalogEntry* catalogEntry, bool isInternal); - void logCreateIndexRecord(catalog::CatalogEntry* catalogEntry, IndexInfo indexInfo, - std::vector treeBytes); - void logDropCatalogEntryRecord(uint64_t tableID, catalog::CatalogEntryType type); - void logAlterCatalogEntryRecord(const binder::BoundAlterInfo* alterInfo); - void logUpdateSequenceRecord(common::sequence_id_t sequenceID, uint64_t kCount); + void logCreateCatalogEntryRecord(const std::string& ownerCatalogName, + catalog::CatalogEntry* catalogEntry, bool isInternal); + void logCreateIndexRecord(const std::string& ownerCatalogName, + catalog::CatalogEntry* catalogEntry, IndexInfo indexInfo, std::vector treeBytes); + void logDropCatalogEntryRecord(const std::string& ownerCatalogName, uint64_t tableID, + catalog::CatalogEntryType type); + void logAlterCatalogEntryRecord(const std::string& ownerCatalogName, + const binder::BoundAlterInfo* alterInfo); + void logUpdateSequenceRecord(const std::string& ownerCatalogName, + common::sequence_id_t sequenceID, uint64_t kCount); - void logTableInsertion(common::table_id_t tableID, common::TableType tableType, - common::row_idx_t numRows, const std::vector& vectors); - void logNodeDeletion(common::table_id_t tableID, common::offset_t nodeOffset, - common::ValueVector* pkVector); - void logNodeUpdate(common::table_id_t tableID, common::column_id_t columnID, - common::offset_t nodeOffset, common::ValueVector* propertyVector); - void logRelDelete(common::table_id_t tableID, common::ValueVector* srcNodeVector, - common::ValueVector* dstNodeVector, common::ValueVector* relIDVector); - void logRelDetachDelete(common::table_id_t tableID, common::RelDataDirection direction, - common::ValueVector* srcNodeVector); - void logRelUpdate(common::table_id_t tableID, common::column_id_t columnID, + void logTableInsertion(const std::string& ownerCatalogName, common::table_id_t tableID, + common::TableType tableType, common::row_idx_t numRows, + const std::vector& vectors); + void logNodeDeletion(const std::string& ownerCatalogName, common::table_id_t tableID, + common::offset_t nodeOffset, common::ValueVector* pkVector); + void logNodeUpdate(const std::string& ownerCatalogName, common::table_id_t tableID, + common::column_id_t columnID, common::offset_t nodeOffset, + common::ValueVector* propertyVector); + void logRelDelete(const std::string& ownerCatalogName, common::table_id_t tableID, common::ValueVector* srcNodeVector, common::ValueVector* dstNodeVector, - common::ValueVector* relIDVector, common::ValueVector* propertyVector); + common::ValueVector* relIDVector); + void logRelDetachDelete(const std::string& ownerCatalogName, common::table_id_t tableID, + common::RelDataDirection direction, common::ValueVector* srcNodeVector); + void logRelUpdate(const std::string& ownerCatalogName, common::table_id_t tableID, + common::column_id_t columnID, common::ValueVector* srcNodeVector, + common::ValueVector* dstNodeVector, common::ValueVector* relIDVector, + common::ValueVector* propertyVector); - void logLoadExtension(std::string path); + void logLoadExtension(const std::string& ownerCatalogName, std::string path); void logCommit(); diff --git a/src/include/storage/wal/record/wal_record_base.h b/src/include/storage/wal/record/wal_record_base.h index 2f40f81601..e0f55b741a 100644 --- a/src/include/storage/wal/record/wal_record_base.h +++ b/src/include/storage/wal/record/wal_record_base.h @@ -2,6 +2,7 @@ #include #include +#include #include "common/cast.h" #include "common/copy_constructors.h" @@ -50,6 +51,7 @@ struct WALHeader { struct WALRecord { WALRecordType type = WALRecordType::INVALID_RECORD; + std::string ownerCatalogName; WALRecord() = default; explicit WALRecord(WALRecordType type) : type{type} {} diff --git a/src/include/storage/wal/wal.h b/src/include/storage/wal/wal.h index 407b8a0eaa..dbe89e9cfb 100644 --- a/src/include/storage/wal/wal.h +++ b/src/include/storage/wal/wal.h @@ -17,7 +17,7 @@ class LocalWAL; class StorageManager; class WAL { public: - static constexpr uint64_t CHECKPOINT_BUNDLE_FORMAT_VERSION = 1; + static constexpr uint64_t CHECKPOINT_BUNDLE_FORMAT_VERSION = 2; // Recovery only: adopt the frozen WAL for one checkpoint, clearing the request on exit even // if the checkpoint fails before rotation. diff --git a/src/include/storage/wal/wal_replayer.h b/src/include/storage/wal/wal_replayer.h index a019018451..d2e028c374 100644 --- a/src/include/storage/wal/wal_replayer.h +++ b/src/include/storage/wal/wal_replayer.h @@ -2,11 +2,17 @@ #include #include +#include +#include #include #include "storage/wal/wal_record.h" namespace lbug { +namespace catalog { +class Catalog; +} // namespace catalog + namespace main { class ClientContext; } // namespace main @@ -47,6 +53,10 @@ class WALReplayer { }; void replayWALRecord(WALRecord& walRecord) const; + void recordReplayedEntryID(catalog::CatalogEntryType entryType, common::oid_t recordedEntryID, + common::oid_t replayedEntryID) const; + common::oid_t getReplayedEntryID(catalog::CatalogEntryType entryType, + common::oid_t recordedEntryID) const; void replayCreateCatalogEntryRecord(WALRecord& walRecord) const; void replayCreateIndexRecord(WALRecord& walRecord) const; void replayDropCatalogEntryRecord(const WALRecord& walRecord) const; @@ -100,6 +110,22 @@ class WALReplayer { std::string walPath; std::string checkpointWalPath; std::string shadowFilePath; + // Owner names whose materialization already failed during this replay pass, so + // their remaining records skip the graph-recovery pass instead of repeating it. + // Cleared whenever replayed graph DDL can change which owners are loadable. + mutable std::unordered_set failedOwnerNames; + // Entry IDs recorded in a graph's WAL come from the recording session's view of its + // catalog, which can differ from the replaying catalog: a standalone session's plain + // catalog lacks the ANY-graph infrastructure entries the graph materializes with, + // and rolled-back CREATEs consume IDs without leaving WAL records. CREATE replay + // records each recorded->replayed ID per entry type (each type has an independent + // ID space); ID-addressed records translate through this map, scoped to the owning + // catalog the record replays against, and fall back to the recorded ID. Cleared per + // replayGraphWAL call. + mutable std::unordered_map>> + replayedEntryIDs; }; } // namespace storage diff --git a/src/include/transaction/transaction.h b/src/include/transaction/transaction.h index 44b0c78edd..bcc7df8044 100644 --- a/src/include/transaction/transaction.h +++ b/src/include/transaction/transaction.h @@ -3,6 +3,7 @@ #include #include #include +#include #include #include "common/types/types.h" @@ -12,6 +13,7 @@ namespace binder { struct BoundAlterInfo; } namespace catalog { +class Catalog; class CatalogEntry; class CatalogSet; class SequenceCatalogEntry; @@ -21,6 +23,7 @@ namespace main { class ClientContext; } // namespace main namespace storage { +class Table; class LocalWAL; class LocalStorage; class UndoBuffer; @@ -126,14 +129,14 @@ class LBUG_API Transaction { storage::LocalStorage* getLocalStorage() const { return localStorage.get(); } LocalCacheManager& getLocalCacheManager() { return localCacheManager; } - bool isUnCommitted(common::table_id_t tableID, common::offset_t nodeOffset) const; - common::row_idx_t getLocalRowIdx(common::table_id_t tableID, + bool isUnCommitted(const storage::Table& table, common::offset_t nodeOffset) const; + common::row_idx_t getLocalRowIdx(const storage::Table& table, common::offset_t nodeOffset) const { - return nodeOffset - getMinUncommittedNodeOffset(tableID); + return nodeOffset - getMinUncommittedNodeOffset(table); } - common::offset_t getUncommittedOffset(common::table_id_t tableID, + common::offset_t getUncommittedOffset(const storage::Table& table, common::row_idx_t localRowIdx) const { - return getMinUncommittedNodeOffset(tableID) + localRowIdx; + return getMinUncommittedNodeOffset(table) + localRowIdx; } main::ClientContext* getClientContext() const { return clientContext; } @@ -154,7 +157,8 @@ class LBUG_API Transaction { static Transaction* Get(const main::ClientContext& context); private: - common::offset_t getMinUncommittedNodeOffset(common::table_id_t tableID) const; + common::offset_t getMinUncommittedNodeOffset(const storage::Table& table) const; + void recordCatalogChange(catalog::Catalog* catalog); private: TransactionType type; @@ -170,7 +174,8 @@ class LBUG_API Transaction { std::vector> commitCallbacks; std::vector> rollbackCallbacks; bool forceCheckpoint; - std::atomic hasCatalogChanges; + std::unordered_set changedCatalogs; + std::mutex changedCatalogsMutex; }; // TODO(bmwinger): These shouldn't need to be exported diff --git a/src/main/database_manager.cpp b/src/main/database_manager.cpp index 651742aff5..60fe791613 100644 --- a/src/main/database_manager.cpp +++ b/src/main/database_manager.cpp @@ -39,6 +39,18 @@ namespace main { DatabaseManager::DatabaseManager() : defaultDatabase{""} {} +DatabaseManager::~DatabaseManager() = default; + +// A graph materialized during an active recovery transaction cannot replay its own WAL +// inline: graph WALs hold standalone-session transactions whose BEGIN/COMMIT would nest +// inside the caller's recovery transaction. The loader queues one of these and +// replayPendingGraphWALs() runs the queued replays at the next transaction-free point. +struct GraphWALReplayRequest { + storage::StorageManager* storageManager = nullptr; + std::string graphName; + storage::WALReplayer::GraphRecoveryState recoveryState; +}; + void DatabaseManager::registerAttachedDatabase(std::unique_ptr attachedDatabase) { if (defaultDatabase == "") { defaultDatabase = attachedDatabase->getDBName(); @@ -111,6 +123,56 @@ DatabaseManager* DatabaseManager::Get(const ClientContext& context) { return context.getDatabase()->getDatabaseManager(); } +static void createAnyGraphTables(catalog::Catalog& catalog) { + // Use DUMMY_CHECKPOINT_TRANSACTION to create tables + auto* dummyTransaction = &transaction::DUMMY_CHECKPOINT_TRANSACTION; + + // Create serial name for the id column: _nodes_id_serial + auto serialName = "_nodes_id_serial"; + auto serialLiteral = + std::make_unique(Value(serialName), serialName); + auto serialDefault = std::make_unique( + function::NextValFunction::name, std::move(serialLiteral), serialName); + + std::vector nodeProperties; + nodeProperties.emplace_back(binder::PropertyDefinition( + binder::ColumnDefinition("id", common::LogicalType::SERIAL()), std::move(serialDefault))); + nodeProperties.emplace_back(binder::PropertyDefinition(binder::ColumnDefinition("label", + common::LogicalType::LIST(common::LogicalType::STRING())))); + nodeProperties.emplace_back( + binder::PropertyDefinition(binder::ColumnDefinition("data", LogicalType::JSON()))); + + auto nodeExtraInfo = std::make_unique("id", + std::move(nodeProperties), ""); + auto nodeTableInfo = binder::BoundCreateTableInfo(catalog::CatalogEntryType::NODE_TABLE_ENTRY, + "_nodes", common::ConflictAction::ON_CONFLICT_THROW, std::move(nodeExtraInfo), false); + auto* nodeEntry = catalog.createTableEntry(dummyTransaction, nodeTableInfo); + // Mark entry as committed so it's visible to all transactions + nodeEntry->setTimestamp(0); + catalog.getStorageManager()->createTable(nodeEntry->ptrCast()); + auto nodeTableID = nodeEntry->ptrCast()->getTableID(); + + std::vector relProperties; + relProperties.emplace_back(binder::ColumnDefinition("_id", common::LogicalType::INTERNAL_ID())); + relProperties.emplace_back(binder::ColumnDefinition("label", common::LogicalType::STRING())); + relProperties.emplace_back(binder::ColumnDefinition("data", LogicalType::JSON())); + + std::vector relTableInfos; + relTableInfos.emplace_back(catalog::NodeTableIDPair(nodeTableID, nodeTableID), + common::RelMultiplicity::MANY, common::RelMultiplicity::MANY); + + auto relExtraInfo = std::unique_ptr( + new binder::BoundExtraCreateRelTableGroupInfo(std::move(relProperties), + common::RelMultiplicity::MANY, common::RelMultiplicity::MANY, + common::ExtendDirection::BOTH, std::move(relTableInfos), std::string(""))); + auto relTableInfo = binder::BoundCreateTableInfo(catalog::CatalogEntryType::REL_GROUP_ENTRY, + "_edges", common::ConflictAction::ON_CONFLICT_THROW, std::move(relExtraInfo), false); + auto* relEntry = catalog.createTableEntry(dummyTransaction, relTableInfo); + // Mark entry as committed so it's visible to all transactions + relEntry->setTimestamp(0); + catalog.getStorageManager()->createTable(relEntry->ptrCast()); +} + void DatabaseManager::createGraph(const std::string& graphName, storage::MemoryManager* memoryManager, main::ClientContext* clientContext, bool isAnyGraph) { if (StringUtils::caseInsensitiveEquals(graphName, "main")) { @@ -154,60 +216,13 @@ void DatabaseManager::createGraph(const std::string& graphName, catalog->setStorageManager(std::move(storageManager)); if (isAnyGraph) { - // Use DUMMY_CHECKPOINT_TRANSACTION to create tables - auto* dummyTransaction = &transaction::DUMMY_CHECKPOINT_TRANSACTION; - - // Create serial name for the id column: _nodes_id_serial - auto serialName = "_nodes_id_serial"; - auto serialLiteral = - std::make_unique(Value(serialName), serialName); - auto serialDefault = std::make_unique( - function::NextValFunction::name, std::move(serialLiteral), serialName); - - std::vector nodeProperties; - nodeProperties.emplace_back(binder::PropertyDefinition( - binder::ColumnDefinition("id", common::LogicalType::SERIAL()), - std::move(serialDefault))); - nodeProperties.emplace_back(binder::PropertyDefinition(binder::ColumnDefinition("label", - common::LogicalType::LIST(common::LogicalType::STRING())))); - nodeProperties.emplace_back( - binder::PropertyDefinition(binder::ColumnDefinition("data", LogicalType::JSON()))); - - auto nodeExtraInfo = std::make_unique("id", - std::move(nodeProperties), ""); - auto nodeTableInfo = - binder::BoundCreateTableInfo(catalog::CatalogEntryType::NODE_TABLE_ENTRY, "_nodes", - common::ConflictAction::ON_CONFLICT_THROW, std::move(nodeExtraInfo), false); - auto* nodeEntry = catalog->createTableEntry(dummyTransaction, nodeTableInfo); - // Mark entry as committed so it's visible to all transactions - nodeEntry->setTimestamp(0); - catalog->getStorageManager()->createTable(nodeEntry->ptrCast()); - auto nodeTableID = nodeEntry->ptrCast()->getTableID(); - - std::vector relProperties; - relProperties.emplace_back( - binder::ColumnDefinition("_id", common::LogicalType::INTERNAL_ID())); - relProperties.emplace_back( - binder::ColumnDefinition("label", common::LogicalType::STRING())); - relProperties.emplace_back(binder::ColumnDefinition("data", LogicalType::JSON())); - - std::vector relTableInfos; - relTableInfos.emplace_back(catalog::NodeTableIDPair(nodeTableID, nodeTableID), - common::RelMultiplicity::MANY, common::RelMultiplicity::MANY); - - auto relExtraInfo = std::unique_ptr( - new binder::BoundExtraCreateRelTableGroupInfo(std::move(relProperties), - common::RelMultiplicity::MANY, common::RelMultiplicity::MANY, - common::ExtendDirection::BOTH, std::move(relTableInfos), std::string(""))); - auto relTableInfo = binder::BoundCreateTableInfo(catalog::CatalogEntryType::REL_GROUP_ENTRY, - "_edges", common::ConflictAction::ON_CONFLICT_THROW, std::move(relExtraInfo), false); - auto* relEntry = catalog->createTableEntry(dummyTransaction, relTableInfo); - // Mark entry as committed so it's visible to all transactions - relEntry->setTimestamp(0); - catalog->getStorageManager()->createTable(relEntry->ptrCast()); - } - - graphs.push_back(std::move(catalog)); + createAnyGraphTables(*catalog); + } + + { + std::unique_lock lck{graphsMutex}; + graphs.push_back(std::move(catalog)); + } // NOTE: Do NOT set defaultGraph here. Setting defaultGraph before the transaction // commits causes Catalog::Get() in Transaction::publishCommit() to return the graph's // catalog instead of the main catalog, so the main catalog's version is never @@ -241,6 +256,7 @@ void DatabaseManager::dropGraph(const std::string& graphName, main::ClientContex // Remove from system catalog mainCatalog->dropGraph(transaction, graphName); + std::unique_lock lck{graphsMutex}; for (auto it = graphs.begin(); it != graphs.end(); ++it) { auto graphNameUpper = StringUtils::getUpper((*it)->getCatalogName()); if (graphNameUpper == upperCaseName) { @@ -257,6 +273,7 @@ void DatabaseManager::dropGraph(const std::string& graphName, main::ClientContex detachDatabase(graphName); } graphs.erase(it); + lck.unlock(); // Delete the physical graph files if (!graphPath.empty() && @@ -290,6 +307,7 @@ void DatabaseManager::setDefaultGraph(const std::string& graphName) { defaultGraph = "main"; return; } + std::shared_lock lck{graphsMutex}; for (auto& graph : graphs) { auto graphNameUpper = StringUtils::getUpper(graph->getCatalogName()); if (graphNameUpper == upperCaseName) { @@ -313,14 +331,23 @@ void DatabaseManager::loadGraphsFromCatalog(storage::MemoryManager* memoryManage main::ClientContext* clientContext, bool mainCheckpointCommitted, bool checkpointBundle, std::optional legacyCheckpointDatabaseID) { auto mainCatalog = clientContext->getDatabase()->getCatalog(); - // Use DUMMY_CHECKPOINT_TRANSACTION since we're loading from disk during startup - // and there's no active transaction yet - auto* transaction = &transaction::DUMMY_CHECKPOINT_TRANSACTION; + // During WAL replay, a tagged record can trigger materialization of a graph whose + // create-graph record is part of the still-uncommitted recovery transaction; enumerate + // with that transaction so such a graph is visible to the loader. Every replayed BEGIN is + // guaranteed to reach its COMMIT record (dryReplay advances the replay bound only past + // COMMIT records), so a graph visible here can never belong to an aborted transaction. + // Outside replay there is no active transaction and DUMMY_CHECKPOINT_TRANSACTION applies. + auto* transaction = TransactionContext::Get(*clientContext)->hasActiveTransaction() ? + Transaction::Get(*clientContext) : + &transaction::DUMMY_CHECKPOINT_TRANSACTION; auto graphEntries = mainCatalog->getGraphEntries(transaction); std::unordered_set loadedGraphNames; - loadedGraphNames.reserve(graphs.size()); - for (const auto& graph : graphs) { - loadedGraphNames.insert(StringUtils::getUpper(graph->getCatalogName())); + { + std::shared_lock lck{graphsMutex}; + loadedGraphNames.reserve(graphs.size()); + for (const auto& graph : graphs) { + loadedGraphNames.insert(StringUtils::getUpper(graph->getCatalogName())); + } } for (auto* graphEntry : graphEntries) { @@ -339,66 +366,155 @@ void DatabaseManager::loadGraphsFromCatalog(storage::MemoryManager* memoryManage continue; } - // Load the graph - auto catalog = std::make_unique(); - catalog->setCatalogName(graphName); - // Extension functions are registered in the main catalog only; let - // function lookup fall back to it while the session is on this graph. - catalog->setFunctionFallback(mainCatalog); - auto dbPath = clientContext->getDatabasePath(); - auto graphPath = DBConfig::isDBPathInMemory(dbPath) ? - ":" + graphName : - storage::StorageUtils::getGraphPath(dbPath, graphName); - - // Check if graph file exists before trying to load - auto vfs = common::VirtualFileSystem::GetUnsafe(*clientContext); - if (!DBConfig::isDBPathInMemory(dbPath) && !vfs->fileOrPathExists(graphPath)) { - if (mainCheckpointCommitted && - vfs->fileOrPathExists(storage::StorageUtils::getShadowFilePath(graphPath))) { - throw RuntimeException(std::format( - "Cannot recover committed checkpoint: graph file {} is missing.", graphPath)); - } - // Graph file doesn't exist, skip this graph - continue; + if (loadGraph(clientContext, memoryManager, graphEntry, mainCheckpointCommitted, + checkpointBundle, legacyCheckpointDatabaseID)) { + loadedGraphNames.insert(std::move(upperCaseName)); } + } +} + +bool DatabaseManager::loadGraph(main::ClientContext* clientContext, + storage::MemoryManager* memoryManager, catalog::GraphCatalogEntry* graphEntry, + bool mainCheckpointCommitted, bool checkpointBundle, + std::optional legacyCheckpointDatabaseID) { + auto graphName = graphEntry->getName(); + auto catalog = std::make_unique(); + catalog->setCatalogName(graphName); + // Extension functions are registered in the main catalog only; let + // function lookup fall back to it while the session is on this graph. + auto* mainCatalog = clientContext->getDatabase()->getCatalog(); + catalog->setFunctionFallback(mainCatalog); + auto dbPath = clientContext->getDatabasePath(); + auto graphPath = DBConfig::isDBPathInMemory(dbPath) ? + ":" + graphName : + storage::StorageUtils::getGraphPath(dbPath, graphName); - auto storageManager = std::make_unique(graphPath, - clientContext->getDBConfig()->readOnly, false, *memoryManager, false, - clientContext->getDBConfig()->enableDefaultHashIndex, vfs); - storageManager->initDataFileHandle(vfs, clientContext); - storage::WALReplayer walReplayer{*clientContext}; - auto recoveryState = walReplayer.prepareGraphCheckpoint(*storageManager, checkpointBundle, - legacyCheckpointDatabaseID); - if (storageManager->getDataFH()->getNumPages() > 0) { - storage::Checkpointer::readCheckpoint(clientContext, catalog.get(), - storageManager.get()); + // Check if graph file exists before trying to load + auto vfs = common::VirtualFileSystem::GetUnsafe(*clientContext); + if (!DBConfig::isDBPathInMemory(dbPath) && !vfs->fileOrPathExists(graphPath)) { + if (mainCheckpointCommitted && + vfs->fileOrPathExists(storage::StorageUtils::getShadowFilePath(graphPath))) { + throw RuntimeException(std::format( + "Cannot recover committed checkpoint: graph file {} is missing.", graphPath)); } - catalog->setStorageManager(std::move(storageManager)); - auto* graphStorageManager = catalog->getStorageManager(); + // Graph file doesn't exist, skip this graph + return false; + } + + auto storageManager = std::make_unique(graphPath, + clientContext->getDBConfig()->readOnly, false, *memoryManager, false, + clientContext->getDBConfig()->enableDefaultHashIndex, vfs); + storageManager->initDataFileHandle(vfs, clientContext); + storage::WALReplayer walReplayer{*clientContext}; + auto recoveryState = walReplayer.prepareGraphCheckpoint(*storageManager, checkpointBundle, + legacyCheckpointDatabaseID); + bool hasPersistedCatalog = false; + if (storageManager->getDataFH()->getNumPages() > 0) { + auto persistedHeader = storage::DatabaseHeader::readDatabaseHeader( + *storageManager->getDataFH()->getFileInfo()); + hasPersistedCatalog = + persistedHeader.has_value() && + persistedHeader->catalogPageRange.startPageIdx != common::INVALID_PAGE_IDX; + storage::Checkpointer::readCheckpoint(clientContext, catalog.get(), storageManager.get()); + } + catalog->setStorageManager(std::move(storageManager)); + if (graphEntry->isAnyGraphType() && !hasPersistedCatalog) { + // A graph whose create-graph record replayed from the WAL (never checkpointed) + // materializes without the ANY-graph infrastructure that createGraph built in + // memory: it was not WAL-logged. Recreate it so the catalog's ID space matches + // logging time and ANY-graph label queries keep working. An empty table set in + // a persisted snapshot is deliberate state, not missing initialization. + createAnyGraphTables(*catalog); + } + auto* graphStorageManager = catalog->getStorageManager(); + { + std::unique_lock lck{graphsMutex}; graphs.push_back(std::move(catalog)); - loadedGraphNames.insert(std::move(upperCaseName)); + } + + if (TransactionContext::Get(*clientContext)->hasActiveTransaction()) { + // Lazy materialization runs inside the caller's recovery transaction, and a + // graph WAL holds standalone-session transactions: replaying it here would + // begin a second recovery transaction nested inside the caller's. Queue the + // replay for the next transaction-free point (replayPendingGraphWALs) instead. + pendingGraphWALReplays.push_back(std::make_unique( + GraphWALReplayRequest{graphStorageManager, graphName, std::move(recoveryState)})); + return true; + } + const auto previousDefaultGraph = defaultGraph; + defaultGraph = graphName; + try { + walReplayer.replayGraphWAL(*graphStorageManager, recoveryState); + } catch (...) { + defaultGraph = previousDefaultGraph; + throw; + } + defaultGraph = previousDefaultGraph; + walReplayer.retireGraphCheckpointWALs(*graphStorageManager, recoveryState); + if (!mainCheckpointCommitted) { + walReplayer.removeGraphCheckpointShadow(*graphStorageManager); + } + // NOTE: defaultGraph is only set around replayGraphWAL above (see createGraph for + // the rationale against leaving it set). Users must explicitly USE GRAPH to work + // in the graph. + return true; +} + +bool DatabaseManager::loadGraphFromCatalog(storage::MemoryManager* memoryManager, + main::ClientContext* clientContext, const std::string& graphName) { + auto mainCatalog = clientContext->getDatabase()->getCatalog(); + // Lazy materialization is only requested from inside WAL replay, where the active + // recovery transaction must see graphs whose create-graph record already replayed + // (see loadGraphsFromCatalog for the same visibility argument). + auto* transaction = TransactionContext::Get(*clientContext)->hasActiveTransaction() ? + Transaction::Get(*clientContext) : + &transaction::DUMMY_CHECKPOINT_TRANSACTION; + if (!mainCatalog->containsGraph(transaction, graphName)) { + return false; + } + // A partition subgraph's data files are owned by the partition storage registry and + // must not be opened here (see the loadGraphsFromCatalog loop). + if (mainCatalog->containsTable(transaction, graphName)) { + return false; + } + if (hasGraph(graphName)) { + return true; + } + auto* graphEntry = mainCatalog->getGraphEntry(transaction, graphName); + return loadGraph(clientContext, memoryManager, graphEntry, false /* mainCheckpointCommitted */, + false /* checkpointBundle */, std::nullopt); +} + +void DatabaseManager::replayPendingGraphWALs(main::ClientContext* clientContext) { + if (pendingGraphWALReplays.empty()) { + return; + } + auto pendingReplays = std::move(pendingGraphWALReplays); + pendingGraphWALReplays.clear(); + storage::WALReplayer walReplayer{*clientContext}; + for (auto& replayRequest : pendingReplays) { const auto previousDefaultGraph = defaultGraph; - defaultGraph = graphName; + defaultGraph = replayRequest->graphName; try { - walReplayer.replayGraphWAL(*graphStorageManager, recoveryState); + walReplayer.replayGraphWAL(*replayRequest->storageManager, + replayRequest->recoveryState); } catch (...) { defaultGraph = previousDefaultGraph; throw; } defaultGraph = previousDefaultGraph; - walReplayer.retireGraphCheckpointWALs(*graphStorageManager, recoveryState); - if (!mainCheckpointCommitted) { - walReplayer.removeGraphCheckpointShadow(*graphStorageManager); - } - - // NOTE: Do NOT set defaultGraph here (see createGraph for rationale). - // Users must explicitly USE GRAPH to work in the graph. + walReplayer.retireGraphCheckpointWALs(*replayRequest->storageManager, + replayRequest->recoveryState); + // Deferred requests only come from lazy materialization during main-WAL replay, + // which always runs with mainCheckpointCommitted=false. + walReplayer.removeGraphCheckpointShadow(*replayRequest->storageManager); } } bool DatabaseManager::hasGraph(const std::string& graphName) { auto upperCaseName = StringUtils::getUpper(graphName); + std::shared_lock lck{graphsMutex}; for (auto& graph : graphs) { auto graphNameUpper = StringUtils::getUpper(graph->getCatalogName()); if (graphNameUpper == upperCaseName) { @@ -410,6 +526,7 @@ bool DatabaseManager::hasGraph(const std::string& graphName) { catalog::Catalog* DatabaseManager::getGraphCatalog(const std::string& graphName) { auto upperCaseName = StringUtils::getUpper(graphName); + std::shared_lock lck{graphsMutex}; for (auto& graph : graphs) { auto graphNameUpper = StringUtils::getUpper(graph->getCatalogName()); if (graphNameUpper == upperCaseName) { @@ -419,11 +536,26 @@ catalog::Catalog* DatabaseManager::getGraphCatalog(const std::string& graphName) throw BinderException{std::format("No graph named {}.", graphName)}; } +void DatabaseManager::withGraphCatalog(const std::string& graphName, + const std::function& action) { + auto upperCaseName = StringUtils::getUpper(graphName); + std::shared_lock lck{graphsMutex}; + for (auto& graph : graphs) { + auto graphNameUpper = StringUtils::getUpper(graph->getCatalogName()); + if (graphNameUpper == upperCaseName) { + action(graph.get()); + return; + } + } + throw BinderException{std::format("No graph named {}.", graphName)}; +} + catalog::Catalog* DatabaseManager::getDefaultGraphCatalog() const { if (defaultGraph == "" || defaultGraph == "main") { return nullptr; } auto upperCaseName = StringUtils::getUpper(defaultGraph); + std::shared_lock lck{graphsMutex}; for (auto& graph : graphs) { auto graphNameUpper = StringUtils::getUpper(graph->getCatalogName()); if (graphNameUpper == upperCaseName) { @@ -443,12 +575,23 @@ storage::StorageManager* DatabaseManager::getDefaultGraphStorageManager() const std::vector DatabaseManager::getGraphs() const { std::vector result; + std::shared_lock lck{graphsMutex}; for (auto& graph : graphs) { result.push_back(graph.get()); } return result; } +void DatabaseManager::bumpGraphCatalogVersions( + const std::unordered_set& catalogs) { + std::shared_lock lck{graphsMutex}; + for (auto& graph : graphs) { + if (catalogs.contains(graph.get())) { + graph->incrementVersion(); + } + } +} + std::pair DatabaseManager::resolveTableStorage( const ClientContext& context, common::table_id_t tableID, const std::string& dbName) { if (!dbName.empty()) { diff --git a/src/processor/operator/ddl/create_index.cpp b/src/processor/operator/ddl/create_index.cpp index bcfb14093f..22c4a8a447 100644 --- a/src/processor/operator/ddl/create_index.cpp +++ b/src/processor/operator/ddl/create_index.cpp @@ -131,8 +131,8 @@ void CreateIndex::executeInternal(ExecutionContext* context) { auto treeBytes = physicalIndex.value()->cast().serializeTreeToBytes(); auto* indexEntry = catalog->getIndex(transaction, info.tableID, info.indexName); - transaction->getLocalWAL().logCreateIndexRecord(indexEntry, - physicalIndex.value()->getIndexInfo(), std::move(treeBytes)); + transaction->getLocalWAL().logCreateIndexRecord(indexEntry->getOwningCatalogName(), + indexEntry, physicalIndex.value()->getIndexInfo(), std::move(treeBytes)); } appendMessage(std::format("Index {} has been created.", info.indexName), memoryManager); return; diff --git a/src/processor/operator/scan/count_rel_table.cpp b/src/processor/operator/scan/count_rel_table.cpp index 4a8e648f58..dee39da5fe 100644 --- a/src/processor/operator/scan/count_rel_table.cpp +++ b/src/processor/operator/scan/count_rel_table.cpp @@ -127,8 +127,7 @@ bool CountRelTable::getNextTuplesInternal(ExecutionContext* context) { // Add uncommitted insertions from local storage if (transaction->isWriteTransaction()) { - if (auto* localTable = - transaction->getLocalStorage()->getLocalTable(relTable->getTableID())) { + if (auto* localTable = transaction->getLocalStorage()->getLocalTable(*relTable)) { auto& localRelTable = localTable->cast(); // Count entries in the CSR index for this direction. // We can't use getNumTotalRows() because it includes deleted rows. diff --git a/src/processor/operator/scan/scan_node_table.cpp b/src/processor/operator/scan/scan_node_table.cpp index 29f27da630..9a1457d6bd 100644 --- a/src/processor/operator/scan/scan_node_table.cpp +++ b/src/processor/operator/scan/scan_node_table.cpp @@ -92,8 +92,7 @@ void ScanNodeTableSharedState::initialize(const transaction::Transaction* transa this->numCommittedNodeGroups = table->getNumCommittedNodeGroups(); } if (transaction->isWriteTransaction()) { - if (const auto localTable = - transaction->getLocalStorage()->getLocalTable(this->table->getTableID())) { + if (const auto localTable = transaction->getLocalStorage()->getLocalTable(*this->table)) { auto& localNodeTable = localTable->cast(); this->numUnCommittedNodeGroups = localNodeTable.getNumNodeGroups(); } diff --git a/src/storage/local_storage/local_storage.cpp b/src/storage/local_storage/local_storage.cpp index f12e5eff66..500eb17959 100644 --- a/src/storage/local_storage/local_storage.cpp +++ b/src/storage/local_storage/local_storage.cpp @@ -1,6 +1,9 @@ #include "storage/local_storage/local_storage.h" #include "catalog/catalog.h" +#include "main/client_context.h" +#include "main/database.h" +#include "main/database_manager.h" #include "storage/local_storage/local_node_table.h" #include "storage/local_storage/local_rel_table.h" #include "storage/local_storage/local_table.h" @@ -16,33 +19,58 @@ using namespace lbug::transaction; namespace lbug { namespace storage { +namespace { +catalog::Catalog* getMainCatalog(main::ClientContext& clientContext) { + return clientContext.getDatabase()->getCatalog(); +} + +std::unique_ptr makeLocalTable(catalog::Catalog* ownerCatalog, + transaction::Transaction* transaction, const LocalTableKey& key, Table& table, + MemoryManager& mm) { + switch (table.getTableType()) { + case TableType::NODE: { + auto tableEntry = ownerCatalog->getTableCatalogEntry(transaction, key.tableID); + return std::make_unique(tableEntry, table, mm); + } break; + case TableType::REL: { + // We have to fetch the rel group entry from the catalog to based on the relGroupID. + auto tableEntry = + ownerCatalog->getTableCatalogEntry(transaction, table.cast().getRelGroupID()); + return std::make_unique(tableEntry, table, mm); + } break; + default: + UNREACHABLE_CODE; + } +} +} // namespace + LocalTable* LocalStorage::getOrCreateLocalTable(Table& table) { const auto tableID = table.getTableID(); - auto catalog = catalog::Catalog::Get(clientContext); + const auto key = LocalTableKey{table.getOwnerCatalogName(), tableID}; auto transaction = transaction::Transaction::Get(clientContext); auto& mm = *MemoryManager::Get(clientContext); - if (!tables.contains(tableID)) { - switch (table.getTableType()) { - case TableType::NODE: { - auto tableEntry = catalog->getTableCatalogEntry(transaction, table.getTableID()); - tables[tableID] = std::make_unique(tableEntry, table, mm); - } break; - case TableType::REL: { - // We have to fetch the rel group entry from the catalog to based on the relGroupID. - auto tableEntry = - catalog->getTableCatalogEntry(transaction, table.cast().getRelGroupID()); - tables[tableID] = std::make_unique(tableEntry, table, mm); - } break; - default: - UNREACHABLE_CODE; + if (!tables.contains(key)) { + // The owner catalog must stay alive while the local table reads its entry, so a + // graph owner is resolved under the registry's shared lock (withGraphCatalog) — + // a concurrent DROP GRAPH can destroy the catalog the moment a plain + // getGraphCatalog lookup releases it. + if (key.ownerCatalogName.empty()) { + tables[key] = + makeLocalTable(getMainCatalog(clientContext), transaction, key, table, mm); + } else { + main::DatabaseManager::Get(clientContext) + ->withGraphCatalog(key.ownerCatalogName, [&](catalog::Catalog* ownerCatalog) { + tables[key] = makeLocalTable(ownerCatalog, transaction, key, table, mm); + }); } } - return tables.at(tableID).get(); + return tables.at(key).get(); } -LocalTable* LocalStorage::getLocalTable(table_id_t tableID) const { - if (tables.contains(tableID)) { - return tables.at(tableID).get(); +LocalTable* LocalStorage::getLocalTable(const Table& table) const { + const auto key = LocalTableKey{table.getOwnerCatalogName(), table.getTableID()}; + if (tables.contains(key)) { + return tables.at(key).get(); } return nullptr; } @@ -66,23 +94,43 @@ PageAllocator* LocalStorage::addOptimisticAllocator(StorageManager* sm) { } void LocalStorage::commit() { - auto catalog = catalog::Catalog::Get(clientContext); auto transaction = transaction::Transaction::Get(clientContext); - auto storageManager = StorageManager::Get(clientContext); - for (auto& [tableID, localTable] : tables) { + for (auto& [key, localTable] : tables) { if (localTable->getTableType() == TableType::NODE) { - const auto tableEntry = catalog->getTableCatalogEntry(transaction, tableID); - const auto table = - storage::PartitionStorageRegistry::resolveNodeTableByID(&clientContext, tableID); - table->commit(&clientContext, tableEntry, localTable.get()); + // Catalog read, table resolution, and the commit below must run under one + // hold of the graph registry's shared lock: a concurrent DROP GRAPH must not + // destroy the owner catalog while staged graph data is being committed. + const auto commitInto = [&](catalog::Catalog* ownerCatalog) { + const auto tableEntry = + ownerCatalog->getTableCatalogEntry(transaction, key.tableID); + const auto table = storage::PartitionStorageRegistry::resolveNodeTableByID( + &clientContext, key.tableID, ownerCatalog); + table->commit(&clientContext, tableEntry, localTable.get()); + }; + if (key.ownerCatalogName.empty()) { + commitInto(getMainCatalog(clientContext)); + } else { + main::DatabaseManager::Get(clientContext) + ->withGraphCatalog(key.ownerCatalogName, commitInto); + } } } - for (auto& [tableID, localTable] : tables) { + for (auto& [key, localTable] : tables) { if (localTable->getTableType() == TableType::REL) { - const auto table = storageManager->getTable(tableID); - const auto tableEntry = - catalog->getTableCatalogEntry(transaction, table->cast().getRelGroupID()); - table->commit(&clientContext, tableEntry, localTable.get()); + const auto commitInto = [&](catalog::Catalog* ownerCatalog) { + const auto table = PartitionStorageRegistry::resolveOwnerStorageManager( + &clientContext, ownerCatalog) + ->getTable(key.tableID); + const auto tableEntry = ownerCatalog->getTableCatalogEntry(transaction, + table->cast().getRelGroupID()); + table->commit(&clientContext, tableEntry, localTable.get()); + }; + if (key.ownerCatalogName.empty()) { + commitInto(getMainCatalog(clientContext)); + } else { + main::DatabaseManager::Get(clientContext) + ->withGraphCatalog(key.ownerCatalogName, commitInto); + } } } for (auto& optimisticAllocator : optimisticAllocators) { diff --git a/src/storage/partition_storage_registry.cpp b/src/storage/partition_storage_registry.cpp index e7b90230e3..95d67f71bf 100644 --- a/src/storage/partition_storage_registry.cpp +++ b/src/storage/partition_storage_registry.cpp @@ -9,6 +9,7 @@ #include "common/serializer/buffered_file.h" #include "common/serializer/deserializer.h" #include "main/client_context.h" +#include "main/database.h" #include "main/database_manager.h" #include "main/db_config.h" #include "storage/database_header.h" @@ -114,7 +115,13 @@ bool PartitionStorageRegistry::isRemotelyRouted(table_id_t parentTableID, uint64 NodeTable* PartitionStorageRegistry::resolveNodeTable(main::ClientContext* context, TableCatalogEntry& entry) { - auto* mainSM = StorageManager::Get(*context); + return resolveNodeTable(context, entry, nullptr /* ambient */); +} + +NodeTable* PartitionStorageRegistry::resolveNodeTable(main::ClientContext* context, + TableCatalogEntry& entry, catalog::Catalog* ownerCatalog) { + auto* mainSM = ownerCatalog != nullptr ? resolveOwnerStorageManager(context, ownerCatalog) : + StorageManager::Get(*context); const auto tableID = entry.getTableID(); if (entry.getType() != CatalogEntryType::NODE_TABLE_ENTRY || !entry.ptrCast()->isPartitionChild()) { @@ -152,6 +159,25 @@ NodeTable* PartitionStorageRegistry::resolveNodeTableByID(main::ClientContext* c return resolveNodeTable(context, *entry); } +NodeTable* PartitionStorageRegistry::resolveNodeTableByID(main::ClientContext* context, + table_id_t tableID, catalog::Catalog* ownerCatalog) { + auto* ownerSM = resolveOwnerStorageManager(context, ownerCatalog); + if (ownerSM->containsTable(tableID)) { + return ownerSM->getTable(tableID)->ptrCast(); + } + auto* entry = + ownerCatalog->getTableCatalogEntry(transaction::Transaction::Get(*context), tableID); + return resolveNodeTable(context, *entry, ownerCatalog); +} + +StorageManager* PartitionStorageRegistry::resolveOwnerStorageManager(main::ClientContext* context, + catalog::Catalog* ownerCatalog) { + if (auto* sm = ownerCatalog->getStorageManager()) { + return sm; + } + return context->getDatabase()->getStorageManager(); +} + std::vector PartitionStorageRegistry::getAllManagers() { std::shared_lock lck{mtx}; std::vector out; diff --git a/src/storage/storage_manager.cpp b/src/storage/storage_manager.cpp index 7157c650dd..61a002d936 100644 --- a/src/storage/storage_manager.cpp +++ b/src/storage/storage_manager.cpp @@ -747,6 +747,10 @@ StorageManager* StorageManager::Get(const main::ClientContext& context) { return context.getAttachedDatabase()->getStorageManager(); } auto dbManager = main::DatabaseManager::Get(context); + if (auto* replayOwnerCatalog = dbManager->getReplayOwnerCatalog(); + replayOwnerCatalog != nullptr) { + return replayOwnerCatalog->getStorageManager(); + } auto graphStorageManager = dbManager->getDefaultGraphStorageManager(); if (graphStorageManager != nullptr) { return graphStorageManager; diff --git a/src/storage/table/node_table.cpp b/src/storage/table/node_table.cpp index 0f7f299942..b3e20f70c4 100644 --- a/src/storage/table/node_table.cpp +++ b/src/storage/table/node_table.cpp @@ -73,7 +73,7 @@ NodeGroupScanResult NodeTableScanState::scanNext(Transaction* transaction, offse auto nodeGroupStartOffset = StorageUtils::getStartOffsetOfNodeGroup(nodeGroupIdx); const auto tableID = table->getTableID(); if (source == TableScanSource::UNCOMMITTED) { - nodeGroupStartOffset = transaction->getUncommittedOffset(tableID, nodeGroupStartOffset); + nodeGroupStartOffset = transaction->getUncommittedOffset(*table, nodeGroupStartOffset); } auto startOffsetInGroup = startOffset - nodeGroupStartOffset; const NodeGroupScanResult scanResult = @@ -252,7 +252,7 @@ bool NodeTableScanState::scanNext(Transaction* transaction) { auto nodeGroupStartOffset = StorageUtils::getStartOffsetOfNodeGroup(nodeGroupIdx); const auto tableID = table->getTableID(); if (source == TableScanSource::UNCOMMITTED) { - nodeGroupStartOffset = transaction->getUncommittedOffset(tableID, nodeGroupStartOffset); + nodeGroupStartOffset = transaction->getUncommittedOffset(*table, nodeGroupStartOffset); } for (auto i = 0u; i < scanResult.numRows; i++) { auto& nodeID = nodeIDVector->getValue(i); @@ -298,7 +298,7 @@ NodeTable::NodeTable(const StorageManager* storageManager, row_idx_t NodeTable::getNumTotalRows(const Transaction* transaction) { auto numLocalRows = 0u; if (transaction && transaction->getLocalStorage()) { - if (const auto localTable = transaction->getLocalStorage()->getLocalTable(tableID)) { + if (const auto localTable = transaction->getLocalStorage()->getLocalTable(*this)) { numLocalRows = localTable->getNumTotalRows(); } } @@ -313,7 +313,7 @@ void NodeTable::initScanState(Transaction* transaction, TableScanState& scanStat nodeGroup = nodeGroups->getNodeGroup(nodeScanState.nodeGroupIdx); } break; case TableScanSource::UNCOMMITTED: { - const auto localTable = transaction->getLocalStorage()->getLocalTable(tableID); + const auto localTable = transaction->getLocalStorage()->getLocalTable(*this); // An UNCOMMITTED morsel without a local table means a stale scan shared state // survived from an earlier (committed) write transaction (see // https://github.com/LadybugDB/ladybug/issues/1030). Throw instead of @@ -339,10 +339,10 @@ void NodeTable::initScanState(Transaction* transaction, TableScanState& scanStat void NodeTable::initScanState(Transaction* transaction, TableScanState& scanState, table_id_t tableID, offset_t startOffset) const { - if (transaction->isUnCommitted(tableID, startOffset)) { + if (transaction->isUnCommitted(*this, startOffset)) { scanState.source = TableScanSource::UNCOMMITTED; scanState.nodeGroupIdx = - StorageUtils::getNodeGroupIdx(transaction->getLocalRowIdx(tableID, startOffset)); + StorageUtils::getNodeGroupIdx(transaction->getLocalRowIdx(*this, startOffset)); } else { scanState.source = TableScanSource::COMMITTED; scanState.nodeGroupIdx = StorageUtils::getNodeGroupIdx(startOffset); @@ -364,8 +364,8 @@ bool NodeTable::lookup(const Transaction* transaction, const TableScanState& sca } const auto nodeOffset = scanState.nodeIDVector->readNodeOffset(nodeIDPos); const offset_t rowIdxInGroup = - transaction->isUnCommitted(tableID, nodeOffset) ? - transaction->getLocalRowIdx(tableID, nodeOffset) - + transaction->isUnCommitted(*this, nodeOffset) ? + transaction->getLocalRowIdx(*this, nodeOffset) - StorageUtils::getStartOffsetOfNodeGroup(scanState.nodeGroupIdx) : nodeOffset - StorageUtils::getStartOffsetOfNodeGroup(scanState.nodeGroupIdx); scanState.rowIdxVector->setValue(nodeIDPos, rowIdxInGroup); @@ -391,15 +391,15 @@ bool NodeTable::lookupMultiple(Transaction* transaction, TableScanState& scanSta continue; } const auto nodeOffset = scanState.nodeIDVector->readNodeOffset(nodeIDPos); - const auto isUnCommitted = transaction->isUnCommitted(tableID, nodeOffset); + const auto isUnCommitted = transaction->isUnCommitted(*this, nodeOffset); const auto source = isUnCommitted ? TableScanSource::UNCOMMITTED : TableScanSource::COMMITTED; const auto nodeGroupIdx = isUnCommitted ? - StorageUtils::getNodeGroupIdx(transaction->getLocalRowIdx(tableID, nodeOffset)) : + StorageUtils::getNodeGroupIdx(transaction->getLocalRowIdx(*this, nodeOffset)) : StorageUtils::getNodeGroupIdx(nodeOffset); const offset_t rowIdxInGroup = - isUnCommitted ? transaction->getLocalRowIdx(tableID, nodeOffset) - + isUnCommitted ? transaction->getLocalRowIdx(*this, nodeOffset) - StorageUtils::getStartOffsetOfNodeGroup(nodeGroupIdx) : nodeOffset - StorageUtils::getStartOffsetOfNodeGroup(nodeGroupIdx); if (scanState.source == source && scanState.nodeGroupIdx == nodeGroupIdx) { @@ -434,7 +434,7 @@ offset_t NodeTable::validateUniquenessConstraint(const Transaction* transaction, lookupPK(transaction, propertyVectors[pkColumnID], pkVectorPos, offset)) { return offset; } - if (const auto localTable = transaction->getLocalStorage()->getLocalTable(tableID)) { + if (const auto localTable = transaction->getLocalStorage()->getLocalTable(*this)) { return localTable->cast().validateUniquenessConstraint(transaction, *pkVector); } @@ -503,7 +503,7 @@ void NodeTable::insert(Transaction* transaction, TableInsertState& insertState) if (insertState.logToWAL && transaction->shouldLogToWAL()) { DASSERT(transaction->isWriteTransaction()); auto& wal = transaction->getLocalWAL(); - wal.logTableInsertion(tableID, TableType::NODE, + wal.logTableInsertion(getOwnerCatalogName(), tableID, TableType::NODE, nodeInsertState.nodeIDVector.state->getSelVector().getSelSize(), insertState.propertyVectors); } @@ -563,8 +563,8 @@ void NodeTable::update(Transaction* transaction, TableUpdateState& updateState) } // Indexes that re-scan the node table to observe the NEW value during update (e.g. HNSW) // must be updated after the row is written. - if (transaction->isUnCommitted(tableID, nodeOffset)) { - const auto localTable = transaction->getLocalStorage()->getLocalTable(tableID); + if (transaction->isUnCommitted(*this, nodeOffset)) { + const auto localTable = transaction->getLocalStorage()->getLocalTable(*this); DASSERT(localTable); localTable->update(&DUMMY_TRANSACTION, updateState); } else { @@ -590,7 +590,7 @@ void NodeTable::update(Transaction* transaction, TableUpdateState& updateState) if (updateState.logToWAL && transaction->shouldLogToWAL()) { DASSERT(transaction->isWriteTransaction()); auto& wal = transaction->getLocalWAL(); - wal.logNodeUpdate(tableID, nodeUpdateState.columnID, nodeOffset, + wal.logNodeUpdate(getOwnerCatalogName(), tableID, nodeUpdateState.columnID, nodeOffset, &nodeUpdateState.propertyVector); } setHasChanges(); @@ -616,8 +616,8 @@ bool NodeTable::delete_(Transaction* transaction, TableDeleteState& deleteState) index.getIndex()->delete_(transaction, nodeDeleteState.nodeIDVector, *indexDeleteState); } - if (transaction->isUnCommitted(tableID, nodeOffset)) { - const auto localTable = transaction->getLocalStorage()->getLocalTable(tableID); + if (transaction->isUnCommitted(*this, nodeOffset)) { + const auto localTable = transaction->getLocalStorage()->getLocalTable(*this); isDeleted = localTable->delete_(&DUMMY_TRANSACTION, deleteState); } else { const auto nodeGroupIdx = StorageUtils::getNodeGroupIdx(nodeOffset); @@ -633,7 +633,8 @@ bool NodeTable::delete_(Transaction* transaction, TableDeleteState& deleteState) if (deleteState.logToWAL && transaction->shouldLogToWAL()) { DASSERT(transaction->isWriteTransaction()); auto& wal = transaction->getLocalWAL(); - wal.logNodeDeletion(tableID, nodeOffset, &nodeDeleteState.pkVector); + wal.logNodeDeletion(getOwnerCatalogName(), tableID, nodeOffset, + &nodeDeleteState.pkVector); } } return isDeleted; @@ -646,7 +647,7 @@ void NodeTable::addColumn(Transaction* transaction, TableAddColumnState& addColu pageAllocator.getDataFH(), memoryManager, shadowFile, enableCompression)); LocalTable* localTable = nullptr; if (transaction->getLocalStorage()) { - localTable = transaction->getLocalStorage()->getLocalTable(tableID); + localTable = transaction->getLocalStorage()->getLocalTable(*this); } if (localTable) { localTable->addColumn(addColumnState); @@ -692,7 +693,7 @@ void NodeTable::commit(main::ClientContext* context, TableCatalogEntry* tableEnt // 2. Set deleted flag for tuples that are deleted in local storage. row_idx_t numLocalRows = 0u; for (auto localNodeGroupIdx = 0u; localNodeGroupIdx < localNodeTable.getNumNodeGroups(); - localNodeGroupIdx++) { + localNodeGroupIdx++) { const auto localNodeGroup = localNodeTable.getNodeGroup(localNodeGroupIdx); if (localNodeGroup->hasDeletions(transaction)) { // TODO(Guodong): Assume local storage is small here. Should optimize the loop away by @@ -863,7 +864,7 @@ void NodeTable::reclaimStorage(PageAllocator& pageAllocator) const { TableStats NodeTable::getStats(const Transaction* transaction) const { auto stats = nodeGroups->getStats(); - if (const auto localTable = transaction->getLocalStorage()->getLocalTable(tableID)) { + if (const auto localTable = transaction->getLocalStorage()->getLocalTable(*this)) { const auto localStats = localTable->cast().getStats(); stats.merge(localStats); } @@ -871,8 +872,8 @@ TableStats NodeTable::getStats(const Transaction* transaction) const { } bool NodeTable::isVisible(const Transaction* transaction, offset_t offset) const { - if (transaction && transaction->isUnCommitted(tableID, offset)) { - const auto localTable = transaction->getLocalStorage()->getLocalTable(tableID); + if (transaction && transaction->isUnCommitted(*this, offset)) { + const auto localTable = transaction->getLocalStorage()->getLocalTable(*this); DASSERT(localTable); return localTable->cast().isVisible(transaction, offset); } @@ -889,8 +890,8 @@ bool NodeTable::isVisibleNoLock(const Transaction* transaction, offset_t offset) throw RuntimeException( "Index contains an invalid node offset. Please drop and rebuild the index."); } - if (transaction && transaction->isUnCommitted(tableID, offset)) { - const auto localTable = transaction->getLocalStorage()->getLocalTable(tableID); + if (transaction && transaction->isUnCommitted(*this, offset)) { + const auto localTable = transaction->getLocalStorage()->getLocalTable(*this); DASSERT(localTable); return localTable->cast().isVisible(transaction, offset); } @@ -947,7 +948,7 @@ Index* NodeTable::tryGetPrimaryKeyIndex() const { bool NodeTable::lookupPK(const Transaction* transaction, ValueVector* keyVector, uint64_t vectorPos, offset_t& result) const { if (transaction->getLocalStorage()) { - if (const auto localTable = transaction->getLocalStorage()->getLocalTable(tableID); + if (const auto localTable = transaction->getLocalStorage()->getLocalTable(*this); localTable && localTable->cast().lookupPK(transaction, keyVector, vectorPos, result)) { return true; @@ -1059,7 +1060,7 @@ void NodeTable::scanIndexColumns(main::ClientContext* context, IndexScanHelper& progressBar->updateProgress(queryID.value(), numNodeGroups == 0 ? 1.0 : 0.01); } for (node_group_idx_t nodeGroupToScan = 0u; nodeGroupToScan < numNodeGroups; - ++nodeGroupToScan) { + ++nodeGroupToScan) { scanState->nodeGroup = nodeGroups_.getNodeGroupNoLock(nodeGroupToScan); // It is possible for the node group to have no chunked groups if we are rolling back due to diff --git a/src/storage/table/rel_table.cpp b/src/storage/table/rel_table.cpp index bbf86e6dba..9252c64540 100644 --- a/src/storage/table/rel_table.cpp +++ b/src/storage/table/rel_table.cpp @@ -68,8 +68,7 @@ void RelTableScanState::setToTable(const Transaction* transaction, Table* table_ csrOffsetColumn = table->cast().getCSROffsetColumn(direction); csrLengthColumn = table->cast().getCSRLengthColumn(direction); nodeGroupIdx = INVALID_NODE_GROUP_IDX; - if (const auto localRelTable = - transaction->getLocalStorage()->getLocalTable(table->getTableID())) { + if (const auto localRelTable = transaction->getLocalStorage()->getLocalTable(*table)) { auto localTableColumnIDs = LocalRelTable::rewriteLocalColumnIDs(direction, columnIDs); localTableScanState = std::make_unique(*this, localRelTable->ptrCast(), localTableColumnIDs); @@ -344,7 +343,7 @@ void RelTable::insert(Transaction* transaction, TableInsertState& insertState) { relInsertState.propertyVectors.end()); DASSERT(relInsertState.srcNodeIDVector.state->getSelVector().getSelSize() == 1); auto& wal = transaction->getLocalWAL(); - wal.logTableInsertion(tableID, TableType::REL, + wal.logTableInsertion(getOwnerCatalogName(), tableID, TableType::REL, relInsertState.srcNodeIDVector.state->getSelVector().getSelSize(), vectorsToLog); } setHasChanges(); @@ -356,7 +355,7 @@ void RelTable::update(Transaction* transaction, TableUpdateState& updateState) { const auto relIDPos = relUpdateState.relIDVector.state->getSelVector()[0]; if (const auto relOffset = relUpdateState.relIDVector.readNodeOffset(relIDPos); relOffset >= StorageConstants::MAX_NUM_ROWS_IN_TABLE) { - const auto localTable = transaction->getLocalStorage()->getLocalTable(tableID); + const auto localTable = transaction->getLocalStorage()->getLocalTable(*this); DASSERT(localTable); localTable->update(&DUMMY_TRANSACTION, updateState); } else { @@ -369,9 +368,9 @@ void RelTable::update(Transaction* transaction, TableUpdateState& updateState) { if (updateState.logToWAL && transaction->shouldLogToWAL()) { DASSERT(transaction->isWriteTransaction()); auto& wal = transaction->getLocalWAL(); - wal.logRelUpdate(tableID, relUpdateState.columnID, &relUpdateState.srcNodeIDVector, - &relUpdateState.dstNodeIDVector, &relUpdateState.relIDVector, - &relUpdateState.propertyVector); + wal.logRelUpdate(getOwnerCatalogName(), tableID, relUpdateState.columnID, + &relUpdateState.srcNodeIDVector, &relUpdateState.dstNodeIDVector, + &relUpdateState.relIDVector, &relUpdateState.propertyVector); } setHasChanges(); } @@ -383,7 +382,7 @@ bool RelTable::delete_(Transaction* transaction, TableDeleteState& deleteState) bool isDeleted = false; if (const auto relOffset = relDeleteState.relIDVector.readNodeOffset(relIDPos); relOffset >= StorageConstants::MAX_NUM_ROWS_IN_TABLE) { - const auto localTable = transaction->getLocalStorage()->getLocalTable(tableID); + const auto localTable = transaction->getLocalStorage()->getLocalTable(*this); DASSERT(localTable); isDeleted = localTable->delete_(transaction, deleteState); } else { @@ -401,7 +400,7 @@ bool RelTable::delete_(Transaction* transaction, TableDeleteState& deleteState) if (deleteState.logToWAL && transaction->shouldLogToWAL()) { DASSERT(transaction->isWriteTransaction()); auto& wal = transaction->getLocalWAL(); - wal.logRelDelete(tableID, &relDeleteState.srcNodeIDVector, + wal.logRelDelete(getOwnerCatalogName(), tableID, &relDeleteState.srcNodeIDVector, &relDeleteState.dstNodeIDVector, &relDeleteState.relIDVector); } } @@ -433,7 +432,8 @@ void RelTable::detachDelete(Transaction* transaction, RelTableDeleteState* delet if (deleteState->logToWAL && transaction->shouldLogToWAL()) { DASSERT(transaction->isWriteTransaction()); auto& wal = transaction->getLocalWAL(); - wal.logRelDetachDelete(tableID, direction, &deleteState->srcNodeIDVector); + wal.logRelDetachDelete(getOwnerCatalogName(), tableID, direction, + &deleteState->srcNodeIDVector); } setHasChanges(); } @@ -466,7 +466,7 @@ void RelTable::detachDeleteBatch(Transaction* transaction, ValueVector& srcNodeI directedRelData.size() == NUM_REL_DIRECTIONS ? getDirectedTableData(RelDirectionUtils::getOppositeDirection(direction)) : nullptr; - const auto localTable = transaction->getLocalStorage()->getLocalTable(tableID); + const auto localTable = transaction->getLocalStorage()->getLocalTable(*this); for (const auto srcNodeID : srcNodeIDs) { const auto srcState = std::make_shared(); @@ -515,7 +515,8 @@ void RelTable::detachDeleteBatch(Transaction* transaction, ValueVector& srcNodeI } if (transaction->shouldLogToWAL()) { DASSERT(transaction->isWriteTransaction()); - transaction->getLocalWAL().logRelDetachDelete(tableID, direction, &srcNodeIDVector); + transaction->getLocalWAL().logRelDetachDelete(getOwnerCatalogName(), tableID, direction, + &srcNodeIDVector); } setHasChanges(); } @@ -531,7 +532,7 @@ std::vector RelTable::getStorageDirections() const { bool RelTable::checkIfNodeHasRels(Transaction* transaction, RelDataDirection direction, ValueVector* srcNodeIDVector) const { bool hasRels = false; - const auto localTable = transaction->getLocalStorage()->getLocalTable(tableID); + const auto localTable = transaction->getLocalStorage()->getLocalTable(*this); if (localTable) { hasRels = localTable->cast().checkIfNodeHasRels(srcNodeIDVector, direction); } @@ -552,7 +553,7 @@ void RelTable::throwIfNodeHasRels(Transaction* transaction, RelDataDirection dir void RelTable::detachDeleteForCSRRels(Transaction* transaction, RelTableData* tableData, RelTableData* reverseTableData, RelTableScanState* relDataReadState, RelTableDeleteState* deleteState) { - const auto localTable = transaction->getLocalStorage()->getLocalTable(tableID); + const auto localTable = transaction->getLocalStorage()->getLocalTable(*this); const auto tempState = deleteState->dstNodeIDVector.state.get(); while (scan(transaction, *relDataReadState)) { const auto numRelsScanned = tempState->getSelVector().getSelSize(); @@ -593,7 +594,7 @@ void RelTable::addColumn(Transaction* transaction, TableAddColumnState& addColum PageAllocator& pageAllocator) { LocalTable* localTable = nullptr; if (transaction->getLocalStorage()) { - localTable = transaction->getLocalStorage()->getLocalTable(tableID); + localTable = transaction->getLocalStorage()->getLocalTable(*this); } if (localTable) { localTable->addColumn(addColumnState); @@ -742,7 +743,7 @@ bool RelTable::checkpoint(main::ClientContext*, TableCatalogEntry* tableEntry, row_idx_t RelTable::getNumTotalRows(const Transaction* transaction) { auto numLocalRows = 0u; - if (auto localTable = transaction->getLocalStorage()->getLocalTable(tableID)) { + if (auto localTable = transaction->getLocalStorage()->getLocalTable(*this)) { numLocalRows = localTable->getNumTotalRows(); } return numLocalRows + nextRelOffset; @@ -755,7 +756,7 @@ std::vector> RelTable::getDegreeEntries( auto* relTableData = getDirectedTableData(direction); auto* csrLengthColumn = relTableData->getCSRLengthColumn(); for (node_group_idx_t nodeGroupIdx = 0; nodeGroupIdx < relTableData->getNumNodeGroups(); - nodeGroupIdx++) { + nodeGroupIdx++) { auto* nodeGroup = relTableData->getNodeGroup(nodeGroupIdx); if (!nodeGroup) { continue; @@ -790,7 +791,7 @@ std::vector> RelTable::getDegreeEntries( } } if (transaction->isWriteTransaction()) { - if (auto* localTable = transaction->getLocalStorage()->getLocalTable(tableID)) { + if (auto* localTable = transaction->getLocalStorage()->getLocalTable(*this)) { auto& localRelTable = localTable->cast(); for (const auto& [nodeOffset, rowIndices] : localRelTable.getCSRIndex(direction)) { degrees[nodeOffset] += rowIndices.size(); @@ -867,7 +868,7 @@ row_idx_t RelTable::getDegreeForOffset(const Transaction* transaction, RelDataDi count += csrIndex->getNumRows(offsetInGroup); } if (transaction->isWriteTransaction()) { - if (auto* localTable = transaction->getLocalStorage()->getLocalTable(tableID)) { + if (auto* localTable = transaction->getLocalStorage()->getLocalTable(*this)) { auto& localCSRIndex = localTable->cast().getCSRIndex(direction); if (auto it = localCSRIndex.find(nodeOffset); it != localCSRIndex.end()) { count += it->second.size(); diff --git a/src/storage/table/table.cpp b/src/storage/table/table.cpp index 238dcd96e2..3b70edc527 100644 --- a/src/storage/table/table.cpp +++ b/src/storage/table/table.cpp @@ -42,8 +42,9 @@ TableDeleteState::~TableDeleteState() = default; Table::Table(const catalog::TableCatalogEntry* tableEntry, const StorageManager* storageManager, MemoryManager* memoryManager) : tableType{tableEntry->getTableType()}, tableID{tableEntry->getTableID()}, - tableName{tableEntry->getName()}, enableCompression{storageManager->compressionEnabled()}, - memoryManager{memoryManager}, storageManager{const_cast(storageManager)}, + tableName{tableEntry->getName()}, ownerCatalogName{tableEntry->getOwningCatalogName()}, + enableCompression{storageManager->compressionEnabled()}, memoryManager{memoryManager}, + storageManager{const_cast(storageManager)}, shadowFile{&storageManager->getShadowFile()}, changeEpoch{0} {} Table::~Table() = default; diff --git a/src/storage/wal/local_wal.cpp b/src/storage/wal/local_wal.cpp index a313b4715b..65534dfeae 100644 --- a/src/storage/wal/local_wal.cpp +++ b/src/storage/wal/local_wal.cpp @@ -27,70 +27,88 @@ void LocalWAL::logCommit() { addNewWALRecordNoLock(walRecord); } -void LocalWAL::logCreateCatalogEntryRecord(CatalogEntry* catalogEntry, bool isInternal) { +void LocalWAL::logCreateCatalogEntryRecord(const std::string& ownerCatalogName, + CatalogEntry* catalogEntry, bool isInternal) { CreateCatalogEntryRecord walRecord(catalogEntry, isInternal); + walRecord.ownerCatalogName = ownerCatalogName; addNewWALRecord(walRecord); } -void LocalWAL::logCreateIndexRecord(CatalogEntry* catalogEntry, IndexInfo indexInfo, - std::vector treeBytes) { +void LocalWAL::logCreateIndexRecord(const std::string& ownerCatalogName, CatalogEntry* catalogEntry, + IndexInfo indexInfo, std::vector treeBytes) { CreateIndexRecord walRecord(catalogEntry, std::move(indexInfo), std::move(treeBytes)); + walRecord.ownerCatalogName = ownerCatalogName; addNewWALRecord(walRecord); } -void LocalWAL::logDropCatalogEntryRecord(table_id_t tableID, CatalogEntryType type) { +void LocalWAL::logDropCatalogEntryRecord(const std::string& ownerCatalogName, table_id_t tableID, + CatalogEntryType type) { DropCatalogEntryRecord walRecord(tableID, type); + walRecord.ownerCatalogName = ownerCatalogName; addNewWALRecord(walRecord); } -void LocalWAL::logAlterCatalogEntryRecord(const BoundAlterInfo* alterInfo) { +void LocalWAL::logAlterCatalogEntryRecord(const std::string& ownerCatalogName, + const BoundAlterInfo* alterInfo) { AlterTableEntryRecord walRecord(alterInfo); + walRecord.ownerCatalogName = ownerCatalogName; addNewWALRecord(walRecord); } -void LocalWAL::logTableInsertion(table_id_t tableID, TableType tableType, row_idx_t numRows, - const std::vector& vectors) { +void LocalWAL::logTableInsertion(const std::string& ownerCatalogName, table_id_t tableID, + TableType tableType, row_idx_t numRows, const std::vector& vectors) { TableInsertionRecord walRecord(tableID, tableType, numRows, vectors); + walRecord.ownerCatalogName = ownerCatalogName; addNewWALRecord(walRecord); } -void LocalWAL::logNodeDeletion(table_id_t tableID, offset_t nodeOffset, ValueVector* pkVector) { +void LocalWAL::logNodeDeletion(const std::string& ownerCatalogName, table_id_t tableID, + offset_t nodeOffset, ValueVector* pkVector) { NodeDeletionRecord walRecord(tableID, nodeOffset, pkVector); + walRecord.ownerCatalogName = ownerCatalogName; addNewWALRecord(walRecord); } -void LocalWAL::logNodeUpdate(table_id_t tableID, column_id_t columnID, offset_t nodeOffset, - ValueVector* propertyVector) { +void LocalWAL::logNodeUpdate(const std::string& ownerCatalogName, table_id_t tableID, + column_id_t columnID, offset_t nodeOffset, ValueVector* propertyVector) { NodeUpdateRecord walRecord(tableID, columnID, nodeOffset, propertyVector); + walRecord.ownerCatalogName = ownerCatalogName; addNewWALRecord(walRecord); } -void LocalWAL::logRelDelete(table_id_t tableID, ValueVector* srcNodeVector, - ValueVector* dstNodeVector, ValueVector* relIDVector) { +void LocalWAL::logRelDelete(const std::string& ownerCatalogName, table_id_t tableID, + ValueVector* srcNodeVector, ValueVector* dstNodeVector, ValueVector* relIDVector) { RelDeletionRecord walRecord(tableID, srcNodeVector, dstNodeVector, relIDVector); + walRecord.ownerCatalogName = ownerCatalogName; addNewWALRecord(walRecord); } -void LocalWAL::logRelDetachDelete(table_id_t tableID, RelDataDirection direction, - ValueVector* srcNodeVector) { +void LocalWAL::logRelDetachDelete(const std::string& ownerCatalogName, table_id_t tableID, + RelDataDirection direction, ValueVector* srcNodeVector) { RelDetachDeleteRecord walRecord(tableID, direction, srcNodeVector); + walRecord.ownerCatalogName = ownerCatalogName; addNewWALRecord(walRecord); } -void LocalWAL::logRelUpdate(table_id_t tableID, column_id_t columnID, ValueVector* srcNodeVector, - ValueVector* dstNodeVector, ValueVector* relIDVector, ValueVector* propertyVector) { +void LocalWAL::logRelUpdate(const std::string& ownerCatalogName, table_id_t tableID, + column_id_t columnID, ValueVector* srcNodeVector, ValueVector* dstNodeVector, + ValueVector* relIDVector, ValueVector* propertyVector) { RelUpdateRecord walRecord(tableID, columnID, srcNodeVector, dstNodeVector, relIDVector, propertyVector); + walRecord.ownerCatalogName = ownerCatalogName; addNewWALRecord(walRecord); } -void LocalWAL::logUpdateSequenceRecord(sequence_id_t sequenceID, uint64_t kCount) { +void LocalWAL::logUpdateSequenceRecord(const std::string& ownerCatalogName, + sequence_id_t sequenceID, uint64_t kCount) { UpdateSequenceRecord walRecord(sequenceID, kCount); + walRecord.ownerCatalogName = ownerCatalogName; addNewWALRecord(walRecord); } -void LocalWAL::logLoadExtension(std::string path) { +void LocalWAL::logLoadExtension(const std::string& ownerCatalogName, std::string path) { LoadExtensionRecord walRecord(std::move(path)); + walRecord.ownerCatalogName = ownerCatalogName; addNewWALRecord(walRecord); } diff --git a/src/storage/wal/records/create_catalog_entry_record_replay.cpp b/src/storage/wal/records/create_catalog_entry_record_replay.cpp index 1f25b80947..77e82ff708 100644 --- a/src/storage/wal/records/create_catalog_entry_record_replay.cpp +++ b/src/storage/wal/records/create_catalog_entry_record_replay.cpp @@ -1,6 +1,8 @@ +#include "binder/ddl/bound_create_table_info.h" #include "catalog/catalog.h" #include "catalog/catalog_entry/index_catalog_entry.h" #include "catalog/catalog_entry/node_table_catalog_entry.h" +#include "catalog/catalog_entry/rel_group_catalog_entry.h" #include "catalog/catalog_entry/scalar_macro_catalog_entry.h" #include "catalog/catalog_entry/sequence_catalog_entry.h" #include "catalog/catalog_entry/table_catalog_entry.h" @@ -53,9 +55,33 @@ void WALReplayer::replayCreateCatalogEntryRecord(WALRecord& walRecord) const { case CatalogEntryType::NODE_TABLE_ENTRY: case CatalogEntryType::REL_GROUP_ENTRY: { auto& entry = record.ownedCatalogEntry->constCast(); - auto newEntry = catalog->createTableEntry(transaction, - entry.getBoundCreateTableInfo(transaction, record.isInternal)); - storageManager->createTable(newEntry->ptrCast(), &clientContext); + auto boundInfo = entry.getBoundCreateTableInfo(transaction, record.isInternal); + const auto isRelGroup = entry.getType() == CatalogEntryType::REL_GROUP_ENTRY; + if (isRelGroup) { + for (auto& relTableInfo : + boundInfo.extraInfo->ptrCast() + ->relTableInfos) { + relTableInfo.nodePair.srcTableID = getReplayedEntryID( + CatalogEntryType::NODE_TABLE_ENTRY, relTableInfo.nodePair.srcTableID); + relTableInfo.nodePair.dstTableID = getReplayedEntryID( + CatalogEntryType::NODE_TABLE_ENTRY, relTableInfo.nodePair.dstTableID); + } + } + auto newEntry = catalog->createTableEntry(transaction, boundInfo); + auto* newTableEntry = newEntry->ptrCast(); + recordReplayedEntryID(entry.getType(), entry.getTableID(), newTableEntry->getTableID()); + if (isRelGroup) { + const auto& recordedRelEntryInfos = + entry.constCast().getRelEntryInfos(); + const auto& replayedRelEntryInfos = + newTableEntry->constCast().getRelEntryInfos(); + DASSERT(recordedRelEntryInfos.size() == replayedRelEntryInfos.size()); + for (auto i = 0u; i < recordedRelEntryInfos.size(); i++) { + recordReplayedEntryID(entry.getType(), recordedRelEntryInfos[i].oid, + replayedRelEntryInfos[i].oid); + } + } + storageManager->createTable(newTableEntry, &clientContext); } break; case CatalogEntryType::SCALAR_MACRO_ENTRY: { auto& macroEntry = record.ownedCatalogEntry->constCast(); @@ -64,17 +90,24 @@ void WALReplayer::replayCreateCatalogEntryRecord(WALRecord& walRecord) const { } break; case CatalogEntryType::SEQUENCE_ENTRY: { auto& sequenceEntry = record.ownedCatalogEntry->constCast(); - catalog->createSequence(transaction, + const auto replayedSequenceID = catalog->createSequence(transaction, sequenceEntry.getBoundCreateSequenceInfo(record.isInternal)); + recordReplayedEntryID(CatalogEntryType::SEQUENCE_ENTRY, sequenceEntry.getOID(), + replayedSequenceID); } break; case CatalogEntryType::TYPE_ENTRY: { auto& typeEntry = record.ownedCatalogEntry->constCast(); catalog->createType(transaction, typeEntry.getName(), typeEntry.getLogicalType().copy()); } break; case CatalogEntryType::INDEX_ENTRY: { - auto& indexEntry = record.ownedCatalogEntry->constCast(); - auto indexEntryCopy = indexEntry.copy(); - catalog->createIndex(transaction, std::move(record.ownedCatalogEntry)); + auto* indexEntry = record.ownedCatalogEntry->ptrCast(); + indexEntry->setTableID( + getReplayedEntryID(CatalogEntryType::NODE_TABLE_ENTRY, indexEntry->getTableID())); + auto indexEntryCopy = indexEntry->copy(); + const auto recordedIndexID = indexEntry->getOID(); + const auto replayedIndexID = + catalog->createIndex(transaction, std::move(record.ownedCatalogEntry)); + recordReplayedEntryID(CatalogEntryType::INDEX_ENTRY, recordedIndexID, replayedIndexID); rebuildArtIndexFromCatalog(clientContext, storageManager, catalog, transaction, *indexEntryCopy); } break; diff --git a/src/storage/wal/records/create_index_record_replay.cpp b/src/storage/wal/records/create_index_record_replay.cpp index 34e9eb4064..501d20e268 100644 --- a/src/storage/wal/records/create_index_record_replay.cpp +++ b/src/storage/wal/records/create_index_record_replay.cpp @@ -19,11 +19,22 @@ void WALReplayer::replayCreateIndexRecord(WALRecord& walRecord) const { auto* catalog = Catalog::Get(clientContext); auto* trx = transaction::Transaction::Get(clientContext); auto* storageManager = StorageManager::Get(clientContext); - auto& indexCatalogEntry = record.ownedCatalogEntry->constCast(); - if (!catalog->containsIndex(trx, indexCatalogEntry.getTableID(), - indexCatalogEntry.getIndexName())) { - catalog->createIndex(trx, std::move(record.ownedCatalogEntry)); + auto* indexCatalogEntry = record.ownedCatalogEntry->ptrCast(); + const auto recordedIndexID = indexCatalogEntry->getOID(); + indexCatalogEntry->setTableID( + getReplayedEntryID(CatalogEntryType::NODE_TABLE_ENTRY, indexCatalogEntry->getTableID())); + if (!catalog->containsIndex(trx, indexCatalogEntry->getTableID(), + indexCatalogEntry->getIndexName())) { + const auto replayedIndexID = catalog->createIndex(trx, std::move(record.ownedCatalogEntry)); + recordReplayedEntryID(CatalogEntryType::INDEX_ENTRY, recordedIndexID, replayedIndexID); + } else { + auto* existingIndex = catalog->getIndex(trx, indexCatalogEntry->getTableID(), + indexCatalogEntry->getIndexName()); + recordReplayedEntryID(CatalogEntryType::INDEX_ENTRY, recordedIndexID, + existingIndex->getOID()); } + record.indexInfo->tableID = + getReplayedEntryID(CatalogEntryType::NODE_TABLE_ENTRY, record.indexInfo->tableID); auto* table = storage::PartitionStorageRegistry::resolveNodeTableByID(&clientContext, record.indexInfo->tableID) ->ptrCast(); diff --git a/src/storage/wal/records/drop_catalog_entry_record_replay.cpp b/src/storage/wal/records/drop_catalog_entry_record_replay.cpp index 107f954a6e..0686b7615d 100644 --- a/src/storage/wal/records/drop_catalog_entry_record_replay.cpp +++ b/src/storage/wal/records/drop_catalog_entry_record_replay.cpp @@ -14,7 +14,7 @@ void WALReplayer::replayDropCatalogEntryRecord(const WALRecord& walRecord) const auto& dropEntryRecord = walRecord.constCast(); auto catalog = Catalog::Get(clientContext); auto transaction = transaction::Transaction::Get(clientContext); - const auto entryID = dropEntryRecord.entryID; + const auto entryID = getReplayedEntryID(dropEntryRecord.entryType, dropEntryRecord.entryID); switch (dropEntryRecord.entryType) { case CatalogEntryType::NODE_TABLE_ENTRY: case CatalogEntryType::REL_GROUP_ENTRY: { diff --git a/src/storage/wal/records/node_deletion_record_replay.cpp b/src/storage/wal/records/node_deletion_record_replay.cpp index 494b3ad255..732234050f 100644 --- a/src/storage/wal/records/node_deletion_record_replay.cpp +++ b/src/storage/wal/records/node_deletion_record_replay.cpp @@ -1,3 +1,4 @@ +#include "catalog/catalog_entry/catalog_entry_type.h" #include "storage/partition_storage_registry.h" #include "storage/storage_manager.h" #include "storage/table/node_table.h" @@ -11,15 +12,15 @@ namespace storage { void WALReplayer::replayNodeDeletionRecord(const WALRecord& walRecord) const { const auto& deletionRecord = walRecord.constCast(); - const auto tableID = deletionRecord.tableID; + const auto tableID = + getReplayedEntryID(catalog::CatalogEntryType::NODE_TABLE_ENTRY, deletionRecord.tableID); auto& table = storage::PartitionStorageRegistry::resolveNodeTableByID(&clientContext, tableID) ->cast(); const auto anchorState = deletionRecord.ownedPKVector->state; DASSERT(anchorState->getSelVector().getSelSize() == 1); const auto nodeIDVector = std::make_unique(LogicalType::INTERNAL_ID()); nodeIDVector->setState(anchorState); - nodeIDVector->setValue(0, - internalID_t{deletionRecord.nodeOffset, deletionRecord.tableID}); + nodeIDVector->setValue(0, internalID_t{deletionRecord.nodeOffset, tableID}); const auto deleteState = std::make_unique(*nodeIDVector, *deletionRecord.ownedPKVector); DASSERT(transaction::Transaction::Get(clientContext) && diff --git a/src/storage/wal/records/node_update_record_replay.cpp b/src/storage/wal/records/node_update_record_replay.cpp index 9582bb62f3..17fbe63d59 100644 --- a/src/storage/wal/records/node_update_record_replay.cpp +++ b/src/storage/wal/records/node_update_record_replay.cpp @@ -1,3 +1,4 @@ +#include "catalog/catalog_entry/catalog_entry_type.h" #include "storage/partition_storage_registry.h" #include "storage/storage_manager.h" #include "storage/table/node_table.h" @@ -11,15 +12,15 @@ namespace storage { void WALReplayer::replayNodeUpdateRecord(const WALRecord& walRecord) const { const auto& updateRecord = walRecord.constCast(); - const auto tableID = updateRecord.tableID; + const auto tableID = + getReplayedEntryID(catalog::CatalogEntryType::NODE_TABLE_ENTRY, updateRecord.tableID); auto& table = storage::PartitionStorageRegistry::resolveNodeTableByID(&clientContext, tableID) ->cast(); const auto anchorState = updateRecord.ownedPropertyVector->state; DASSERT(anchorState->getSelVector().getSelSize() == 1); const auto nodeIDVector = std::make_unique(LogicalType::INTERNAL_ID()); nodeIDVector->setState(anchorState); - nodeIDVector->setValue(0, - internalID_t{updateRecord.nodeOffset, updateRecord.tableID}); + nodeIDVector->setValue(0, internalID_t{updateRecord.nodeOffset, tableID}); const auto updateState = std::make_unique(updateRecord.columnID, *nodeIDVector, *updateRecord.ownedPropertyVector); DASSERT(transaction::Transaction::Get(clientContext) && diff --git a/src/storage/wal/records/rel_deletion_record_replay.cpp b/src/storage/wal/records/rel_deletion_record_replay.cpp index aa91a5a6e7..a5b494567e 100644 --- a/src/storage/wal/records/rel_deletion_record_replay.cpp +++ b/src/storage/wal/records/rel_deletion_record_replay.cpp @@ -1,3 +1,4 @@ +#include "catalog/catalog_entry/catalog_entry_type.h" #include "storage/storage_manager.h" #include "storage/table/rel_table.h" #include "storage/wal/wal_replayer.h" @@ -10,7 +11,8 @@ namespace storage { void WALReplayer::replayRelDeletionRecord(const WALRecord& walRecord) const { const auto& deletionRecord = walRecord.constCast(); - const auto tableID = deletionRecord.tableID; + const auto tableID = + getReplayedEntryID(catalog::CatalogEntryType::REL_GROUP_ENTRY, deletionRecord.tableID); auto& table = StorageManager::Get(clientContext)->getTable(tableID)->cast(); const auto anchorState = deletionRecord.ownedRelIDVector->state; DASSERT(anchorState->getSelVector().getSelSize() == 1); diff --git a/src/storage/wal/records/rel_detach_delete_record_replay.cpp b/src/storage/wal/records/rel_detach_delete_record_replay.cpp index 9e55c92c6b..cbc1d0c977 100644 --- a/src/storage/wal/records/rel_detach_delete_record_replay.cpp +++ b/src/storage/wal/records/rel_detach_delete_record_replay.cpp @@ -1,3 +1,4 @@ +#include "catalog/catalog_entry/catalog_entry_type.h" #include "storage/storage_manager.h" #include "storage/table/rel_table.h" #include "storage/wal/wal_replayer.h" @@ -10,7 +11,8 @@ namespace storage { void WALReplayer::replayRelDetachDeletionRecord(const WALRecord& walRecord) const { const auto& deletionRecord = walRecord.constCast(); - const auto tableID = deletionRecord.tableID; + const auto tableID = + getReplayedEntryID(catalog::CatalogEntryType::REL_GROUP_ENTRY, deletionRecord.tableID); auto& table = StorageManager::Get(clientContext)->getTable(tableID)->cast(); DASSERT(transaction::Transaction::Get(clientContext) && transaction::Transaction::Get(clientContext)->isRecovery()); diff --git a/src/storage/wal/records/rel_update_record_replay.cpp b/src/storage/wal/records/rel_update_record_replay.cpp index fb91bea565..b3e9c920d0 100644 --- a/src/storage/wal/records/rel_update_record_replay.cpp +++ b/src/storage/wal/records/rel_update_record_replay.cpp @@ -1,3 +1,4 @@ +#include "catalog/catalog_entry/catalog_entry_type.h" #include "storage/storage_manager.h" #include "storage/table/rel_table.h" #include "storage/wal/wal_replayer.h" @@ -10,7 +11,8 @@ namespace storage { void WALReplayer::replayRelUpdateRecord(const WALRecord& walRecord) const { const auto& updateRecord = walRecord.constCast(); - const auto tableID = updateRecord.tableID; + const auto tableID = + getReplayedEntryID(catalog::CatalogEntryType::REL_GROUP_ENTRY, updateRecord.tableID); auto& table = StorageManager::Get(clientContext)->getTable(tableID)->cast(); const auto anchorState = updateRecord.ownedRelIDVector->state; DASSERT(anchorState == updateRecord.ownedSrcNodeIDVector->state && diff --git a/src/storage/wal/records/table_insertion_record_replay.cpp b/src/storage/wal/records/table_insertion_record_replay.cpp index 59eb94f14e..1179e56144 100644 --- a/src/storage/wal/records/table_insertion_record_replay.cpp +++ b/src/storage/wal/records/table_insertion_record_replay.cpp @@ -1,3 +1,4 @@ +#include "catalog/catalog_entry/catalog_entry_type.h" #include "common/exception/runtime.h" #include "storage/local_storage/local_rel_table.h" #include "storage/partition_storage_registry.h" @@ -29,7 +30,8 @@ void WALReplayer::replayTableInsertionRecord(const WALRecord& walRecord) const { void WALReplayer::replayNodeTableInsertRecord(const WALRecord& walRecord) const { const auto& insertionRecord = walRecord.constCast(); - const auto tableID = insertionRecord.tableID; + const auto tableID = + getReplayedEntryID(catalog::CatalogEntryType::NODE_TABLE_ENTRY, insertionRecord.tableID); // A torn WAL tail can deserialize to a record with zero vectors. operator[] on an empty // vector is UB (native crash, uncatchable by dry-replay truncation), so reject it here. if (insertionRecord.ownedVectors.empty() || insertionRecord.ownedVectors[0] == nullptr || @@ -69,7 +71,8 @@ void WALReplayer::replayNodeTableInsertRecord(const WALRecord& walRecord) const void WALReplayer::replayRelTableInsertRecord(const WALRecord& walRecord) const { const auto& insertionRecord = walRecord.constCast(); - const auto tableID = insertionRecord.tableID; + const auto tableID = + getReplayedEntryID(catalog::CatalogEntryType::REL_GROUP_ENTRY, insertionRecord.tableID); if (insertionRecord.ownedVectors.empty() || insertionRecord.ownedVectors[0] == nullptr || insertionRecord.ownedVectors[0]->state == nullptr) { throw RuntimeException( diff --git a/src/storage/wal/wal_record.cpp b/src/storage/wal/wal_record.cpp index 7dad5b2bfe..58f69a8dec 100644 --- a/src/storage/wal/wal_record.cpp +++ b/src/storage/wal/wal_record.cpp @@ -20,6 +20,8 @@ void WALRecord::serializeWithLength(Serializer& serializer, const WALRecord& rec auto bufferWriter = std::make_shared(); Serializer bufferSerializer{bufferWriter}; record.serialize(bufferSerializer); + bufferSerializer.writeDebuggingInfo("ownerCatalogName"); + bufferSerializer.write(record.ownerCatalogName); const auto recordLength = bufferWriter->getSize(); serializer.write(recordLength); @@ -90,6 +92,10 @@ std::unique_ptr WALRecord::deserialize(Deserializer& deserializer, throw RuntimeException("Corrupted wal file. Read out invalid WAL record type."); } } + if (deserializer.hasRemainingData()) { + deserializer.validateDebuggingInfo(key, "ownerCatalogName"); + deserializer.deserializeValue(walRecord->ownerCatalogName); + } walRecord->type = type; deserializer.skipReadLimit(); deserializer.getReader()->onObjectEnd(); diff --git a/src/storage/wal/wal_replayer.cpp b/src/storage/wal/wal_replayer.cpp index 06c54ba9f8..4bde24128e 100644 --- a/src/storage/wal/wal_replayer.cpp +++ b/src/storage/wal/wal_replayer.cpp @@ -1,7 +1,9 @@ #include "storage/wal/wal_replayer.h" +#include #include +#include "catalog/catalog.h" #include "common/constants.h" #include "common/exception/checkpoint.h" #include "common/exception/internal.h" @@ -54,6 +56,24 @@ static std::string unsupportedCheckpointFormatMessage(uint64_t version) { version, WAL::CHECKPOINT_BUNDLE_FORMAT_VERSION); } +namespace { +class ReplayOwnerScope { +public: + ReplayOwnerScope(main::DatabaseManager& dbManager, catalog::Catalog* ownerCatalog) + : dbManager{dbManager}, previous{dbManager.getReplayOwnerCatalog()} { + dbManager.setReplayOwnerCatalog(ownerCatalog); + } + ~ReplayOwnerScope() { dbManager.setReplayOwnerCatalog(previous); } + + ReplayOwnerScope(const ReplayOwnerScope&) = delete; + ReplayOwnerScope& operator=(const ReplayOwnerScope&) = delete; + +private: + main::DatabaseManager& dbManager; + catalog::Catalog* previous; +}; +} // namespace + static void removePartitionChildShadowFiles(main::ClientContext& clientContext) { auto* dbManager = main::DatabaseManager::Get(clientContext); if (dbManager == nullptr) { @@ -87,6 +107,13 @@ static void recoverGraphCheckpoints(main::ClientContext& clientContext, } } +static void replayPendingGraphWALs(main::ClientContext& clientContext) { + auto* databaseManager = main::DatabaseManager::Get(clientContext); + if (databaseManager != nullptr) { + databaseManager->replayPendingGraphWALs(&clientContext); + } +} + static void removeGraphCheckpointShadows(main::ClientContext& clientContext) { auto* databaseManager = main::DatabaseManager::Get(clientContext); if (databaseManager == nullptr) { @@ -470,6 +497,7 @@ WALReplayer::GraphRecoveryState WALReplayer::prepareGraphCheckpoint(StorageManag void WALReplayer::replayGraphWAL(StorageManager& storageManager, const GraphRecoveryState& recoveryState) const { + replayedEntryIDs.clear(); try { for (const auto& range : recoveryState.walReplayRanges) { auto flags = FileFlags::READ_ONLY; @@ -505,6 +533,29 @@ void WALReplayer::replayGraphWAL(StorageManager& storageManager, } } +void WALReplayer::recordReplayedEntryID(catalog::CatalogEntryType entryType, + common::oid_t recordedEntryID, common::oid_t replayedEntryID) const { + replayedEntryIDs[catalog::Catalog::Get(clientContext)][entryType][recordedEntryID] = + replayedEntryID; +} + +common::oid_t WALReplayer::getReplayedEntryID(catalog::CatalogEntryType entryType, + common::oid_t recordedEntryID) const { + const auto catalogIt = replayedEntryIDs.find(catalog::Catalog::Get(clientContext)); + if (catalogIt == replayedEntryIDs.end()) { + return recordedEntryID; + } + const auto typeIt = catalogIt->second.find(entryType); + if (typeIt == catalogIt->second.end()) { + return recordedEntryID; + } + const auto idIt = typeIt->second.find(recordedEntryID); + if (idIt == typeIt->second.end()) { + return recordedEntryID; + } + return idIt->second; +} + void WALReplayer::retireGraphCheckpointWALs(StorageManager& storageManager, const GraphRecoveryState& recoveryState) const { if (storageManager.isReadOnly()) { @@ -566,6 +617,7 @@ void WALReplayer::replayFrozenWAL(Checkpointer& checkpointer, bool throwOnWalRep auto walRecord = WALRecord::deserialize(deserializer, clientContext); replayWALRecord(*walRecord); } + replayPendingGraphWALs(clientContext); recoverGraphCheckpoints(clientContext, false, false); if (offsetDeserialized == 0) { // Nothing was committed, so the frozen WAL holds nothing to keep. @@ -645,6 +697,7 @@ void WALReplayer::replayActiveWAL(Checkpointer& checkpointer, bool throwOnWalRep auto walRecord = WALRecord::deserialize(deserializer, clientContext); replayWALRecord(*walRecord); } + replayPendingGraphWALs(clientContext); recoverGraphCheckpoints(clientContext, false, false); truncateWALFile(*fileInfo, offsetDeserialized); } @@ -673,7 +726,8 @@ void WALReplayer::replayCommittedCheckpoint(Checkpointer& checkpointer, } const auto mainDatabaseID = readPersistedDatabaseID(*mainStorageManager); const bool checkpointBundle = - replayInfo.checkpointFormatVersion == WAL::CHECKPOINT_BUNDLE_FORMAT_VERSION; + replayInfo.checkpointFormatVersion >= 1 && + replayInfo.checkpointFormatVersion <= WAL::CHECKPOINT_BUNDLE_FORMAT_VERSION; if (checkpointBundle && replayInfo.walDatabaseID.value != mainDatabaseID.value) { throw RuntimeException(std::format( "Cannot recover committed checkpoint: WAL {} does not match the Database ID of {}. Do " @@ -682,7 +736,7 @@ void WALReplayer::replayCommittedCheckpoint(Checkpointer& checkpointer, checkpointWALPath, clientContext.getDatabasePath())); } if (!VirtualFileSystem::GetUnsafe(clientContext) - ->fileOrPathExists(shadowFilePath, &clientContext)) { + ->fileOrPathExists(shadowFilePath, &clientContext)) { throw RuntimeException(std::format( "Cannot recover committed checkpoint: shadow file {} is missing.", shadowFilePath)); } @@ -757,20 +811,58 @@ WALReplayer::WALReplayInfo WALReplayer::dryReplay(FileInfo& fileInfo, bool throw } void WALReplayer::replayWALRecord(WALRecord& walRecord) const { + std::optional replayOwnerScope; + if (!walRecord.ownerCatalogName.empty()) { + auto dbManager = main::DatabaseManager::Get(clientContext); + if (dbManager == nullptr || !dbManager->hasGraph(walRecord.ownerCatalogName)) { + if (dbManager != nullptr && !failedOwnerNames.contains(walRecord.ownerCatalogName)) { + // A graph created earlier in this same WAL is not registered yet: its + // create-graph record only adds the catalog entry. Materialize that one + // graph (not the full graph-recovery pass) before routing this record into + // its catalog. A name whose materialization failed is remembered so its + // remaining records skip this load instead of repeating it; replayed graph + // DDL clears that memory (see the catalog-entry cases). The graph's own + // WAL replay is deferred by the loader to the next transaction-free point, + // because its standalone-session transactions cannot nest inside this one. + dbManager->loadGraphFromCatalog(MemoryManager::Get(clientContext), &clientContext, + walRecord.ownerCatalogName); + if (!dbManager->hasGraph(walRecord.ownerCatalogName)) { + failedOwnerNames.insert(walRecord.ownerCatalogName); + } + } + } + if (dbManager == nullptr || !dbManager->hasGraph(walRecord.ownerCatalogName)) { + // DROP GRAPH reclaims the graph's files immediately, while records its + // earlier committed transactions logged stay in the main WAL until the + // next checkpoint. Such an owner cannot be materialized (no files to load) + // and no catalog remains to apply the record to. Skip it, exactly as + // loadGraphsFromCatalog skips graphs whose files are gone; throwing here + // would wedge recovery permanently, because the committed prefix can + // never replay past this record and the WAL can never be retired. + return; + } + replayOwnerScope.emplace(*dbManager, + dbManager->getGraphCatalog(walRecord.ownerCatalogName)); + } switch (walRecord.type) { case WALRecordType::BEGIN_TRANSACTION_RECORD: { TransactionContext::Get(clientContext)->beginRecoveryTransaction(); } break; case WALRecordType::COMMIT_RECORD: { TransactionContext::Get(clientContext)->commit(); + // The enclosing recovery transaction just completed, so graph WALs queued by lazy + // materialization inside it can now replay without nesting their transactions. + replayPendingGraphWALs(clientContext); } break; case WALRecordType::CREATE_CATALOG_ENTRY_RECORD: { + failedOwnerNames.clear(); replayCreateCatalogEntryRecord(walRecord); } break; case WALRecordType::CREATE_INDEX_RECORD: { replayCreateIndexRecord(walRecord); } break; case WALRecordType::DROP_CATALOG_ENTRY_RECORD: { + failedOwnerNames.clear(); replayDropCatalogEntryRecord(walRecord); } break; case WALRecordType::ALTER_TABLE_ENTRY_RECORD: { diff --git a/src/transaction/transaction.cpp b/src/transaction/transaction.cpp index eeafc5c905..37b71b610e 100644 --- a/src/transaction/transaction.cpp +++ b/src/transaction/transaction.cpp @@ -4,6 +4,8 @@ #include "catalog/catalog_entry/sequence_catalog_entry.h" #include "common/exception/runtime.h" #include "main/client_context.h" +#include "main/database.h" +#include "main/database_manager.h" #include "main/db_config.h" #include "storage/local_storage/local_node_table.h" #include "storage/local_storage/local_storage.h" @@ -31,7 +33,7 @@ bool LocalCacheManager::put(std::unique_ptr object) { Transaction::Transaction(main::ClientContext& clientContext, TransactionType transactionType, common::transaction_t transactionID, common::transaction_t startTS) : type{transactionType}, ID{transactionID}, startTS{startTS}, - commitTS{common::INVALID_TRANSACTION}, forceCheckpoint{false}, hasCatalogChanges{false} { + commitTS{common::INVALID_TRANSACTION}, forceCheckpoint{false} { this->clientContext = &clientContext; localStorage = std::make_unique(clientContext); undoBuffer = std::make_unique(storage::MemoryManager::Get(clientContext)); @@ -43,15 +45,14 @@ Transaction::Transaction(main::ClientContext& clientContext, TransactionType tra Transaction::Transaction(TransactionType transactionType) noexcept : type{transactionType}, ID{DUMMY_TRANSACTION_ID}, startTS{DUMMY_START_TIMESTAMP}, commitTS{common::INVALID_TRANSACTION}, clientContext{nullptr}, undoBuffer{nullptr}, - forceCheckpoint{false}, hasCatalogChanges{false} { + forceCheckpoint{false} { currentTS = common::Timestamp::getCurrentTimestamp().value; } Transaction::Transaction(TransactionType transactionType, common::transaction_t ID, common::transaction_t startTS) noexcept : type{transactionType}, ID{ID}, startTS{startTS}, commitTS{common::INVALID_TRANSACTION}, - clientContext{nullptr}, undoBuffer{nullptr}, forceCheckpoint{false}, - hasCatalogChanges{false} { + clientContext{nullptr}, undoBuffer{nullptr}, forceCheckpoint{false} { currentTS = common::Timestamp::getCurrentTimestamp().value; } @@ -89,9 +90,31 @@ void Transaction::publishCommit() { } localStorage->commit(); undoBuffer->commit(commitTS); - if (hasCatalogChanges) { - Catalog::Get(*clientContext)->incrementVersion(); - hasCatalogChanges = false; + { + std::lock_guard lck{changedCatalogsMutex}; + if (!changedCatalogs.empty()) { + auto* mainCatalog = clientContext->getDatabase()->getCatalog(); + for (auto it = changedCatalogs.begin(); it != changedCatalogs.end();) { + if (*it == mainCatalog) { + mainCatalog->incrementVersion(); + it = changedCatalogs.erase(it); + } else { + ++it; + } + } + // A graph dropped after its change was recorded - in this or a concurrent + // transaction - destroys its catalog. Route the graph bumps through the + // database manager: under its registry lock, only catalogs it currently + // owns are advanced, so a catalog being dropped concurrently is either + // still owned (bumped before removal) or already gone (never touched). + if (!changedCatalogs.empty()) { + auto* dbManager = main::DatabaseManager::Get(*clientContext); + if (dbManager != nullptr) { + dbManager->bumpGraphCatalogVersions(changedCatalogs); + } + } + } + changedCatalogs.clear(); } for (auto& callback : commitCallbacks) { callback(*this); @@ -106,7 +129,10 @@ void Transaction::rollback(storage::WAL*) { // this must be rolled back first undoBuffer->rollback(clientContext); localStorage->rollback(); - hasCatalogChanges = false; + { + std::lock_guard lck{changedCatalogsMutex}; + changedCatalogs.clear(); + } for (auto& callback : rollbackCallbacks) { callback(*this); } @@ -122,15 +148,15 @@ void Transaction::pushRollbackCallback(std::function callbac rollbackCallbacks.push_back(std::move(callback)); } -bool Transaction::isUnCommitted(common::table_id_t tableID, common::offset_t nodeOffset) const { - return localStorage && localStorage->getLocalTable(tableID) && - nodeOffset >= getMinUncommittedNodeOffset(tableID); +bool Transaction::isUnCommitted(const storage::Table& table, common::offset_t nodeOffset) const { + return localStorage && localStorage->getLocalTable(table) && + nodeOffset >= getMinUncommittedNodeOffset(table); } void Transaction::pushCreateDropCatalogEntry(CatalogSet& catalogSet, CatalogEntry& catalogEntry, bool isInternal, bool skipLoggingToWAL) { undoBuffer->createCatalogEntry(catalogSet, catalogEntry); - hasCatalogChanges = true; + recordCatalogChange(catalogSet.getCatalog()); if (!shouldLogToWAL() || skipLoggingToWAL) { return; } @@ -142,7 +168,8 @@ void Transaction::pushCreateDropCatalogEntry(CatalogSet& catalogSet, CatalogEntr case CatalogEntryType::REL_GROUP_ENTRY: { if (catalogEntry.getType() == CatalogEntryType::DUMMY_ENTRY) { DASSERT(catalogEntry.isDeleted()); - localWAL->logCreateCatalogEntryRecord(newCatalogEntry, isInternal); + localWAL->logCreateCatalogEntryRecord(catalogSet.getOwnerCatalogName(), newCatalogEntry, + isInternal); } else { throw common::RuntimeException("This shouldn't happen. Alter table is not supported."); } @@ -154,14 +181,16 @@ void Transaction::pushCreateDropCatalogEntry(CatalogSet& catalogSet, CatalogEntr // We don't log SERIAL catalog entry creation as it is implicit return; } - localWAL->logCreateCatalogEntryRecord(newCatalogEntry, isInternal); + localWAL->logCreateCatalogEntryRecord(catalogSet.getOwnerCatalogName(), newCatalogEntry, + isInternal); } break; case CatalogEntryType::SCALAR_MACRO_ENTRY: case CatalogEntryType::TYPE_ENTRY: case CatalogEntryType::GRAPH_ENTRY: { DASSERT( catalogEntry.getType() == CatalogEntryType::DUMMY_ENTRY && catalogEntry.isDeleted()); - localWAL->logCreateCatalogEntryRecord(newCatalogEntry, isInternal); + localWAL->logCreateCatalogEntryRecord(catalogSet.getOwnerCatalogName(), newCatalogEntry, + isInternal); } break; case CatalogEntryType::DUMMY_ENTRY: { DASSERT(newCatalogEntry->isDeleted()); @@ -175,7 +204,8 @@ void Transaction::pushCreateDropCatalogEntry(CatalogSet& catalogSet, CatalogEntr case CatalogEntryType::REL_GROUP_ENTRY: case CatalogEntryType::SEQUENCE_ENTRY: case CatalogEntryType::GRAPH_ENTRY: { - localWAL->logDropCatalogEntryRecord(catalogEntry.getOID(), catalogEntry.getType()); + localWAL->logDropCatalogEntryRecord(catalogSet.getOwnerCatalogName(), + catalogEntry.getOID(), catalogEntry.getType()); } break; case CatalogEntryType::SCALAR_FUNCTION_ENTRY: case CatalogEntryType::TABLE_FUNCTION_ENTRY: @@ -204,20 +234,21 @@ void Transaction::pushCreateDropCatalogEntry(CatalogSet& catalogSet, CatalogEntr void Transaction::pushAlterCatalogEntry(CatalogSet& catalogSet, CatalogEntry& catalogEntry, const binder::BoundAlterInfo& alterInfo, bool skipLoggingToWAL) { undoBuffer->createCatalogEntry(catalogSet, catalogEntry); - hasCatalogChanges = true; + recordCatalogChange(catalogSet.getCatalog()); if (shouldLogToWAL() && !skipLoggingToWAL) { DASSERT(localWAL); - localWAL->logAlterCatalogEntryRecord(&alterInfo); + localWAL->logAlterCatalogEntryRecord(catalogSet.getOwnerCatalogName(), &alterInfo); } } void Transaction::pushSequenceChange(SequenceCatalogEntry* sequenceEntry, int64_t kCount, const SequenceRollbackData& data) { undoBuffer->createSequenceChange(*sequenceEntry, data); - hasCatalogChanges = true; + recordCatalogChange(sequenceEntry->getOwningCatalog()); if (shouldLogToWAL()) { DASSERT(localWAL); - localWAL->logUpdateSequenceRecord(sequenceEntry->getOID(), kCount); + localWAL->logUpdateSequenceRecord(sequenceEntry->getOwningCatalogName(), + sequenceEntry->getOID(), kCount); } } @@ -239,15 +270,21 @@ void Transaction::pushVectorUpdateInfo(storage::UpdateInfo& updateInfo, Transaction::~Transaction() = default; -common::offset_t Transaction::getMinUncommittedNodeOffset(common::table_id_t tableID) const { - if (localStorage && localStorage->getLocalTable(tableID)) { - return localStorage->getLocalTable(tableID) - ->cast() - .getStartOffset(); +common::offset_t Transaction::getMinUncommittedNodeOffset(const storage::Table& table) const { + if (localStorage && localStorage->getLocalTable(table)) { + return localStorage->getLocalTable(table)->cast().getStartOffset(); } return 0; } +void Transaction::recordCatalogChange(catalog::Catalog* catalog) { + if (catalog == nullptr) { + return; + } + std::lock_guard lck{changedCatalogsMutex}; + changedCatalogs.insert(catalog); +} + Transaction* Transaction::Get(const main::ClientContext& context) { return TransactionContext::Get(context)->getActiveTransaction(); } From 9e792155a4c47215e36c5f48e448477c036e9f01 Mon Sep 17 00:00:00 2001 From: Aikins Laryea Date: Sat, 3 Oct 2026 15:39:53 +0000 Subject: [PATCH 02/17] test(transaction): cover graph-owned WAL record replay across owner catalogs Replay graph-owned uncheckpointed records into the owner's catalog, cross-graph manual transactions into each owner, and a graph's DDL plus first row committed to a single WAL. Assert row content per graph so cross-graph swaps are caught, including rel inserts staged for two graphs in one manual transaction. A checkpoint bundle persisted by an earlier bundle-format version must recover through the bundle path and keep applying graph shadows. --- src/storage/table/node_table.cpp | 4 +- src/storage/table/rel_table.cpp | 2 +- src/storage/wal/wal_replayer.cpp | 2 +- test/transaction/checkpoint_test.cpp | 866 ++++++++++++++++++++++++++- test/transaction/wal_test.cpp | 16 +- 5 files changed, 867 insertions(+), 23 deletions(-) diff --git a/src/storage/table/node_table.cpp b/src/storage/table/node_table.cpp index b3e20f70c4..d1f2e72c99 100644 --- a/src/storage/table/node_table.cpp +++ b/src/storage/table/node_table.cpp @@ -693,7 +693,7 @@ void NodeTable::commit(main::ClientContext* context, TableCatalogEntry* tableEnt // 2. Set deleted flag for tuples that are deleted in local storage. row_idx_t numLocalRows = 0u; for (auto localNodeGroupIdx = 0u; localNodeGroupIdx < localNodeTable.getNumNodeGroups(); - localNodeGroupIdx++) { + localNodeGroupIdx++) { const auto localNodeGroup = localNodeTable.getNodeGroup(localNodeGroupIdx); if (localNodeGroup->hasDeletions(transaction)) { // TODO(Guodong): Assume local storage is small here. Should optimize the loop away by @@ -1060,7 +1060,7 @@ void NodeTable::scanIndexColumns(main::ClientContext* context, IndexScanHelper& progressBar->updateProgress(queryID.value(), numNodeGroups == 0 ? 1.0 : 0.01); } for (node_group_idx_t nodeGroupToScan = 0u; nodeGroupToScan < numNodeGroups; - ++nodeGroupToScan) { + ++nodeGroupToScan) { scanState->nodeGroup = nodeGroups_.getNodeGroupNoLock(nodeGroupToScan); // It is possible for the node group to have no chunked groups if we are rolling back due to diff --git a/src/storage/table/rel_table.cpp b/src/storage/table/rel_table.cpp index 9252c64540..7a018359b3 100644 --- a/src/storage/table/rel_table.cpp +++ b/src/storage/table/rel_table.cpp @@ -756,7 +756,7 @@ std::vector> RelTable::getDegreeEntries( auto* relTableData = getDirectedTableData(direction); auto* csrLengthColumn = relTableData->getCSRLengthColumn(); for (node_group_idx_t nodeGroupIdx = 0; nodeGroupIdx < relTableData->getNumNodeGroups(); - nodeGroupIdx++) { + nodeGroupIdx++) { auto* nodeGroup = relTableData->getNodeGroup(nodeGroupIdx); if (!nodeGroup) { continue; diff --git a/src/storage/wal/wal_replayer.cpp b/src/storage/wal/wal_replayer.cpp index 4bde24128e..013d99b6ab 100644 --- a/src/storage/wal/wal_replayer.cpp +++ b/src/storage/wal/wal_replayer.cpp @@ -736,7 +736,7 @@ void WALReplayer::replayCommittedCheckpoint(Checkpointer& checkpointer, checkpointWALPath, clientContext.getDatabasePath())); } if (!VirtualFileSystem::GetUnsafe(clientContext) - ->fileOrPathExists(shadowFilePath, &clientContext)) { + ->fileOrPathExists(shadowFilePath, &clientContext)) { throw RuntimeException(std::format( "Cannot recover committed checkpoint: shadow file {} is missing.", shadowFilePath)); } diff --git a/test/transaction/checkpoint_test.cpp b/test/transaction/checkpoint_test.cpp index 8998915266..71e5f90b02 100644 --- a/test/transaction/checkpoint_test.cpp +++ b/test/transaction/checkpoint_test.cpp @@ -17,6 +17,7 @@ #include "api_test/private_api_test.h" #include "catalog/catalog.h" #include "catalog/catalog_entry/rel_group_catalog_entry.h" +#include "catalog/catalog_entry/sequence_catalog_entry.h" #include "catalog/catalog_entry/table_catalog_entry.h" #include "common/checksum.h" #include "common/exception/runtime.h" @@ -1094,11 +1095,23 @@ static BinaryData serializeCheckpointRecord(const WALRecord& record) { return writer->getData(); } +// The pre-owner-tagging WAL record layout: a length-prefixed payload with no owner suffix. +static BinaryData serializeLegacyFormatRecord(const WALRecord& record) { + auto recordBufferWriter = std::make_shared(); + Serializer recordSerializer{recordBufferWriter}; + record.serialize(recordSerializer); + auto writer = std::make_shared(); + Serializer serializer{writer}; + const auto recordLength = recordBufferWriter->getSize(); + serializer.write(recordLength); + serializer.write(recordBufferWriter->getBlobData(), recordLength); + return writer->getData(); +} + static void rewriteCheckpointRecord(main::ClientContext& context, const std::string& walPath, - const WALRecord& replacement, std::optional enableChecksumsOverride = std::nullopt) { + BinaryData replacementRecord, std::optional enableChecksumsOverride = std::nullopt) { const auto currentRecord = serializeCheckpointRecord(CheckpointRecord{WAL::CHECKPOINT_BUNDLE_FORMAT_VERSION}); - const auto legacyRecord = serializeCheckpointRecord(replacement); const auto enableChecksums = enableChecksumsOverride.value_or(context.getDBConfig()->enableChecksums); const auto checksumSize = enableChecksums ? sizeof(uint64_t) : 0; @@ -1123,18 +1136,20 @@ static void rewriteCheckpointRecord(main::ClientContext& context, const std::str } } fileInfo->truncate(recordOffset); - fileInfo->writeFile(legacyRecord.data.get(), legacyRecord.size, recordOffset); + fileInfo->writeFile(replacementRecord.data.get(), replacementRecord.size, recordOffset); if (enableChecksums) { - const auto legacyChecksum = checksum(legacyRecord.data.get(), legacyRecord.size); - fileInfo->writeFile(reinterpret_cast(&legacyChecksum), - sizeof(legacyChecksum), recordOffset + legacyRecord.size); + const auto replacementChecksum = + checksum(replacementRecord.data.get(), replacementRecord.size); + fileInfo->writeFile(reinterpret_cast(&replacementChecksum), + sizeof(replacementChecksum), recordOffset + replacementRecord.size); } fileInfo->syncFile(); } static void rewriteCheckpointRecordAsLegacy(main::ClientContext& context, const std::string& walPath, std::optional enableChecksumsOverride) { - rewriteCheckpointRecord(context, walPath, LegacyCheckpointRecord{}, enableChecksumsOverride); + rewriteCheckpointRecord(context, walPath, serializeLegacyFormatRecord(LegacyCheckpointRecord{}), + enableChecksumsOverride); } static void writeShadowDatabaseID(main::ClientContext& context, const std::string& shadowPath, @@ -1361,7 +1376,7 @@ TEST_F(CheckpointRetryAfterFailureTest, RejectsNewerCheckpointFormatVersion) { auto* context = getClientContext(*conn); rewriteCheckpointRecord(*context, StorageUtils::getCheckpointWALFilePath(databasePath), - CheckpointRecord{WAL::CHECKPOINT_BUNDLE_FORMAT_VERSION + 1}); + serializeCheckpointRecord(CheckpointRecord{WAL::CHECKPOINT_BUNDLE_FORMAT_VERSION + 1})); conn.reset(); database.reset(); const auto databaseBeforeRecovery = readFile(databasePath); @@ -1369,13 +1384,43 @@ TEST_F(CheckpointRetryAfterFailureTest, RejectsNewerCheckpointFormatVersion) { createDBAndConn(); FAIL() << "Expected a newer checkpoint format version to be rejected."; } catch (const RuntimeException& e) { - EXPECT_NE(std::string{e.what()}.find("unsupported checkpoint format version 2"), + EXPECT_NE(std::string{e.what()}.find(std::format("unsupported checkpoint format version {}", + WAL::CHECKPOINT_BUNDLE_FORMAT_VERSION + 1)), std::string::npos) << e.what(); } EXPECT_EQ(readFile(databasePath), databaseBeforeRecovery); } +// A checkpoint bundle persisted by an earlier bundle-format version (e.g. 1) must recover +// through the bundle path, not the pre-bundle legacy path that discards graph shadows. +TEST_F(CheckpointRetryAfterFailureTest, RecoversOlderCheckpointBundleFormatVersion) { + if (inMemMode || systemConfig->checkpointThreshold == 0) { + GTEST_SKIP(); + } + insertNodes(0, 100); + ASSERT_TRUE(conn->query("CREATE GRAPH bundle_graph ANY;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH bundle_graph;")->isSuccess()); + ASSERT_TRUE( + conn->query("CREATE NODE TABLE User(id INT64 PRIMARY KEY, name STRING);")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE (:User {id: 1, name: 'Alice'});")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + failCheckpointWith(); + + auto* context = getClientContext(*conn); + rewriteCheckpointRecord(*context, StorageUtils::getCheckpointWALFilePath(databasePath), + serializeCheckpointRecord(CheckpointRecord{1})); + conn.reset(); + database.reset(); + createDBAndConn(); + + checkNodes(100); + ASSERT_TRUE(conn->query("USE GRAPH bundle_graph;")->isSuccess()); + auto result = conn->query("MATCH (n:User {name: 'Alice'}) RETURN COUNT(n);"); + ASSERT_TRUE(result->isSuccess()) << result->getErrorMessage(); + ASSERT_EQ(result->getNext()->getValue(0)->getValue(), 1); +} + TEST_F(CheckpointRetryAfterFailureTest, BundledShadowUsesCompatibilityGuardAndRecovers) { if (inMemMode || systemConfig->checkpointThreshold == 0) { GTEST_SKIP(); @@ -2229,7 +2274,7 @@ TEST_F(FlakyCheckpointerTest, RejectsNewerGraphCheckpointFormatVersion) { FlakyCheckpointer::resetCheckpointer(*graphContext); ASSERT_TRUE(std::filesystem::exists(graphCheckpointWALPath)); rewriteCheckpointRecord(*graphContext, graphCheckpointWALPath, - CheckpointRecord{WAL::CHECKPOINT_BUNDLE_FORMAT_VERSION + 1}); + serializeCheckpointRecord(CheckpointRecord{WAL::CHECKPOINT_BUNDLE_FORMAT_VERSION + 1})); graphConnection.reset(); graphDatabase.reset(); @@ -2243,7 +2288,9 @@ TEST_F(FlakyCheckpointerTest, RejectsNewerGraphCheckpointFormatVersion) { } catch (const RuntimeException& e) { const std::string message = e.what(); EXPECT_NE(message.find("Cannot recover graph WAL"), std::string::npos) << message; - EXPECT_NE(message.find("unsupported checkpoint format version 2"), std::string::npos) + EXPECT_NE(message.find(std::format("unsupported checkpoint format version {}", + WAL::CHECKPOINT_BUNDLE_FORMAT_VERSION + 1)), + std::string::npos) << message; } EXPECT_EQ(readFile(graphPath), graphDataBefore); @@ -2421,6 +2468,803 @@ TEST_P(TornGraphWALHeaderTest, FollowsReplayFailureMode) { INSTANTIATE_TEST_SUITE_P(ReplayFailureMode, TornGraphWALHeaderTest, ::testing::Bool(), [](const ::testing::TestParamInfo& info) { return info.param ? "Strict" : "NonStrict"; }); +// Graph-table commits are logged to the main WAL. Replaying that WAL must route each record +// through its owning graph's catalog: table IDs are per-catalog, so resolving a graph's row +// against main's catalog misroutes it into a same-ID main table. +TEST_F(FlakyCheckpointerTest, ReplaysGraphOwnedRecordsIntoOwnerCatalog) { + if (inMemMode || systemConfig->checkpointThreshold == 0) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE NODE TABLE main_test(id INT64 PRIMARY KEY);")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH owner_graph ANY;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH owner_graph;")->isSuccess()); + ASSERT_TRUE( + conn->query("CREATE NODE TABLE User(id INT64 PRIMARY KEY, name STRING);")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(conn->query("CHECKPOINT;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH owner_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE (:User {id: 1, name: 'Alice'});")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE (:main_test {id: 1});")->isSuccess()); + + conn.reset(); + database.reset(); + createDBAndConn(); + + auto result = conn->query("MATCH (n:main_test) RETURN COUNT(n);"); + ASSERT_TRUE(result->isSuccess()) << result->getErrorMessage(); + ASSERT_EQ(result->getNext()->getValue(0)->getValue(), 1); + result = conn->query("MATCH (n:User) RETURN COUNT(n);"); + EXPECT_FALSE(result->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH owner_graph;")->isSuccess()); + result = conn->query("MATCH (n:User {name: 'Alice'}) RETURN COUNT(n);"); + ASSERT_TRUE(result->isSuccess()) << result->getErrorMessage(); + ASSERT_EQ(result->getNext()->getValue(0)->getValue(), 1); + + ASSERT_TRUE(conn->query("CREATE (:User {id: 2, name: 'Bob'});")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + conn.reset(); + database.reset(); + createDBAndConn(); + ASSERT_TRUE(conn->query("USE GRAPH owner_graph;")->isSuccess()); + result = conn->query("MATCH (n:User) RETURN COUNT(n);"); + ASSERT_TRUE(result->isSuccess()) << result->getErrorMessage(); + ASSERT_EQ(result->getNext()->getValue(0)->getValue(), 2); +} + +// A manual transaction that inserts into two graphs stages both graphs' local tables in one +// local storage and commits all its records into one WAL. The per-owner local-storage key +// keeps the two inserts distinct at commit, and the per-record owner tags route replay of +// the single WAL into each graph's catalog. +TEST_F(FlakyCheckpointerTest, ReplaysCrossGraphTransactionIntoEachOwnerCatalog) { + if (inMemMode || systemConfig->checkpointThreshold == 0) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH left_graph ANY;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH right_graph ANY;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH left_graph;")->isSuccess()); + ASSERT_TRUE( + conn->query("CREATE NODE TABLE User(id INT64 PRIMARY KEY, name STRING);")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH right_graph;")->isSuccess()); + ASSERT_TRUE( + conn->query("CREATE NODE TABLE User(id INT64 PRIMARY KEY, name STRING);")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(conn->query("CHECKPOINT;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH left_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE (:User {id: 1, name: 'A'});")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH right_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE (:User {id: 1, name: 'B'});")->isSuccess()); + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH left_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE (:User {id: 2, name: 'C'});")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH right_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE (:User {id: 2, name: 'D'});")->isSuccess()); + ASSERT_TRUE(conn->query("COMMIT;")->isSuccess()); + + conn.reset(); + database.reset(); + createDBAndConn(); + + ASSERT_TRUE(conn->query("USE GRAPH left_graph;")->isSuccess()); + auto result = conn->query("MATCH (n:User) RETURN n.name ORDER BY n.name;"); + ASSERT_TRUE(result->isSuccess()) << result->getErrorMessage(); + std::vector names; + while (result->hasNext()) { + names.push_back(result->getNext()->getValue(0)->getValue()); + } + ASSERT_EQ(names, (std::vector{"A", "C"})); + + ASSERT_TRUE(conn->query("USE GRAPH right_graph;")->isSuccess()); + result = conn->query("MATCH (n:User) RETURN n.name ORDER BY n.name;"); + ASSERT_TRUE(result->isSuccess()) << result->getErrorMessage(); + names.clear(); + while (result->hasNext()) { + names.push_back(result->getNext()->getValue(0)->getValue()); + } + ASSERT_EQ(names, (std::vector{"B", "D"})); +} + +// A manual transaction that stages rel inserts for two graphs routes each rel to its own +// graph's rel table at commit, even though both graphs' rel tables share the same table ID. +// The two rels use different directions (left: L3->L4, right: R4->R3) so a swapped owner +// replay is detectable instead of producing an isomorphic end state. +TEST_F(FlakyCheckpointerTest, ReplaysCrossGraphRelTransactionIntoEachOwnerCatalog) { + if (inMemMode || systemConfig->checkpointThreshold == 0) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH left_graph ANY;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH right_graph ANY;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH left_graph;")->isSuccess()); + ASSERT_TRUE( + conn->query("CREATE NODE TABLE User(id INT64 PRIMARY KEY, name STRING);")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE REL TABLE Follows(FROM User TO User);")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH right_graph;")->isSuccess()); + ASSERT_TRUE( + conn->query("CREATE NODE TABLE User(id INT64 PRIMARY KEY, name STRING);")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE REL TABLE Follows(FROM User TO User);")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(conn->query("CHECKPOINT;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH left_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE (:User {id: 1, name: 'L3'});")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE (:User {id: 2, name: 'L4'});")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH right_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE (:User {id: 1, name: 'R3'});")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE (:User {id: 2, name: 'R4'});")->isSuccess()); + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH left_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("MATCH (a:User {name: 'L3'}), (b:User {name: 'L4'}) " + "CREATE (a)-[:Follows]->(b);") + ->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH right_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("MATCH (a:User {name: 'R4'}), (b:User {name: 'R3'}) " + "CREATE (a)-[:Follows]->(b);") + ->isSuccess()); + ASSERT_TRUE(conn->query("COMMIT;")->isSuccess()); + + conn.reset(); + database.reset(); + createDBAndConn(); + + ASSERT_TRUE(conn->query("USE GRAPH left_graph;")->isSuccess()); + auto result = conn->query("MATCH (:User {name: 'L3'})-[:Follows]->(b:User) RETURN b.name;"); + ASSERT_TRUE(result->isSuccess()) << result->getErrorMessage(); + ASSERT_TRUE(result->hasNext()); + ASSERT_EQ(result->getNext()->getValue(0)->getValue(), "L4"); + ASSERT_FALSE(result->hasNext()); + result = conn->query("MATCH (:User {name: 'L4'})-[:Follows]->(b:User) RETURN COUNT(b);"); + ASSERT_TRUE(result->isSuccess()) << result->getErrorMessage(); + ASSERT_EQ(result->getNext()->getValue(0)->getValue(), 0); + + ASSERT_TRUE(conn->query("USE GRAPH right_graph;")->isSuccess()); + result = conn->query("MATCH (:User {name: 'R4'})-[:Follows]->(b:User) RETURN b.name;"); + ASSERT_TRUE(result->isSuccess()) << result->getErrorMessage(); + ASSERT_TRUE(result->hasNext()); + ASSERT_EQ(result->getNext()->getValue(0)->getValue(), "R3"); + ASSERT_FALSE(result->hasNext()); + result = conn->query("MATCH (:User {name: 'R3'})-[:Follows]->(b:User) RETURN COUNT(b);"); + ASSERT_TRUE(result->isSuccess()) << result->getErrorMessage(); + ASSERT_EQ(result->getNext()->getValue(0)->getValue(), 0); +} + +// CREATE GRAPH and the graph's first table and row all commit into the main WAL before any +// checkpoint. Replaying the WAL must rebuild the graph's catalog entry and load the row +// through the freshly created entry. +TEST_F(FlakyCheckpointerTest, ReplaysGraphDDLAndDataInOneWAL) { + if (inMemMode || systemConfig->checkpointThreshold == 0) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH ddl_graph ANY;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH ddl_graph;")->isSuccess()); + ASSERT_TRUE( + conn->query("CREATE NODE TABLE Person(id INT64 PRIMARY KEY, name STRING);")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE (:Person {id: 1, name: 'Alice'});")->isSuccess()); + + conn.reset(); + database.reset(); + createDBAndConn(); + + ASSERT_TRUE(conn->query("USE GRAPH ddl_graph;")->isSuccess()); + auto result = conn->query("MATCH (n:Person {id: 1}) RETURN n.name;"); + ASSERT_TRUE(result->isSuccess()) << result->getErrorMessage(); + ASSERT_EQ(result->getNext()->getValue(0)->getValue(), "Alice"); + result = conn->query("MATCH (n:Person) RETURN COUNT(n);"); + ASSERT_TRUE(result->isSuccess()) << result->getErrorMessage(); + ASSERT_EQ(result->getNext()->getValue(0)->getValue(), 1); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + result = conn->query("MATCH (n:Person) RETURN COUNT(n);"); + EXPECT_FALSE(result->isSuccess()); +} + +// A graph created and used inside one committed transaction also recovers: during replay +// the graph's catalog entry lives in the still-active recovery transaction, so graph +// materialization triggered by the transaction's own tagged records must see it. +TEST_F(FlakyCheckpointerTest, ReplaysGraphCreatedAndUsedInOneTransaction) { + if (inMemMode || systemConfig->checkpointThreshold == 0) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH same_txn_graph ANY;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH same_txn_graph;")->isSuccess()); + ASSERT_TRUE( + conn->query("CREATE NODE TABLE User(id INT64 PRIMARY KEY, name STRING);")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE (:User {id: 1, name: 'Alice'});")->isSuccess()); + ASSERT_TRUE(conn->query("COMMIT;")->isSuccess()); + + conn.reset(); + database.reset(); + createDBAndConn(); + + ASSERT_TRUE(conn->query("USE GRAPH same_txn_graph;")->isSuccess()); + auto result = conn->query("MATCH (n:User {id: 1}) RETURN n.name;"); + ASSERT_TRUE(result->isSuccess()) << result->getErrorMessage(); + ASSERT_EQ(result->getNext()->getValue(0)->getValue(), "Alice"); + result = conn->query("MATCH (n:User) RETURN COUNT(n);"); + ASSERT_TRUE(result->isSuccess()) << result->getErrorMessage(); + ASSERT_EQ(result->getNext()->getValue(0)->getValue(), 1); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + result = conn->query("MATCH (n:User) RETURN COUNT(n);"); + EXPECT_FALSE(result->isSuccess()); +} + +// A sequence advance records the owning catalog for a version bump at commit. If the same +// transaction later drops that graph, the catalog is destroyed before commit, and the +// commit's version pass must not touch the destroyed catalog. +TEST_F(FlakyCheckpointerTest, CommitSkipsCatalogDroppedLaterInTransaction) { + if (inMemMode || systemConfig->checkpointThreshold == 0) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH seq_graph ANY;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH seq_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE SEQUENCE s;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH seq_graph;")->isSuccess()); + auto result = conn->query("RETURN nextval('s');"); + ASSERT_TRUE(result->isSuccess()) << result->getErrorMessage(); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(conn->query("DROP GRAPH seq_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("COMMIT;")->isSuccess()); + + result = conn->query("USE GRAPH seq_graph;"); + EXPECT_FALSE(result->isSuccess()); + ASSERT_TRUE( + conn->query("CREATE NODE TABLE T(id INT64 PRIMARY KEY, name STRING);")->isSuccess()); + result = conn->query("MATCH (n:T) RETURN COUNT(n);"); + ASSERT_TRUE(result->isSuccess()) << result->getErrorMessage(); + ASSERT_EQ(result->getNext()->getValue(0)->getValue(), 0); +} + +// Parallel evaluators of nextval share one transaction, so catalog-change recording from +// concurrent threads must be synchronized. Each thread advances the same sequence through +// the active transaction; every advance must survive commit exactly once. +TEST_F(FlakyCheckpointerTest, ConcurrentNextvalRecordsCatalogChangesWithoutRacing) { + if (inMemMode || systemConfig->checkpointThreshold == 0) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE SEQUENCE s;")->isSuccess()); + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + auto* context = getClientContext(*conn); + auto* transaction = Transaction::Get(*context); + auto* catalog = context->getDatabase()->getCatalog(); + const auto catalogVersion0 = catalog->getVersion(); + auto* sequenceEntry = catalog->getSequenceEntry(transaction, "s", false); + constexpr auto threadCount = 8; + constexpr auto callsPerThread = 500; + std::vector threads; + for (auto t = 0; t < threadCount; ++t) { + threads.emplace_back([&] { + for (auto i = 0; i < callsPerThread; ++i) { + sequenceEntry->nextKVal(transaction, 1); + } + }); + } + for (auto& thread : threads) { + thread.join(); + } + ASSERT_TRUE(conn->query("COMMIT;")->isSuccess()); + ASSERT_EQ(catalog->getVersion(), catalogVersion0 + 1) + << "commit must advance the changed catalog's version exactly once"; + auto result = conn->query("RETURN nextval('s');"); + ASSERT_TRUE(result->isSuccess()) << result->getErrorMessage(); + ASSERT_EQ(result->getNext()->getValue(0)->getValue(), + threadCount * callsPerThread + 1); +} + +// A transaction's catalog-version bump at commit must land on exactly the catalogs it +// changed, regardless of which graph is current when the transaction commits. +TEST_F(FlakyCheckpointerTest, CommitBumpsVersionOnlyOfChangedGraphCatalog) { + if (inMemMode || systemConfig->checkpointThreshold == 0) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH version_graph_a ANY;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH version_graph_a;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE SEQUENCE s;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH version_graph_b ANY;")->isSuccess()); + auto* context = getClientContext(*conn); + auto* dbManager = main::DatabaseManager::Get(*context); + const auto mainVersion0 = context->getDatabase()->getCatalog()->getVersion(); + const auto graphAVersion0 = dbManager->getGraphCatalog("version_graph_a")->getVersion(); + const auto graphBVersion0 = dbManager->getGraphCatalog("version_graph_b")->getVersion(); + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH version_graph_a;")->isSuccess()); + auto result = conn->query("RETURN nextval('s');"); + ASSERT_TRUE(result->isSuccess()) << result->getErrorMessage(); + ASSERT_TRUE(conn->query("USE GRAPH version_graph_b;")->isSuccess()); + ASSERT_TRUE(conn->query("COMMIT;")->isSuccess()); + ASSERT_EQ(dbManager->getGraphCatalog("version_graph_a")->getVersion(), graphAVersion0 + 1) + << "commit must advance the changed graph's catalog version exactly once"; + ASSERT_EQ(dbManager->getGraphCatalog("version_graph_b")->getVersion(), graphBVersion0); + ASSERT_EQ(context->getDatabase()->getCatalog()->getVersion(), mainVersion0); +} + +// DROP GRAPH reclaims the graph's files immediately, while the owner-tagged records its +// earlier committed transactions logged stay in the main WAL until the next checkpoint. +// Recovery must skip those records instead of failing to reopen the database forever. +TEST_F(FlakyCheckpointerTest, DroppedGraphRecordsDoNotBlockRecovery) { + if (inMemMode || systemConfig->checkpointThreshold == 0) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH drop_recovery_graph ANY;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH drop_recovery_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE SEQUENCE s;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH keep_recovery_graph ANY;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH keep_recovery_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE SEQUENCE k;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(conn->query("CHECKPOINT;")->isSuccess()); + for (auto i = 0; i < 3; ++i) { + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH drop_recovery_graph;")->isSuccess()); + auto droppedResult = conn->query("RETURN nextval('s');"); + ASSERT_TRUE(droppedResult->isSuccess()) << droppedResult->getErrorMessage(); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(conn->query("COMMIT;")->isSuccess()); + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH keep_recovery_graph;")->isSuccess()); + auto keptResult = conn->query("RETURN nextval('k');"); + ASSERT_TRUE(keptResult->isSuccess()) << keptResult->getErrorMessage(); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(conn->query("COMMIT;")->isSuccess()); + } + ASSERT_TRUE(conn->query("DROP GRAPH drop_recovery_graph;")->isSuccess()); + + conn.reset(); + database.reset(); + createDBAndConn(); + + auto openResult = conn->query("RETURN 1;"); + ASSERT_TRUE(openResult->isSuccess()) << openResult->getErrorMessage(); + ASSERT_EQ(openResult->getNext()->getValue(0)->getValue(), 1); + ASSERT_TRUE(conn->query("USE GRAPH keep_recovery_graph;")->isSuccess()); + auto keptResult = conn->query("RETURN nextval('k');"); + ASSERT_TRUE(keptResult->isSuccess()) << keptResult->getErrorMessage(); + ASSERT_EQ(keptResult->getNext()->getValue(0)->getValue(), 4); + auto useResult = conn->query("USE GRAPH drop_recovery_graph;"); + EXPECT_FALSE(useResult->isSuccess()); +} + +// A commit bumping a graph catalog's version must not touch a catalog that a concurrent +// DROP GRAPH on another connection destroys mid-commit. The registry lock serializes +// the commit-time bump against the drop's removal. +TEST_F(FlakyCheckpointerTest, ConcurrentDropGraphDuringCommitKeepsCatalogsAlive) { + if (inMemMode || systemConfig->checkpointThreshold == 0) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL debug_enable_multi_writes=true;")->isSuccess()); + main::Connection dropConn(database.get()); + constexpr auto iterations = 100; + for (auto i = 0; i < iterations; ++i) { + ASSERT_TRUE(conn->query("CREATE GRAPH race_graph ANY;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH race_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE SEQUENCE s;")->isSuccess()); + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH race_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("RETURN nextval('s');")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + bool dropSucceeded = false; + std::thread dropper( + [&] { dropSucceeded = dropConn.query("DROP GRAPH race_graph;")->isSuccess(); }); + auto commitResult = conn->query("COMMIT;"); + dropper.join(); + ASSERT_TRUE(commitResult->isSuccess()) + << "iteration " << i << ": " << commitResult->getErrorMessage(); + ASSERT_TRUE(dropSucceeded) << "iteration " << i; + } + auto result = conn->query("RETURN 1;"); + ASSERT_TRUE(result->isSuccess()) << result->getErrorMessage(); + ASSERT_EQ(result->getNext()->getValue(0)->getValue(), 1); +} + +// A standalone session opened directly on a graph's data file commits into the graph's own +// WAL. Replaying the main WAL hits owner-tagged records for that graph, which must +// materialize only the named graph and defer its WAL replay to the enclosing transaction's +// commit instead of forcing a full catalog load inside the active recovery transaction. The +// standalone session writes catalog-entry DDL only: a node table it creates has a plain +// schema that no longer matches the ANY-graph catalog the records replay into, and an +// UPDATE_SEQUENCE record it logs resolves by raw sequence ID against the standalone +// catalog's incompatible ID space, so only its by-name DDL records are assertable. +TEST_F(FlakyCheckpointerTest, StandaloneGraphFileCommitDoesNotWedgeMainWALReplay) { + if (inMemMode || systemConfig->checkpointThreshold == 0) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH wedge_graph ANY;")->isSuccess()); + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH wedge_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE SEQUENCE parent_seq;")->isSuccess()); + auto nextResult = conn->query("RETURN nextval('parent_seq');"); + ASSERT_TRUE(nextResult->isSuccess()) << nextResult->getErrorMessage(); + ASSERT_EQ(nextResult->getNext()->getValue(0)->getValue(), 1); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(conn->query("COMMIT;")->isSuccess()); + + const auto graphPath = StorageUtils::getGraphPath(databasePath, "wedge_graph"); + const auto graphWALPath = StorageUtils::getWALFilePath(graphPath); + const auto graphShadowPath = StorageUtils::getShadowFilePath(graphPath); + ASSERT_FALSE(std::filesystem::exists(graphWALPath)); + + conn.reset(); + database.reset(); + + auto graphConfig = *systemConfig; + graphConfig.autoCheckpoint = false; + graphConfig.forceCheckpointOnClose = false; + auto graphDatabase = std::make_unique(graphPath, graphConfig); + auto graphConnection = std::make_unique(graphDatabase.get()); + auto tableResult = graphConnection->query("CREATE SEQUENCE carol_seq;"); + ASSERT_TRUE(tableResult->isSuccess()) << tableResult->getErrorMessage(); + ASSERT_TRUE(std::filesystem::exists(graphWALPath)); + // DDL results hold empty-schema factorized tables, which must be destroyed while their + // database is still open. + tableResult.reset(); + graphConnection.reset(); + graphDatabase.reset(); + + createDBAndConn(); + + auto openResult = conn->query("RETURN 1;"); + ASSERT_TRUE(openResult->isSuccess()) << openResult->getErrorMessage(); + ASSERT_EQ(openResult->getNext()->getValue(0)->getValue(), 1); + ASSERT_TRUE(conn->query("USE GRAPH wedge_graph;")->isSuccess()); + nextResult = conn->query("RETURN nextval('parent_seq');"); + ASSERT_TRUE(nextResult->isSuccess()) << nextResult->getErrorMessage(); + ASSERT_EQ(nextResult->getNext()->getValue(0)->getValue(), 2); + nextResult = conn->query("RETURN nextval('carol_seq');"); + ASSERT_TRUE(nextResult->isSuccess()) << nextResult->getErrorMessage(); + ASSERT_EQ(nextResult->getNext()->getValue(0)->getValue(), 1); + EXPECT_FALSE(std::filesystem::exists(graphShadowPath)); +} + +// A standalone session's graph WAL records entry IDs from that session's catalog, which +// lacks the ANY-graph infrastructure entries the graph materializes with, so raw +// recorded IDs can point at infrastructure entries here. Replay must translate the +// DROP SEQUENCE ID to the entry the standalone CREATE installed; resolving it raw +// drops the graph's infra serial, the following CREATE SEQUENCE record then fails on +// the name collision, and every later open wedges. The nextval results also pin the +// translation: both sequences replay fresh here, so any value other than 1 means a +// replay targeted the wrong entry. +TEST_F(FlakyCheckpointerTest, StandaloneSequenceDropRecreateDoesNotPoisonRecovery) { + if (inMemMode || systemConfig->checkpointThreshold == 0) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH wedge_graph ANY;")->isSuccess()); + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH wedge_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE SEQUENCE parent_seq;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(conn->query("COMMIT;")->isSuccess()); + + const auto graphPath = StorageUtils::getGraphPath(databasePath, "wedge_graph"); + conn.reset(); + database.reset(); + + auto graphConfig = *systemConfig; + graphConfig.autoCheckpoint = false; + graphConfig.forceCheckpointOnClose = false; + auto graphDatabase = std::make_unique(graphPath, graphConfig); + auto graphConnection = std::make_unique(graphDatabase.get()); + auto ddl1 = graphConnection->query("CREATE SEQUENCE carol_seq;"); + ASSERT_TRUE(ddl1->isSuccess()) << ddl1->getErrorMessage(); + auto ddl2 = graphConnection->query("DROP SEQUENCE carol_seq;"); + ASSERT_TRUE(ddl2->isSuccess()) << ddl2->getErrorMessage(); + auto ddl3 = graphConnection->query("CREATE SEQUENCE carol_seq;"); + ASSERT_TRUE(ddl3->isSuccess()) << ddl3->getErrorMessage(); + ddl1.reset(); + ddl2.reset(); + ddl3.reset(); + graphConnection.reset(); + graphDatabase.reset(); + + createDBAndConn(); + auto openResult = conn->query("RETURN 1;"); + ASSERT_TRUE(openResult->isSuccess()) << openResult->getErrorMessage(); + ASSERT_EQ(openResult->getNext()->getValue(0)->getValue(), 1); + ASSERT_TRUE(conn->query("USE GRAPH wedge_graph;")->isSuccess()); + auto parentResult = conn->query("RETURN nextval('parent_seq');"); + ASSERT_TRUE(parentResult->isSuccess()) << parentResult->getErrorMessage(); + ASSERT_EQ(parentResult->getNext()->getValue(0)->getValue(), 1); + auto carolResult = conn->query("RETURN nextval('carol_seq');"); + ASSERT_TRUE(carolResult->isSuccess()) << carolResult->getErrorMessage(); + ASSERT_EQ(carolResult->getNext()->getValue(0)->getValue(), 1); +} + +// A standalone session's graph WAL addresses rel data by the per-direction physical +// rel-table ID and binds a rel group through its endpoint node-table IDs, all recorded +// in that session's plain catalog, which lacks the ANY-graph infrastructure entries +// the graph materializes with. Replayed raw, the physical ID can land on a node table +// (the cast to RelTable fails and wedges every later open), and the recorded endpoints +// bind the group to infrastructure tables instead of the replayed node tables. CREATE +// replay must translate the group's endpoints and record the physical-table IDs so rel +// data replays against the tables this recovery created. +TEST_F(FlakyCheckpointerTest, StandaloneRelTableReplayDoesNotPoisonRecovery) { + if (inMemMode || systemConfig->checkpointThreshold == 0) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH rel_graph ANY;")->isSuccess()); + conn.reset(); + database.reset(); + + const auto graphPath = StorageUtils::getGraphPath(databasePath, "rel_graph"); + auto graphConfig = *systemConfig; + graphConfig.autoCheckpoint = false; + graphConfig.forceCheckpointOnClose = false; + auto graphDatabase = std::make_unique(graphPath, graphConfig); + auto graphConnection = std::make_unique(graphDatabase.get()); + auto ddl1 = graphConnection->query("CREATE NODE TABLE Person(id INT64 PRIMARY KEY);"); + ASSERT_TRUE(ddl1->isSuccess()) << ddl1->getErrorMessage(); + auto ddl2 = graphConnection->query("CREATE REL TABLE Knows(FROM Person TO Person);"); + ASSERT_TRUE(ddl2->isSuccess()) << ddl2->getErrorMessage(); + auto nodeInsert = graphConnection->query("CREATE (:Person {id: 1});"); + ASSERT_TRUE(nodeInsert->isSuccess()) << nodeInsert->getErrorMessage(); + nodeInsert = graphConnection->query("CREATE (:Person {id: 2});"); + ASSERT_TRUE(nodeInsert->isSuccess()) << nodeInsert->getErrorMessage(); + auto relInsert = graphConnection->query( + "MATCH (a:Person {id: 1}), (b:Person {id: 2}) CREATE (a)-[:Knows]->(b);"); + ASSERT_TRUE(relInsert->isSuccess()) << relInsert->getErrorMessage(); + ddl1.reset(); + ddl2.reset(); + nodeInsert.reset(); + relInsert.reset(); + graphConnection.reset(); + graphDatabase.reset(); + + createDBAndConn(); + auto openResult = conn->query("RETURN 1;"); + ASSERT_TRUE(openResult->isSuccess()) << openResult->getErrorMessage(); + ASSERT_EQ(openResult->getNext()->getValue(0)->getValue(), 1); + ASSERT_TRUE(conn->query("USE GRAPH rel_graph;")->isSuccess()); + auto nodeResult = conn->query("MATCH (n:Person) RETURN COUNT(n);"); + ASSERT_TRUE(nodeResult->isSuccess()) << nodeResult->getErrorMessage(); + ASSERT_EQ(nodeResult->getNext()->getValue(0)->getValue(), 2); + auto relResult = conn->query("MATCH (:Person)-[k:Knows]->(:Person) RETURN COUNT(k);"); + ASSERT_TRUE(relResult->isSuccess()) << relResult->getErrorMessage(); + ASSERT_EQ(relResult->getNext()->getValue(0)->getValue(), 1); +} + +// A standalone session's graph WAL records index operations in the standalone catalog's own +// index-ID space, which starts empty because the parent session's graph-owned records live in +// the main WAL. Parent recovery installs those parent-created indexes first, so replaying the +// standalone DROP INDEX by its recorded raw ID drops a parent-created index instead. CREATE +// INDEX replay must record the recorded-to-replayed index-ID mapping for the DROP to resolve. +TEST_F(FlakyCheckpointerTest, StandaloneIndexReplayDoesNotDropParentIndex) { + if (inMemMode || systemConfig->checkpointThreshold == 0) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH idx_graph ANY;")->isSuccess()); + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH idx_graph;")->isSuccess()); + ASSERT_TRUE( + conn->query("CREATE NODE TABLE ParentT(id INT64 PRIMARY KEY, val INT64);")->isSuccess()); + ASSERT_TRUE( + conn->query("CREATE ART INDEX parent_idx FOR (p:ParentT) ON (p.val);")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(conn->query("COMMIT;")->isSuccess()); + + const auto graphPath = StorageUtils::getGraphPath(databasePath, "idx_graph"); + conn.reset(); + database.reset(); + + auto graphConfig = *systemConfig; + graphConfig.autoCheckpoint = false; + graphConfig.forceCheckpointOnClose = false; + auto graphDatabase = std::make_unique(graphPath, graphConfig); + auto graphConnection = std::make_unique(graphDatabase.get()); + auto ddl1 = + graphConnection->query("CREATE NODE TABLE Person(id INT64 PRIMARY KEY, val INT64);"); + ASSERT_TRUE(ddl1->isSuccess()) << ddl1->getErrorMessage(); + auto ddl2 = graphConnection->query("CREATE ART INDEX person_idx FOR (p:Person) ON (p.val);"); + ASSERT_TRUE(ddl2->isSuccess()) << ddl2->getErrorMessage(); + auto ddl3 = graphConnection->query("DROP INDEX Person.person_idx;"); + ASSERT_TRUE(ddl3->isSuccess()) << ddl3->getErrorMessage(); + auto ddl4 = graphConnection->query("CREATE ART INDEX person_idx FOR (p:Person) ON (p.val);"); + ASSERT_TRUE(ddl4->isSuccess()) << ddl4->getErrorMessage(); + ddl1.reset(); + ddl2.reset(); + ddl3.reset(); + ddl4.reset(); + graphConnection.reset(); + graphDatabase.reset(); + + createDBAndConn(); + auto openResult = conn->query("RETURN 1;"); + ASSERT_TRUE(openResult->isSuccess()) << openResult->getErrorMessage(); + ASSERT_EQ(openResult->getNext()->getValue(0)->getValue(), 1); + ASSERT_TRUE(conn->query("USE GRAPH idx_graph;")->isSuccess()); + auto parentIdxResult = + conn->query("CALL SHOW_INDEXES() WHERE index_name = 'parent_idx' RETURN count(*);"); + ASSERT_TRUE(parentIdxResult->isSuccess()) << parentIdxResult->getErrorMessage(); + ASSERT_EQ(parentIdxResult->getNext()->getValue(0)->getValue(), 1); + parentIdxResult.reset(); + auto personIdxResult = + conn->query("CALL SHOW_INDEXES() WHERE index_name = 'person_idx' RETURN count(*);"); + ASSERT_TRUE(personIdxResult->isSuccess()) << personIdxResult->getErrorMessage(); + ASSERT_EQ(personIdxResult->getNext()->getValue(0)->getValue(), 1); + personIdxResult.reset(); + auto dropParentResult = conn->query("DROP INDEX ParentT.parent_idx;"); + ASSERT_TRUE(dropParentResult->isSuccess()) << dropParentResult->getErrorMessage(); + auto dropPersonResult = conn->query("DROP INDEX Person.person_idx;"); + ASSERT_TRUE(dropPersonResult->isSuccess()) << dropPersonResult->getErrorMessage(); +} + +// A staged rel insert into a graph-owned table resolves its owning catalog under the registry +// lock at commit time. When DROP GRAPH wins the race the commit must fail cleanly with the +// missing-graph binder error and leave the connection usable; when the commit wins it must +// succeed. Only the rel table is staged: a staged node table would crash the post-failure +// rollback, whose clear() dereferences the dropped table's freed storage (a pre-existing +// issue to fix separately). The drop-first phase pins the failure deterministically; the +// concurrent iterations alone accept whichever operation wins and so cannot detect a commit +// that silently discards the staged insert instead of failing. +TEST_F(FlakyCheckpointerTest, ConcurrentDropGraphDuringStagedInsertCommitFailsCleanly) { + if (inMemMode || systemConfig->checkpointThreshold == 0) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL debug_enable_multi_writes=true;")->isSuccess()); + main::Connection dropConn(database.get()); + + auto stageRelInsert = [&] { + ASSERT_TRUE(conn->query("CREATE GRAPH race_graph ANY;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH race_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE NODE TABLE t(id INT64 PRIMARY KEY);")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE REL TABLE e(FROM t TO t);")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE (:t {id: 1});")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE (:t {id: 2});")->isSuccess()); + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH race_graph;")->isSuccess()); + ASSERT_TRUE( + conn->query("MATCH (a:t {id: 1}), (b:t {id: 2}) CREATE (a)-[:e]->(b);")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + }; + + ASSERT_NO_FATAL_FAILURE(stageRelInsert()); + ASSERT_TRUE(dropConn.query("DROP GRAPH race_graph;")->isSuccess()); + auto droppedCommitResult = conn->query("COMMIT;"); + ASSERT_FALSE(droppedCommitResult->isSuccess()) << droppedCommitResult->getErrorMessage(); + EXPECT_NE(droppedCommitResult->getErrorMessage().find("No graph named race_graph"), + std::string::npos) + << droppedCommitResult->getErrorMessage(); + auto droppedProbe = conn->query("RETURN 1;"); + EXPECT_TRUE(droppedProbe->isSuccess()) << droppedProbe->getErrorMessage(); + + constexpr auto iterations = 100; + for (auto i = 0; i < iterations; ++i) { + ASSERT_NO_FATAL_FAILURE(stageRelInsert()); + bool dropSucceeded = false; + std::thread dropper( + [&] { dropSucceeded = dropConn.query("DROP GRAPH race_graph;")->isSuccess(); }); + auto commitResult = conn->query("COMMIT;"); + dropper.join(); + ASSERT_TRUE(dropSucceeded) << "iteration " << i; + if (commitResult->isSuccess()) { + continue; + } + EXPECT_NE(commitResult->getErrorMessage().find("No graph named race_graph"), + std::string::npos) + << "iteration " << i << ": " << commitResult->getErrorMessage(); + auto probe = conn->query("RETURN 1;"); + EXPECT_TRUE(probe->isSuccess()) << "iteration " << i << ": " << probe->getErrorMessage(); + } + auto result = conn->query("RETURN 1;"); + ASSERT_TRUE(result->isSuccess()) << result->getErrorMessage(); + ASSERT_EQ(result->getNext()->getValue(0)->getValue(), 1); +} + +// The deferred graph-WAL flush must replay into the owning graph only: records committed by +// a standalone session on one graph's file must not appear in a sibling graph's catalog, +// while the sibling's own owner-tagged records from the main WAL must still be applied. +TEST_F(FlakyCheckpointerTest, DeferredGraphWALReplayIsScopedToItsOwningGraph) { + if (inMemMode || systemConfig->checkpointThreshold == 0) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH multi_a ANY;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH multi_a;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE SEQUENCE seq_a;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH multi_b ANY;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH multi_b;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE SEQUENCE seq_b;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + + const auto graphPathB = StorageUtils::getGraphPath(databasePath, "multi_b"); + const auto graphWALPathB = StorageUtils::getWALFilePath(graphPathB); + const auto graphShadowPathB = StorageUtils::getShadowFilePath(graphPathB); + ASSERT_FALSE(std::filesystem::exists(graphWALPathB)); + + conn.reset(); + database.reset(); + + auto graphConfig = *systemConfig; + graphConfig.autoCheckpoint = false; + graphConfig.forceCheckpointOnClose = false; + auto graphDatabase = std::make_unique(graphPathB, graphConfig); + auto graphConnection = std::make_unique(graphDatabase.get()); + auto tableResult = graphConnection->query("CREATE SEQUENCE carol_seq;"); + ASSERT_TRUE(tableResult->isSuccess()) << tableResult->getErrorMessage(); + ASSERT_TRUE(std::filesystem::exists(graphWALPathB)); + // DDL results hold empty-schema factorized tables, which must be destroyed while their + // database is still open. + tableResult.reset(); + graphConnection.reset(); + graphDatabase.reset(); + + createDBAndConn(); + + auto openResult = conn->query("RETURN 1;"); + ASSERT_TRUE(openResult->isSuccess()) << openResult->getErrorMessage(); + ASSERT_TRUE(conn->query("USE GRAPH multi_b;")->isSuccess()); + auto nextB = conn->query("RETURN nextval('seq_b');"); + ASSERT_TRUE(nextB->isSuccess()) << nextB->getErrorMessage(); + ASSERT_EQ(nextB->getNext()->getValue(0)->getValue(), 1); + nextB = conn->query("RETURN nextval('carol_seq');"); + ASSERT_TRUE(nextB->isSuccess()) << nextB->getErrorMessage(); + ASSERT_EQ(nextB->getNext()->getValue(0)->getValue(), 1); + ASSERT_TRUE(conn->query("USE GRAPH multi_a;")->isSuccess()); + auto nextA = conn->query("RETURN nextval('seq_a');"); + ASSERT_TRUE(nextA->isSuccess()) << nextA->getErrorMessage(); + ASSERT_EQ(nextA->getNext()->getValue(0)->getValue(), 1); + auto userA = conn->query("RETURN nextval('carol_seq');"); + EXPECT_FALSE(userA->isSuccess()); + EXPECT_FALSE(std::filesystem::exists(graphShadowPathB)); +} + +// A checkpointed ANY graph whose infra tables were dropped reopens with them still +// dropped: an empty persisted table set is deliberate state, not missing initialization. +TEST_F(FlakyCheckpointerTest, KeepsDroppedAnyGraphTablesDroppedAfterCheckpoint) { + if (inMemMode || systemConfig->checkpointThreshold == 0) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH dropped_graph ANY;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH dropped_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("DROP TABLE _edges;")->isSuccess()); + ASSERT_TRUE(conn->query("DROP TABLE _nodes;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(conn->query("CHECKPOINT;")->isSuccess()); + + conn.reset(); + database.reset(); + createDBAndConn(); + + ASSERT_TRUE(conn->query("USE GRAPH dropped_graph;")->isSuccess()); + auto result = conn->query("MATCH (n:_nodes) RETURN COUNT(n);"); + EXPECT_FALSE(result->isSuccess()) << "_nodes was recreated after a checkpointed drop"; + result = conn->query("MATCH ()-[e:_edges]->() RETURN COUNT(e);"); + EXPECT_FALSE(result->isSuccess()) << "_edges was recreated after a checkpointed drop"; +} + class LegacyGraphMarkerWithoutShadowTest : public FlakyCheckpointerTest, public ::testing::WithParamInterface {}; diff --git a/test/transaction/wal_test.cpp b/test/transaction/wal_test.cpp index 070d1e91b1..9ef4313885 100644 --- a/test/transaction/wal_test.cpp +++ b/test/transaction/wal_test.cpp @@ -214,7 +214,7 @@ TEST_F(WalTest, WALSyncFailurePoisonsWALAndReturnsAllocatedCommitSequence) { &vfs); lbug::storage::LocalWAL localWAL(*lbug::storage::MemoryManager::Get(*conn->getClientContext()), false /* enableChecksums */); - localWAL.logLoadExtension("dummy"); + localWAL.logLoadExtension("" /* main */, "dummy"); localWAL.logCommit(); uint64_t walCommitSequence = 0; @@ -225,7 +225,7 @@ TEST_F(WalTest, WALSyncFailurePoisonsWALAndReturnsAllocatedCommitSequence) { failingFSPtr->setFailSync(false); lbug::storage::LocalWAL secondLocalWAL( *lbug::storage::MemoryManager::Get(*conn->getClientContext()), false /* enableChecksums */); - secondLocalWAL.logLoadExtension("dummy2"); + secondLocalWAL.logLoadExtension("" /* main */, "dummy2"); secondLocalWAL.logCommit(); walCommitSequence = 0; @@ -270,7 +270,7 @@ TEST_F(WalTest, OrdinaryCommitSyncsRegisteredNamespace) { &vfs); lbug::storage::LocalWAL localWAL(*lbug::storage::MemoryManager::Get(*conn->getClientContext()), false /* enableChecksums */); - localWAL.logLoadExtension("dummy"); + localWAL.logLoadExtension("" /* main */, "dummy"); localWAL.logCommit(); uint64_t walCommitSequence = 0; ASSERT_NO_THROW(wal.logCommittedWAL(localWAL, conn->getClientContext(), walCommitSequence)); @@ -290,7 +290,7 @@ TEST_F(WalTest, FailureAfterCheckpointRenamePoisonsWAL) { lbug::storage::WAL wal(databasePath, false /* readOnly */, false /* enableChecksums */, &vfs); lbug::storage::LocalWAL localWAL(*lbug::storage::MemoryManager::Get(*conn->getClientContext()), false /* enableChecksums */); - localWAL.logLoadExtension("dummy"); + localWAL.logLoadExtension("" /* main */, "dummy"); localWAL.logCommit(); uint64_t walCommitSequence = 0; ASSERT_NO_THROW(wal.logCommittedWAL(localWAL, conn->getClientContext(), walCommitSequence)); @@ -303,7 +303,7 @@ TEST_F(WalTest, FailureAfterCheckpointRenamePoisonsWAL) { lbug::storage::LocalWAL laterWAL(*lbug::storage::MemoryManager::Get(*conn->getClientContext()), false /* enableChecksums */); - laterWAL.logLoadExtension("later"); + laterWAL.logLoadExtension("" /* main */, "later"); laterWAL.logCommit(); walCommitSequence = 0; EXPECT_THROW(wal.logCommittedWAL(laterWAL, conn->getClientContext(), walCommitSequence), @@ -321,7 +321,7 @@ TEST_F(WalTest, DirectorySyncFailureAfterCheckpointRenamePoisonsWAL) { lbug::storage::WAL wal(databasePath, false /* readOnly */, false /* enableChecksums */, &vfs); lbug::storage::LocalWAL localWAL(*lbug::storage::MemoryManager::Get(*conn->getClientContext()), false /* enableChecksums */); - localWAL.logLoadExtension("dummy"); + localWAL.logLoadExtension("" /* main */, "dummy"); localWAL.logCommit(); uint64_t walCommitSequence = 0; ASSERT_NO_THROW(wal.logCommittedWAL(localWAL, conn->getClientContext(), walCommitSequence)); @@ -334,7 +334,7 @@ TEST_F(WalTest, DirectorySyncFailureAfterCheckpointRenamePoisonsWAL) { lbug::storage::LocalWAL laterWAL(*lbug::storage::MemoryManager::Get(*conn->getClientContext()), false /* enableChecksums */); - laterWAL.logLoadExtension("later"); + laterWAL.logLoadExtension("" /* main */, "later"); laterWAL.logCommit(); walCommitSequence = 0; EXPECT_THROW(wal.logCommittedWAL(laterWAL, conn->getClientContext(), walCommitSequence), @@ -366,7 +366,7 @@ TEST_F(WalTest, FrozenWALRemovalFailurePoisonsWALAfterDurableTruncate) { lbug::storage::LocalWAL localWAL(*lbug::storage::MemoryManager::Get(*conn->getClientContext()), false /* enableChecksums */); - localWAL.logLoadExtension("dummy"); + localWAL.logLoadExtension("" /* main */, "dummy"); localWAL.logCommit(); uint64_t walCommitSequence = 0; EXPECT_THROW(wal.logCommittedWAL(localWAL, conn->getClientContext(), walCommitSequence), From 4ac13845117dc6a90ac72c4c5508b3171c7e8e31 Mon Sep 17 00:00:00 2001 From: Aikins Laryea Date: Sun, 4 Oct 2026 21:08:38 +0000 Subject: [PATCH 03/17] fix(storage): silence unused tableID parameter in NodeTable::initScanState --- src/include/storage/table/node_table.h | 2 +- src/storage/table/node_table.cpp | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/src/include/storage/table/node_table.h b/src/include/storage/table/node_table.h index 25fc352b18..8c043c87b2 100644 --- a/src/include/storage/table/node_table.h +++ b/src/include/storage/table/node_table.h @@ -122,7 +122,7 @@ class LBUG_API NodeTable : public Table { void initScanState(transaction::Transaction* transaction, TableScanState& scanState, bool resetCachedBoundNodeIDs = true) const override; void initScanState(transaction::Transaction* transaction, TableScanState& scanState, - common::table_id_t tableID, common::offset_t startOffset) const; + [[maybe_unused]] common::table_id_t tableID, common::offset_t startOffset) const; // Virtual method for operator-level scan coordination initialization // Called once per scan operation (not per scan state) diff --git a/src/storage/table/node_table.cpp b/src/storage/table/node_table.cpp index d1f2e72c99..f44f5cad15 100644 --- a/src/storage/table/node_table.cpp +++ b/src/storage/table/node_table.cpp @@ -338,7 +338,7 @@ void NodeTable::initScanState(Transaction* transaction, TableScanState& scanStat } void NodeTable::initScanState(Transaction* transaction, TableScanState& scanState, - table_id_t tableID, offset_t startOffset) const { + [[maybe_unused]] table_id_t tableID, offset_t startOffset) const { if (transaction->isUnCommitted(*this, startOffset)) { scanState.source = TableScanSource::UNCOMMITTED; scanState.nodeGroupIdx = From 49bc83fc09af1baf44bb2e6e50fda5ebea1231e4 Mon Sep 17 00:00:00 2001 From: Aikins Laryea Date: Sun, 4 Oct 2026 22:28:36 +0000 Subject: [PATCH 04/17] fix(transaction): restore table-ID isUnCommitted/getLocalRowIdx overloads --- src/include/storage/local_storage/local_storage.h | 2 ++ src/include/transaction/transaction.h | 6 ++++++ src/storage/local_storage/local_storage.cpp | 14 ++++++++++++++ src/transaction/transaction.cpp | 14 ++++++++++++++ 4 files changed, 36 insertions(+) diff --git a/src/include/storage/local_storage/local_storage.h b/src/include/storage/local_storage/local_storage.h index 0a445fca71..732ca984e1 100644 --- a/src/include/storage/local_storage/local_storage.h +++ b/src/include/storage/local_storage/local_storage.h @@ -41,6 +41,8 @@ class LocalStorage { LocalTable* getOrCreateLocalTable(Table& table); // Return nullptr if no local table exists. LocalTable* getLocalTable(const Table& table) const; + // Return nullptr if no local table exists, or several owners hold the same table ID. + LocalTable* getLocalTable(common::table_id_t tableID) const; // Optimistic page allocation is scoped to one storage manager (each partition child has // its own data file and page manager). `sm == nullptr` selects the main database file. diff --git a/src/include/transaction/transaction.h b/src/include/transaction/transaction.h index bcc7df8044..1e2a3ee54f 100644 --- a/src/include/transaction/transaction.h +++ b/src/include/transaction/transaction.h @@ -130,10 +130,15 @@ class LBUG_API Transaction { storage::LocalStorage* getLocalStorage() const { return localStorage.get(); } LocalCacheManager& getLocalCacheManager() { return localCacheManager; } bool isUnCommitted(const storage::Table& table, common::offset_t nodeOffset) const; + bool isUnCommitted(common::table_id_t tableID, common::offset_t nodeOffset) const; common::row_idx_t getLocalRowIdx(const storage::Table& table, common::offset_t nodeOffset) const { return nodeOffset - getMinUncommittedNodeOffset(table); } + common::row_idx_t getLocalRowIdx(common::table_id_t tableID, + common::offset_t nodeOffset) const { + return nodeOffset - getMinUncommittedNodeOffset(tableID); + } common::offset_t getUncommittedOffset(const storage::Table& table, common::row_idx_t localRowIdx) const { return getMinUncommittedNodeOffset(table) + localRowIdx; @@ -158,6 +163,7 @@ class LBUG_API Transaction { private: common::offset_t getMinUncommittedNodeOffset(const storage::Table& table) const; + common::offset_t getMinUncommittedNodeOffset(common::table_id_t tableID) const; void recordCatalogChange(catalog::Catalog* catalog); private: diff --git a/src/storage/local_storage/local_storage.cpp b/src/storage/local_storage/local_storage.cpp index 500eb17959..f519da4b5f 100644 --- a/src/storage/local_storage/local_storage.cpp +++ b/src/storage/local_storage/local_storage.cpp @@ -75,6 +75,20 @@ LocalTable* LocalStorage::getLocalTable(const Table& table) const { return nullptr; } +LocalTable* LocalStorage::getLocalTable(common::table_id_t tableID) const { + LocalTable* result = nullptr; + for (const auto& [key, table] : tables) { + if (key.tableID != tableID) { + continue; + } + if (result != nullptr) { + return nullptr; + } + result = table.get(); + } + return result; +} + PageAllocator* LocalStorage::addOptimisticAllocator(StorageManager* sm) { auto* effectiveSM = sm != nullptr ? sm : StorageManager::Get(clientContext); auto* dataFH = effectiveSM->getDataFH(); diff --git a/src/transaction/transaction.cpp b/src/transaction/transaction.cpp index 37b71b610e..5a570a1ad9 100644 --- a/src/transaction/transaction.cpp +++ b/src/transaction/transaction.cpp @@ -153,6 +153,11 @@ bool Transaction::isUnCommitted(const storage::Table& table, common::offset_t no nodeOffset >= getMinUncommittedNodeOffset(table); } +bool Transaction::isUnCommitted(common::table_id_t tableID, common::offset_t nodeOffset) const { + return localStorage && localStorage->getLocalTable(tableID) && + nodeOffset >= getMinUncommittedNodeOffset(tableID); +} + void Transaction::pushCreateDropCatalogEntry(CatalogSet& catalogSet, CatalogEntry& catalogEntry, bool isInternal, bool skipLoggingToWAL) { undoBuffer->createCatalogEntry(catalogSet, catalogEntry); @@ -277,6 +282,15 @@ common::offset_t Transaction::getMinUncommittedNodeOffset(const storage::Table& return 0; } +common::offset_t Transaction::getMinUncommittedNodeOffset(common::table_id_t tableID) const { + if (localStorage) { + if (auto* localTable = localStorage->getLocalTable(tableID)) { + return localTable->cast().getStartOffset(); + } + } + return 0; +} + void Transaction::recordCatalogChange(catalog::Catalog* catalog) { if (catalog == nullptr) { return; From 7ad7fc362f4fb8fd009357c9e962bdde3fedb831 Mon Sep 17 00:00:00 2001 From: Aikins Laryea Date: Sun, 4 Oct 2026 23:02:20 +0000 Subject: [PATCH 05/17] fix(wal): translate UPDATE_SEQUENCE through replayed entry IDs --- .../wal/records/update_sequence_record_replay.cpp | 3 ++- test/transaction/checkpoint_test.cpp | 14 ++++++++++---- 2 files changed, 12 insertions(+), 5 deletions(-) diff --git a/src/storage/wal/records/update_sequence_record_replay.cpp b/src/storage/wal/records/update_sequence_record_replay.cpp index 898e65d9a4..3686b96043 100644 --- a/src/storage/wal/records/update_sequence_record_replay.cpp +++ b/src/storage/wal/records/update_sequence_record_replay.cpp @@ -12,7 +12,8 @@ namespace storage { void WALReplayer::replayUpdateSequenceRecord(const WALRecord& walRecord) const { auto& sequenceEntryRecord = walRecord.constCast(); - const auto sequenceID = sequenceEntryRecord.sequenceID; + const auto sequenceID = + getReplayedEntryID(CatalogEntryType::SEQUENCE_ENTRY, sequenceEntryRecord.sequenceID); const auto entry = Catalog::Get(clientContext) ->getSequenceEntry(transaction::Transaction::Get(clientContext), sequenceID); diff --git a/test/transaction/checkpoint_test.cpp b/test/transaction/checkpoint_test.cpp index 71e5f90b02..f9bcfa5884 100644 --- a/test/transaction/checkpoint_test.cpp +++ b/test/transaction/checkpoint_test.cpp @@ -2882,9 +2882,11 @@ TEST_F(FlakyCheckpointerTest, ConcurrentDropGraphDuringCommitKeepsCatalogsAlive) // materialize only the named graph and defer its WAL replay to the enclosing transaction's // commit instead of forcing a full catalog load inside the active recovery transaction. The // standalone session writes catalog-entry DDL only: a node table it creates has a plain -// schema that no longer matches the ANY-graph catalog the records replay into, and an -// UPDATE_SEQUENCE record it logs resolves by raw sequence ID against the standalone -// catalog's incompatible ID space, so only its by-name DDL records are assertable. +// schema that no longer matches the ANY-graph catalog the records replay into. Its +// UPDATE_SEQUENCE record addresses the sequence by the standalone catalog's ID space, which +// the ANY-graph infrastructure shifts here, so replay must translate it through the entry-ID +// mapping the CREATE record recorded: the standalone session's acknowledged nextval of 1 +// must not be handed out twice after the parent reopen. TEST_F(FlakyCheckpointerTest, StandaloneGraphFileCommitDoesNotWedgeMainWALReplay) { if (inMemMode || systemConfig->checkpointThreshold == 0) { GTEST_SKIP(); @@ -2916,10 +2918,14 @@ TEST_F(FlakyCheckpointerTest, StandaloneGraphFileCommitDoesNotWedgeMainWALReplay auto graphConnection = std::make_unique(graphDatabase.get()); auto tableResult = graphConnection->query("CREATE SEQUENCE carol_seq;"); ASSERT_TRUE(tableResult->isSuccess()) << tableResult->getErrorMessage(); + auto standaloneNextResult = graphConnection->query("RETURN nextval('carol_seq');"); + ASSERT_TRUE(standaloneNextResult->isSuccess()) << standaloneNextResult->getErrorMessage(); + ASSERT_EQ(standaloneNextResult->getNext()->getValue(0)->getValue(), 1); ASSERT_TRUE(std::filesystem::exists(graphWALPath)); // DDL results hold empty-schema factorized tables, which must be destroyed while their // database is still open. tableResult.reset(); + standaloneNextResult.reset(); graphConnection.reset(); graphDatabase.reset(); @@ -2934,7 +2940,7 @@ TEST_F(FlakyCheckpointerTest, StandaloneGraphFileCommitDoesNotWedgeMainWALReplay ASSERT_EQ(nextResult->getNext()->getValue(0)->getValue(), 2); nextResult = conn->query("RETURN nextval('carol_seq');"); ASSERT_TRUE(nextResult->isSuccess()) << nextResult->getErrorMessage(); - ASSERT_EQ(nextResult->getNext()->getValue(0)->getValue(), 1); + ASSERT_EQ(nextResult->getNext()->getValue(0)->getValue(), 2); EXPECT_FALSE(std::filesystem::exists(graphShadowPath)); } From 7773a99f0d2b9f49965af4fd1d0ba97facd824d2 Mon Sep 17 00:00:00 2001 From: Aikins Laryea Date: Mon, 5 Oct 2026 01:38:50 +0000 Subject: [PATCH 06/17] fix(storage): make the graph registry shared lock reentrant --- src/include/main/database_manager.h | 22 +++++++++++++++++++++ src/main/database_manager.cpp | 30 ++++++++++++++++++++++------- 2 files changed, 45 insertions(+), 7 deletions(-) diff --git a/src/include/main/database_manager.h b/src/include/main/database_manager.h index 40ced05737..426946cd81 100644 --- a/src/include/main/database_manager.h +++ b/src/include/main/database_manager.h @@ -1,11 +1,13 @@ #pragma once +#include #include #include #include #include #include "attached_database.h" +#include "common/copy_constructors.h" #include "storage/partition_storage_registry.h" namespace lbug { @@ -95,6 +97,26 @@ class DatabaseManager { std::string defaultDatabase; std::vector> graphs; mutable std::shared_mutex graphsMutex; + // withGraphCatalog callbacks re-enter registry lookups on the same thread (e.g. + // index initialization resolving the default graph catalog); std::shared_mutex is + // not recursive, so nested same-thread acquisitions are counted instead of + // reacquired. + inline static thread_local uint64_t graphsSharedHolds = 0; + + void acquireGraphsShared() const; + void releaseGraphsShared() const; + + class GraphsSharedLock { + public: + explicit GraphsSharedLock(const DatabaseManager& dbManager) : dbManager(dbManager) { + dbManager.acquireGraphsShared(); + } + DELETE_COPY_AND_MOVE(GraphsSharedLock); + ~GraphsSharedLock() { dbManager.releaseGraphsShared(); } + + private: + const DatabaseManager& dbManager; + }; // Owns the per-partition data files of partitioned node tables (phase-B per-partition // storage; see docs/partitioning.md 6b). storage::PartitionStorageRegistry partitionStorageRegistry; diff --git a/src/main/database_manager.cpp b/src/main/database_manager.cpp index 60fe791613..dc7c9d6a1d 100644 --- a/src/main/database_manager.cpp +++ b/src/main/database_manager.cpp @@ -307,7 +307,7 @@ void DatabaseManager::setDefaultGraph(const std::string& graphName) { defaultGraph = "main"; return; } - std::shared_lock lck{graphsMutex}; + GraphsSharedLock lck{*this}; for (auto& graph : graphs) { auto graphNameUpper = StringUtils::getUpper(graph->getCatalogName()); if (graphNameUpper == upperCaseName) { @@ -343,7 +343,7 @@ void DatabaseManager::loadGraphsFromCatalog(storage::MemoryManager* memoryManage auto graphEntries = mainCatalog->getGraphEntries(transaction); std::unordered_set loadedGraphNames; { - std::shared_lock lck{graphsMutex}; + GraphsSharedLock lck{*this}; loadedGraphNames.reserve(graphs.size()); for (const auto& graph : graphs) { loadedGraphNames.insert(StringUtils::getUpper(graph->getCatalogName())); @@ -514,7 +514,7 @@ void DatabaseManager::replayPendingGraphWALs(main::ClientContext* clientContext) bool DatabaseManager::hasGraph(const std::string& graphName) { auto upperCaseName = StringUtils::getUpper(graphName); - std::shared_lock lck{graphsMutex}; + GraphsSharedLock lck{*this}; for (auto& graph : graphs) { auto graphNameUpper = StringUtils::getUpper(graph->getCatalogName()); if (graphNameUpper == upperCaseName) { @@ -526,7 +526,7 @@ bool DatabaseManager::hasGraph(const std::string& graphName) { catalog::Catalog* DatabaseManager::getGraphCatalog(const std::string& graphName) { auto upperCaseName = StringUtils::getUpper(graphName); - std::shared_lock lck{graphsMutex}; + GraphsSharedLock lck{*this}; for (auto& graph : graphs) { auto graphNameUpper = StringUtils::getUpper(graph->getCatalogName()); if (graphNameUpper == upperCaseName) { @@ -536,10 +536,26 @@ catalog::Catalog* DatabaseManager::getGraphCatalog(const std::string& graphName) throw BinderException{std::format("No graph named {}.", graphName)}; } +void DatabaseManager::acquireGraphsShared() const { + if (graphsSharedHolds > 0) { + ++graphsSharedHolds; + return; + } + graphsMutex.lock_shared(); + ++graphsSharedHolds; +} + +void DatabaseManager::releaseGraphsShared() const { + --graphsSharedHolds; + if (graphsSharedHolds == 0) { + graphsMutex.unlock_shared(); + } +} + void DatabaseManager::withGraphCatalog(const std::string& graphName, const std::function& action) { auto upperCaseName = StringUtils::getUpper(graphName); - std::shared_lock lck{graphsMutex}; + GraphsSharedLock lck{*this}; for (auto& graph : graphs) { auto graphNameUpper = StringUtils::getUpper(graph->getCatalogName()); if (graphNameUpper == upperCaseName) { @@ -575,7 +591,7 @@ storage::StorageManager* DatabaseManager::getDefaultGraphStorageManager() const std::vector DatabaseManager::getGraphs() const { std::vector result; - std::shared_lock lck{graphsMutex}; + GraphsSharedLock lck{*this}; for (auto& graph : graphs) { result.push_back(graph.get()); } @@ -584,7 +600,7 @@ std::vector DatabaseManager::getGraphs() const { void DatabaseManager::bumpGraphCatalogVersions( const std::unordered_set& catalogs) { - std::shared_lock lck{graphsMutex}; + GraphsSharedLock lck{*this}; for (auto& graph : graphs) { if (catalogs.contains(graph.get())) { graph->incrementVersion(); From a4db7df2db5aa46111f7fe4176bd0f8c920e8766 Mon Sep 17 00:00:00 2001 From: Aikins Laryea Date: Mon, 5 Oct 2026 01:38:56 +0000 Subject: [PATCH 07/17] fix(wal): tag UPDATE_SEQUENCE records with the sequence name --- src/include/storage/wal/local_wal.h | 2 +- .../wal/record/update_sequence_record.h | 9 ++-- src/storage/wal/local_wal.cpp | 4 +- .../wal/records/update_sequence_record.cpp | 8 ++- .../records/update_sequence_record_replay.cpp | 27 +++++++--- .../records/update_sequence_record.tsp | 3 ++ src/transaction/transaction.cpp | 2 +- test/transaction/checkpoint_test.cpp | 50 +++++++++++++++++++ 8 files changed, 91 insertions(+), 14 deletions(-) diff --git a/src/include/storage/wal/local_wal.h b/src/include/storage/wal/local_wal.h index 0039d799f9..416653a77f 100644 --- a/src/include/storage/wal/local_wal.h +++ b/src/include/storage/wal/local_wal.h @@ -33,7 +33,7 @@ class LocalWAL { void logAlterCatalogEntryRecord(const std::string& ownerCatalogName, const binder::BoundAlterInfo* alterInfo); void logUpdateSequenceRecord(const std::string& ownerCatalogName, - common::sequence_id_t sequenceID, uint64_t kCount); + common::sequence_id_t sequenceID, uint64_t kCount, const std::string& sequenceName); void logTableInsertion(const std::string& ownerCatalogName, common::table_id_t tableID, common::TableType tableType, common::row_idx_t numRows, diff --git a/src/include/storage/wal/record/update_sequence_record.h b/src/include/storage/wal/record/update_sequence_record.h index 536cccdf6b..86b0cf1b11 100644 --- a/src/include/storage/wal/record/update_sequence_record.h +++ b/src/include/storage/wal/record/update_sequence_record.h @@ -7,6 +7,7 @@ #include #include +#include #include #include @@ -19,13 +20,15 @@ namespace storage { struct UpdateSequenceRecord final : WALRecord { common::sequence_id_t sequenceID; uint64_t kCount; + std::string sequenceName; UpdateSequenceRecord() : WALRecord{WALRecordType::UPDATE_SEQUENCE_RECORD}, sequenceID{0}, kCount{0} {} - UpdateSequenceRecord(common::sequence_id_t sequenceID, uint64_t kCount) - : WALRecord{WALRecordType::UPDATE_SEQUENCE_RECORD}, sequenceID{sequenceID}, kCount{kCount} { - } + UpdateSequenceRecord(common::sequence_id_t sequenceID, uint64_t kCount, + std::string sequenceName) + : WALRecord{WALRecordType::UPDATE_SEQUENCE_RECORD}, sequenceID{sequenceID}, kCount{kCount}, + sequenceName{sequenceName} {} void serialize(common::Serializer& serializer) const override; static std::unique_ptr deserialize(common::Deserializer& deserializer); diff --git a/src/storage/wal/local_wal.cpp b/src/storage/wal/local_wal.cpp index 65534dfeae..e4de77d191 100644 --- a/src/storage/wal/local_wal.cpp +++ b/src/storage/wal/local_wal.cpp @@ -100,8 +100,8 @@ void LocalWAL::logRelUpdate(const std::string& ownerCatalogName, table_id_t tabl } void LocalWAL::logUpdateSequenceRecord(const std::string& ownerCatalogName, - sequence_id_t sequenceID, uint64_t kCount) { - UpdateSequenceRecord walRecord(sequenceID, kCount); + sequence_id_t sequenceID, uint64_t kCount, const std::string& sequenceName) { + UpdateSequenceRecord walRecord(sequenceID, kCount, sequenceName); walRecord.ownerCatalogName = ownerCatalogName; addNewWALRecord(walRecord); } diff --git a/src/storage/wal/records/update_sequence_record.cpp b/src/storage/wal/records/update_sequence_record.cpp index 75fc1b1ce3..a16358673b 100644 --- a/src/storage/wal/records/update_sequence_record.cpp +++ b/src/storage/wal/records/update_sequence_record.cpp @@ -15,6 +15,7 @@ void UpdateSequenceRecord::serialize(Serializer& serializer) const { WALRecord::serialize(serializer); serializer.write(sequenceID); serializer.write(kCount); + serializer.write(sequenceName); } std::unique_ptr UpdateSequenceRecord::deserialize( @@ -27,8 +28,13 @@ std::unique_ptr UpdateSequenceRecord::deserialize( if (deserializer.hasRemainingData()) { deserializer.deserializeValue(kCount); } + std::string sequenceName{}; + if (deserializer.hasRemainingData()) { + deserializer.deserializeValue(sequenceName); + } - return std::make_unique(std::move(sequenceID), std::move(kCount)); + return std::make_unique(std::move(sequenceID), std::move(kCount), + std::move(sequenceName)); } } // namespace storage diff --git a/src/storage/wal/records/update_sequence_record_replay.cpp b/src/storage/wal/records/update_sequence_record_replay.cpp index 3686b96043..ba8d872a10 100644 --- a/src/storage/wal/records/update_sequence_record_replay.cpp +++ b/src/storage/wal/records/update_sequence_record_replay.cpp @@ -12,12 +12,27 @@ namespace storage { void WALReplayer::replayUpdateSequenceRecord(const WALRecord& walRecord) const { auto& sequenceEntryRecord = walRecord.constCast(); - const auto sequenceID = - getReplayedEntryID(CatalogEntryType::SEQUENCE_ENTRY, sequenceEntryRecord.sequenceID); - const auto entry = - Catalog::Get(clientContext) - ->getSequenceEntry(transaction::Transaction::Get(clientContext), sequenceID); - entry->nextKVal(transaction::Transaction::Get(clientContext), sequenceEntryRecord.kCount); + auto catalog = Catalog::Get(clientContext); + auto transaction = transaction::Transaction::Get(clientContext); + // sequenceName is empty in records written before the field existed. The name is + // logged precisely because implicit serial sequences have no create record, so their + // entry IDs shift between logging and replay (e.g. ANY-graph infrastructure + // materialized before a standalone graph WAL replays). + SequenceCatalogEntry* entry = nullptr; + if (!sequenceEntryRecord.sequenceName.empty()) { + for (auto* candidate : catalog->getSequenceEntries(transaction)) { + if (candidate->getName() == sequenceEntryRecord.sequenceName) { + entry = candidate; + break; + } + } + } + if (entry == nullptr) { + const auto sequenceID = + getReplayedEntryID(CatalogEntryType::SEQUENCE_ENTRY, sequenceEntryRecord.sequenceID); + entry = catalog->getSequenceEntry(transaction, sequenceID); + } + entry->nextKVal(transaction, sequenceEntryRecord.kCount); } } // namespace storage diff --git a/src/storage/wal/typespec/records/update_sequence_record.tsp b/src/storage/wal/typespec/records/update_sequence_record.tsp index 2c8c2044d5..43846a462d 100644 --- a/src/storage/wal/typespec/records/update_sequence_record.tsp +++ b/src/storage/wal/typespec/records/update_sequence_record.tsp @@ -1,9 +1,12 @@ // @debug_fields=false import "../common.tsp"; import ; +import ; +import ; import "common/types/types.h"; model UpdateSequenceRecord extends WalRecordBase { sequenceID: sequence_id_t; kCount: uint64_t; + sequenceName: string; } diff --git a/src/transaction/transaction.cpp b/src/transaction/transaction.cpp index 5a570a1ad9..dda6ea6707 100644 --- a/src/transaction/transaction.cpp +++ b/src/transaction/transaction.cpp @@ -253,7 +253,7 @@ void Transaction::pushSequenceChange(SequenceCatalogEntry* sequenceEntry, int64_ if (shouldLogToWAL()) { DASSERT(localWAL); localWAL->logUpdateSequenceRecord(sequenceEntry->getOwningCatalogName(), - sequenceEntry->getOID(), kCount); + sequenceEntry->getOID(), kCount, sequenceEntry->getName()); } } diff --git a/test/transaction/checkpoint_test.cpp b/test/transaction/checkpoint_test.cpp index f9bcfa5884..6227bce04c 100644 --- a/test/transaction/checkpoint_test.cpp +++ b/test/transaction/checkpoint_test.cpp @@ -2999,6 +2999,56 @@ TEST_F(FlakyCheckpointerTest, StandaloneSequenceDropRecreateDoesNotPoisonRecover ASSERT_EQ(carolResult->getNext()->getValue(0)->getValue(), 1); } +// A standalone session's CREATE NODE TABLE with a SERIAL column creates the column's +// sequence implicitly, and implicit sequence creates are deliberately not WAL-logged, +// so replay has no create record from which to translate the recorded sequence entry +// ID. The graph materializes the ANY-graph infrastructure before replaying the +// standalone WAL, shifting entry IDs, so the raw-ID fallback advances an infrastructure +// sequence instead: the table's own serial stays unadvanced, the next generated id +// repeats the committed id, and the insert fails on the duplicate primary key. The +// UPDATE_SEQUENCE record's name field resolves the sequence across the shift. +TEST_F(FlakyCheckpointerTest, StandaloneSerialTableSurvivesParentReopen) { + if (inMemMode || systemConfig->checkpointThreshold == 0) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH serial_graph ANY;")->isSuccess()); + conn.reset(); + database.reset(); + + const auto graphPath = StorageUtils::getGraphPath(databasePath, "serial_graph"); + auto graphConfig = *systemConfig; + graphConfig.autoCheckpoint = false; + graphConfig.forceCheckpointOnClose = false; + auto graphDatabase = std::make_unique(graphPath, graphConfig); + auto graphConnection = std::make_unique(graphDatabase.get()); + auto tableResult = + graphConnection->query("CREATE NODE TABLE P(id SERIAL, name STRING, PRIMARY KEY(id));"); + ASSERT_TRUE(tableResult->isSuccess()) << tableResult->getErrorMessage(); + auto insertResult = graphConnection->query("CREATE (:P {name: 'a'});"); + ASSERT_TRUE(insertResult->isSuccess()) << insertResult->getErrorMessage(); + tableResult.reset(); + insertResult.reset(); + graphConnection.reset(); + graphDatabase.reset(); + + createDBAndConn(); + auto openResult = conn->query("RETURN 1;"); + ASSERT_TRUE(openResult->isSuccess()) << openResult->getErrorMessage(); + ASSERT_EQ(openResult->getNext()->getValue(0)->getValue(), 1); + ASSERT_TRUE(conn->query("USE GRAPH serial_graph;")->isSuccess()); + insertResult = conn->query("CREATE (:P {name: 'b'});"); + ASSERT_TRUE(insertResult->isSuccess()) << insertResult->getErrorMessage(); + auto idsResult = conn->query("MATCH (p:P) RETURN p.id ORDER BY p.id;"); + ASSERT_TRUE(idsResult->isSuccess()) << idsResult->getErrorMessage(); + ASSERT_TRUE(idsResult->hasNext()); + ASSERT_EQ(idsResult->getNext()->getValue(0)->getValue(), 0); + ASSERT_TRUE(idsResult->hasNext()); + ASSERT_EQ(idsResult->getNext()->getValue(0)->getValue(), 1); + ASSERT_FALSE(idsResult->hasNext()); +} + // A standalone session's graph WAL addresses rel data by the per-direction physical // rel-table ID and binds a rel group through its endpoint node-table IDs, all recorded // in that session's plain catalog, which lacks the ANY-graph infrastructure entries From e54375bf43e398c152e092bbd8b6357799ca4f81 Mon Sep 17 00:00:00 2001 From: Aikins Laryea Date: Mon, 5 Oct 2026 01:39:03 +0000 Subject: [PATCH 08/17] fix(storage): skip undo version records for graphs dropped mid-transaction --- src/include/main/database_manager.h | 7 ++ src/include/storage/table/node_table.h | 2 +- src/include/storage/table/rel_table_data.h | 4 +- .../storage/table/version_record_handler.h | 10 ++ src/include/storage/undo_buffer.h | 10 +- src/main/database_manager.cpp | 17 ++- src/storage/table/node_table.cpp | 8 +- src/storage/table/rel_table_data.cpp | 15 ++- src/storage/undo_buffer.cpp | 102 +++++++++++------- src/transaction/transaction.cpp | 2 +- 10 files changed, 124 insertions(+), 53 deletions(-) diff --git a/src/include/main/database_manager.h b/src/include/main/database_manager.h index 426946cd81..268c4aa7cc 100644 --- a/src/include/main/database_manager.h +++ b/src/include/main/database_manager.h @@ -68,6 +68,13 @@ class DatabaseManager { // guards the lookup itself. void withGraphCatalog(const std::string& graphName, const std::function& action); + // Runs action with the graph registry's shared lock held if this catalog is still + // owned by the database manager, and returns true; a concurrent DROP GRAPH either + // waited (still owned) or already destroyed it (action skipped, returns false). + // Compares by identity, so a graph recreated under the same name cannot adopt the + // old catalog's outstanding work. + bool withGraphCatalogIfAlive(catalog::Catalog* catalog, + const std::function& action) const; catalog::Catalog* getDefaultGraphCatalog() const; catalog::Catalog* getReplayOwnerCatalog() const { return replayOwnerCatalog; } void setReplayOwnerCatalog(catalog::Catalog* catalog) { replayOwnerCatalog = catalog; } diff --git a/src/include/storage/table/node_table.h b/src/include/storage/table/node_table.h index 8c043c87b2..5752760d3a 100644 --- a/src/include/storage/table/node_table.h +++ b/src/include/storage/table/node_table.h @@ -98,7 +98,7 @@ struct IndexScanHelper { class NodeTableVersionRecordHandler final : public VersionRecordHandler { public: - explicit NodeTableVersionRecordHandler(NodeTable* table); + NodeTableVersionRecordHandler(NodeTable* table, catalog::Catalog* ownerCatalog); void applyFuncToChunkedGroups(version_record_handler_op_t func, common::node_group_idx_t nodeGroupIdx, common::row_idx_t startRow, diff --git a/src/include/storage/table/rel_table_data.h b/src/include/storage/table/rel_table_data.h index c5eb0f8020..f4cea6de8a 100644 --- a/src/include/storage/table/rel_table_data.h +++ b/src/include/storage/table/rel_table_data.h @@ -28,7 +28,7 @@ struct CSRHeaderColumns { class PersistentVersionRecordHandler final : public VersionRecordHandler { public: - explicit PersistentVersionRecordHandler(RelTableData* relTableData); + PersistentVersionRecordHandler(RelTableData* relTableData, catalog::Catalog* ownerCatalog); void applyFuncToChunkedGroups(version_record_handler_op_t func, common::node_group_idx_t nodeGroupIdx, common::row_idx_t startRow, @@ -42,7 +42,7 @@ class PersistentVersionRecordHandler final : public VersionRecordHandler { class InMemoryVersionRecordHandler final : public VersionRecordHandler { public: - explicit InMemoryVersionRecordHandler(RelTableData* relTableData); + InMemoryVersionRecordHandler(RelTableData* relTableData, catalog::Catalog* ownerCatalog); void applyFuncToChunkedGroups(version_record_handler_op_t func, common::node_group_idx_t nodeGroupIdx, common::row_idx_t startRow, diff --git a/src/include/storage/table/version_record_handler.h b/src/include/storage/table/version_record_handler.h index aca176e222..687f9f9c29 100644 --- a/src/include/storage/table/version_record_handler.h +++ b/src/include/storage/table/version_record_handler.h @@ -5,6 +5,10 @@ namespace lbug { +namespace catalog { +class Catalog; +} // namespace catalog + namespace storage { class ChunkedNodeGroup; @@ -15,14 +19,20 @@ using version_record_handler_op_t = void ( // Note: these handlers are not safe to use in multi-threaded contexts without external locking class VersionRecordHandler { public: + explicit VersionRecordHandler(catalog::Catalog* ownerCatalog) : ownerCatalog{ownerCatalog} {} virtual ~VersionRecordHandler() = default; + catalog::Catalog* getOwnerCatalog() const { return ownerCatalog; } + virtual void applyFuncToChunkedGroups(version_record_handler_op_t func, common::node_group_idx_t nodeGroupIdx, common::row_idx_t startRow, common::row_idx_t numRows, common::transaction_t commitTS) const = 0; virtual void rollbackInsert(main::ClientContext* context, common::node_group_idx_t nodeGroupIdx, common::row_idx_t startRow, common::row_idx_t numRows) const; + +private: + catalog::Catalog* ownerCatalog; }; } // namespace storage diff --git a/src/include/storage/undo_buffer.h b/src/include/storage/undo_buffer.h index 255e14789e..badb3498f0 100644 --- a/src/include/storage/undo_buffer.h +++ b/src/include/storage/undo_buffer.h @@ -90,7 +90,7 @@ class UndoBuffer { void createVectorUpdateInfo(UpdateInfo* updateInfo, common::idx_t vectorIdx, VectorUpdateInfo* vectorUpdateInfo, common::transaction_t version); - void commit(common::transaction_t commitTS) const; + void commit(main::ClientContext* context, common::transaction_t commitTS) const; void rollback(main::ClientContext* context) const; private: @@ -100,8 +100,8 @@ class UndoBuffer { common::row_idx_t numRows, const VersionRecordHandler* versionRecordHandler, common::node_group_idx_t nodeGroupIdx = 0); - static void commitRecord(UndoRecordType recordType, const uint8_t* record, - common::transaction_t commitTS); + static void commitRecord(main::ClientContext* context, UndoRecordType recordType, + const uint8_t* record, common::transaction_t commitTS); static void rollbackRecord(main::ClientContext* context, UndoRecordType recordType, const uint8_t* record); @@ -111,8 +111,8 @@ class UndoBuffer { static void commitSequenceEntry(uint8_t const* entry, common::transaction_t commitTS); static void rollbackSequenceEntry(uint8_t const* entry); - static void commitVersionInfo(UndoRecordType recordType, const uint8_t* record, - common::transaction_t commitTS); + static void commitVersionInfo(main::ClientContext* context, UndoRecordType recordType, + const uint8_t* record, common::transaction_t commitTS); static void rollbackVersionInfo(main::ClientContext* context, UndoRecordType recordType, const uint8_t* record); diff --git a/src/main/database_manager.cpp b/src/main/database_manager.cpp index dc7c9d6a1d..ca2ed5f6e0 100644 --- a/src/main/database_manager.cpp +++ b/src/main/database_manager.cpp @@ -566,12 +566,27 @@ void DatabaseManager::withGraphCatalog(const std::string& graphName, throw BinderException{std::format("No graph named {}.", graphName)}; } +bool DatabaseManager::withGraphCatalogIfAlive(catalog::Catalog* catalog, + const std::function& action) const { + if (catalog == nullptr) { + return false; + } + GraphsSharedLock lck{*this}; + for (auto& graph : graphs) { + if (graph.get() == catalog) { + action(); + return true; + } + } + return false; +} + catalog::Catalog* DatabaseManager::getDefaultGraphCatalog() const { if (defaultGraph == "" || defaultGraph == "main") { return nullptr; } auto upperCaseName = StringUtils::getUpper(defaultGraph); - std::shared_lock lck{graphsMutex}; + GraphsSharedLock lck{*this}; for (auto& graph : graphs) { auto graphNameUpper = StringUtils::getUpper(graph->getCatalogName()); if (graphNameUpper == upperCaseName) { diff --git a/src/storage/table/node_table.cpp b/src/storage/table/node_table.cpp index f44f5cad15..a1e11fd842 100644 --- a/src/storage/table/node_table.cpp +++ b/src/storage/table/node_table.cpp @@ -39,7 +39,9 @@ bool updatesAfterTableWrite(const Index& index) { } } // namespace -NodeTableVersionRecordHandler::NodeTableVersionRecordHandler(NodeTable* table) : table(table) {} +NodeTableVersionRecordHandler::NodeTableVersionRecordHandler(NodeTable* table, + catalog::Catalog* ownerCatalog) + : VersionRecordHandler(ownerCatalog), table(table) {} void NodeTableVersionRecordHandler::applyFuncToChunkedGroups(version_record_handler_op_t func, node_group_idx_t nodeGroupIdx, row_idx_t startRow, row_idx_t numRows, @@ -266,7 +268,9 @@ NodeTable::NodeTable(const StorageManager* storageManager, const NodeTableCatalogEntry* nodeTableEntry, MemoryManager* mm) : Table{nodeTableEntry, storageManager, mm}, pkColumnID{nodeTableEntry->getColumnID(nodeTableEntry->getPrimaryKeyName())}, - versionRecordHandler(this) { + versionRecordHandler(this, nodeTableEntry->getOwningCatalogName().empty() ? + nullptr : + nodeTableEntry->getOwningCatalog()) { auto* dataFH = storageManager->getDataFH(); auto& pageAllocator = *dataFH->getPageManager(); const auto maxColumnID = nodeTableEntry->getMaxColumnID(); diff --git a/src/storage/table/rel_table_data.cpp b/src/storage/table/rel_table_data.cpp index 4d3246946f..d7ea0683fa 100644 --- a/src/storage/table/rel_table_data.cpp +++ b/src/storage/table/rel_table_data.cpp @@ -17,8 +17,9 @@ using namespace lbug::transaction; namespace lbug { namespace storage { -PersistentVersionRecordHandler::PersistentVersionRecordHandler(RelTableData* relTableData) - : relTableData(relTableData) {} +PersistentVersionRecordHandler::PersistentVersionRecordHandler(RelTableData* relTableData, + catalog::Catalog* ownerCatalog) + : VersionRecordHandler(ownerCatalog), relTableData(relTableData) {} void PersistentVersionRecordHandler::applyFuncToChunkedGroups(version_record_handler_op_t func, node_group_idx_t nodeGroupIdx, row_idx_t startRow, row_idx_t numRows, @@ -37,8 +38,9 @@ void PersistentVersionRecordHandler::rollbackInsert(main::ClientContext* context relTableData->rollbackGroupCollectionInsert(numRows, true); } -InMemoryVersionRecordHandler::InMemoryVersionRecordHandler(RelTableData* relTableData) - : relTableData(relTableData) {} +InMemoryVersionRecordHandler::InMemoryVersionRecordHandler(RelTableData* relTableData, + catalog::Catalog* ownerCatalog) + : VersionRecordHandler(ownerCatalog), relTableData(relTableData) {} void InMemoryVersionRecordHandler::applyFuncToChunkedGroups(version_record_handler_op_t func, node_group_idx_t nodeGroupIdx, row_idx_t startRow, row_idx_t numRows, @@ -61,7 +63,10 @@ RelTableData::RelTableData(FileHandle* dataFH, MemoryManager* mm, ShadowFile* sh Table& table, RelDataDirection direction, table_id_t nbrTableID, bool enableCompression) : table{table}, mm{mm}, shadowFile{shadowFile}, enableCompression{enableCompression}, direction{direction}, multiplicity{relTableInfo.getMultiplicity(direction)}, - persistentVersionRecordHandler(this), inMemoryVersionRecordHandler(this) { + persistentVersionRecordHandler(this, + table.getOwnerCatalogName().empty() ? nullptr : relGroupEntry.getOwningCatalog()), + inMemoryVersionRecordHandler(this, + table.getOwnerCatalogName().empty() ? nullptr : relGroupEntry.getOwningCatalog()) { initCSRHeaderColumns(dataFH); initPropertyColumns(relGroupEntry, nbrTableID, dataFH); // default to using the persistent version record handler diff --git a/src/storage/undo_buffer.cpp b/src/storage/undo_buffer.cpp index 71633326f7..8f6faf1e3d 100644 --- a/src/storage/undo_buffer.cpp +++ b/src/storage/undo_buffer.cpp @@ -4,6 +4,8 @@ #include "catalog/catalog_entry/sequence_catalog_entry.h" #include "catalog/catalog_entry/table_catalog_entry.h" #include "catalog/catalog_set.h" +#include "main/client_context.h" +#include "main/database_manager.h" #include "storage/table/chunked_node_group.h" #include "storage/table/update_info.h" #include "storage/table/version_record_handler.h" @@ -43,6 +45,7 @@ struct VersionRecord { row_idx_t numRows; node_group_idx_t nodeGroupIdx; const VersionRecordHandler* versionRecordHandler; + catalog::Catalog* ownerCatalog; }; struct VectorUpdateRecord { @@ -131,8 +134,8 @@ void UndoBuffer::createVersionInfo(const UndoRecordType recordType, row_idx_t st const UndoRecordHeader recordHeader{recordType, sizeof(VersionRecord)}; *reinterpret_cast(buffer) = recordHeader; buffer += sizeof(UndoRecordHeader); - *reinterpret_cast(buffer) = - VersionRecord{startRow, numRows, nodeGroupIdx, versionRecordHandler}; + *reinterpret_cast(buffer) = VersionRecord{startRow, numRows, nodeGroupIdx, + versionRecordHandler, versionRecordHandler->getOwnerCatalog()}; } void UndoBuffer::createVectorUpdateInfo(UpdateInfo* updateInfo, const idx_t vectorIdx, @@ -161,10 +164,31 @@ uint8_t* UndoBuffer::createUndoRecord(const uint64_t size) { return res; } -void UndoBuffer::commit(transaction_t commitTS) const { +namespace { +// A version record stores the owning catalog captured when it was pushed. If that +// catalog is no longer in the graph registry, a concurrent DROP GRAPH has destroyed the +// table's storage, so the record's handler dangles and there is nothing to apply: the +// graph's data, the only reader of this version info, died with it. Liveness is decided +// by pointer identity, so a graph recreated under the same name cannot adopt an old +// transaction's records. The apply runs under the registry shared lock, so a DROP GRAPH +// that has not won the race yet cannot destroy the storage mid-apply. +void applyVersionInfoIfOwnerGraphAlive(ClientContext* context, catalog::Catalog* ownerCatalog, + const std::function& applyVersionInfo) { + if (ownerCatalog == nullptr) { + applyVersionInfo(); + return; + } + auto* dbManager = DatabaseManager::Get(*context); + if (dbManager != nullptr) { + dbManager->withGraphCatalogIfAlive(ownerCatalog, applyVersionInfo); + } +} +} // namespace + +void UndoBuffer::commit(ClientContext* context, transaction_t commitTS) const { UndoBufferIterator iterator{*this}; iterator.iterate([&](UndoRecordType entryType, uint8_t const* entry) { - commitRecord(entryType, entry, commitTS); + commitRecord(context, entryType, entry, commitTS); }); } @@ -175,8 +199,8 @@ void UndoBuffer::rollback(ClientContext* context) const { }); } -void UndoBuffer::commitRecord(UndoRecordType recordType, const uint8_t* record, - transaction_t commitTS) { +void UndoBuffer::commitRecord(ClientContext* context, UndoRecordType recordType, + const uint8_t* record, transaction_t commitTS) { switch (recordType) { case UndoRecordType::CATALOG_ENTRY: { commitCatalogEntryRecord(record, commitTS); @@ -186,7 +210,7 @@ void UndoBuffer::commitRecord(UndoRecordType recordType, const uint8_t* record, } break; case UndoRecordType::INSERT_INFO: case UndoRecordType::DELETE_INFO: { - commitVersionInfo(recordType, record, commitTS); + commitVersionInfo(context, recordType, record, commitTS); } break; case UndoRecordType::UPDATE_INFO: { commitVectorUpdateInfo(record, commitTS); @@ -203,22 +227,26 @@ void UndoBuffer::commitCatalogEntryRecord(const uint8_t* record, const transacti newCatalogEntry->setTimestamp(commitTS); } -void UndoBuffer::commitVersionInfo(UndoRecordType recordType, const uint8_t* record, - transaction_t commitTS) { +void UndoBuffer::commitVersionInfo(ClientContext* context, UndoRecordType recordType, + const uint8_t* record, transaction_t commitTS) { const auto& undoRecord = *reinterpret_cast(record); - switch (recordType) { - case UndoRecordType::INSERT_INFO: { - undoRecord.versionRecordHandler->applyFuncToChunkedGroups(&ChunkedNodeGroup::commitInsert, - undoRecord.nodeGroupIdx, undoRecord.startRow, undoRecord.numRows, commitTS); - } break; - case UndoRecordType::DELETE_INFO: { - undoRecord.versionRecordHandler->applyFuncToChunkedGroups(&ChunkedNodeGroup::commitDelete, - undoRecord.nodeGroupIdx, undoRecord.startRow, undoRecord.numRows, commitTS); - } break; - default: { - UNREACHABLE_CODE; - } - } + applyVersionInfoIfOwnerGraphAlive(context, undoRecord.ownerCatalog, [&]() { + switch (recordType) { + case UndoRecordType::INSERT_INFO: { + undoRecord.versionRecordHandler->applyFuncToChunkedGroups( + &ChunkedNodeGroup::commitInsert, undoRecord.nodeGroupIdx, undoRecord.startRow, + undoRecord.numRows, commitTS); + } break; + case UndoRecordType::DELETE_INFO: { + undoRecord.versionRecordHandler->applyFuncToChunkedGroups( + &ChunkedNodeGroup::commitDelete, undoRecord.nodeGroupIdx, undoRecord.startRow, + undoRecord.numRows, commitTS); + } break; + default: { + UNREACHABLE_CODE; + } + } + }); } void UndoBuffer::commitVectorUpdateInfo(const uint8_t* record, transaction_t commitTS) { @@ -283,20 +311,22 @@ void UndoBuffer::rollbackSequenceEntry(const uint8_t* entry) { void UndoBuffer::rollbackVersionInfo(ClientContext* context, UndoRecordType recordType, const uint8_t* record) { auto& undoRecord = *reinterpret_cast(record); - switch (recordType) { - case UndoRecordType::INSERT_INFO: { - undoRecord.versionRecordHandler->rollbackInsert(context, undoRecord.nodeGroupIdx, - undoRecord.startRow, undoRecord.numRows); - } break; - case UndoRecordType::DELETE_INFO: { - undoRecord.versionRecordHandler->applyFuncToChunkedGroups(&ChunkedNodeGroup::rollbackDelete, - undoRecord.nodeGroupIdx, undoRecord.startRow, undoRecord.numRows, - transaction::Transaction::Get(*context)->getCommitTS()); - } break; - default: { - UNREACHABLE_CODE; - } - } + applyVersionInfoIfOwnerGraphAlive(context, undoRecord.ownerCatalog, [&]() { + switch (recordType) { + case UndoRecordType::INSERT_INFO: { + undoRecord.versionRecordHandler->rollbackInsert(context, undoRecord.nodeGroupIdx, + undoRecord.startRow, undoRecord.numRows); + } break; + case UndoRecordType::DELETE_INFO: { + undoRecord.versionRecordHandler->applyFuncToChunkedGroups( + &ChunkedNodeGroup::rollbackDelete, undoRecord.nodeGroupIdx, undoRecord.startRow, + undoRecord.numRows, transaction::Transaction::Get(*context)->getCommitTS()); + } break; + default: { + UNREACHABLE_CODE; + } + } + }); } void UndoBuffer::rollbackVectorUpdateInfo(const uint8_t* record) { diff --git a/src/transaction/transaction.cpp b/src/transaction/transaction.cpp index dda6ea6707..519b48c803 100644 --- a/src/transaction/transaction.cpp +++ b/src/transaction/transaction.cpp @@ -89,7 +89,7 @@ void Transaction::publishCommit() { throw common::RuntimeException{"Cannot publish commit with an invalid commit timestamp."}; } localStorage->commit(); - undoBuffer->commit(commitTS); + undoBuffer->commit(clientContext, commitTS); { std::lock_guard lck{changedCatalogsMutex}; if (!changedCatalogs.empty()) { From 5a29617bc3e18e660c0531f0c4e45d487f2913db Mon Sep 17 00:00:00 2001 From: Aikins Laryea Date: Mon, 5 Oct 2026 02:40:29 +0000 Subject: [PATCH 09/17] fix(wal): encode named UPDATE_SEQUENCE records as a distinct record type --- .../wal/record/update_sequence_named_record.h | 39 +++++++++++++ .../wal/record/update_sequence_record.h | 8 +-- .../storage/wal/record/wal_record_base.h | 1 + src/include/storage/wal/wal_record.h | 1 + src/include/storage/wal/wal_replayer.h | 3 + src/storage/wal/CMakeLists.txt | 1 + src/storage/wal/local_wal.cpp | 2 +- .../records/update_sequence_named_record.cpp | 41 ++++++++++++++ .../wal/records/update_sequence_record.cpp | 8 +-- .../records/update_sequence_record_replay.cpp | 39 ++++++++----- .../records/update_sequence_named_record.tsp | 13 +++++ .../records/update_sequence_record.tsp | 1 - src/storage/wal/wal_record.cpp | 3 + src/storage/wal/wal_replayer.cpp | 3 + test/transaction/wal_test.cpp | 55 +++++++++++++++++++ tools/wal_dump/main.cpp | 10 ++++ 16 files changed, 200 insertions(+), 28 deletions(-) create mode 100644 src/include/storage/wal/record/update_sequence_named_record.h create mode 100644 src/storage/wal/records/update_sequence_named_record.cpp create mode 100644 src/storage/wal/typespec/records/update_sequence_named_record.tsp diff --git a/src/include/storage/wal/record/update_sequence_named_record.h b/src/include/storage/wal/record/update_sequence_named_record.h new file mode 100644 index 0000000000..7048ac3801 --- /dev/null +++ b/src/include/storage/wal/record/update_sequence_named_record.h @@ -0,0 +1,39 @@ +//===----------------------------------------------------------------------===// +// This file is automatically generated by scripts/generate_wal_typespec.py. +// Do not edit this file manually. +//===----------------------------------------------------------------------===// + +#pragma once + +#include +#include +#include +#include +#include + +#include "common/types/types.h" +#include "storage/wal/record/wal_record_base.h" + +namespace lbug { +namespace storage { + +struct UpdateSequenceNamedRecord final : WALRecord { + common::sequence_id_t sequenceID; + uint64_t kCount; + std::string sequenceName; + + UpdateSequenceNamedRecord() + : WALRecord{WALRecordType::UPDATE_SEQUENCE_NAMED_RECORD}, sequenceID{0}, kCount{0} {} + + UpdateSequenceNamedRecord(common::sequence_id_t sequenceID, uint64_t kCount, + std::string sequenceName) + : WALRecord{WALRecordType::UPDATE_SEQUENCE_NAMED_RECORD}, sequenceID{sequenceID}, + kCount{kCount}, sequenceName{sequenceName} {} + + void serialize(common::Serializer& serializer) const override; + static std::unique_ptr deserialize( + common::Deserializer& deserializer); +}; + +} // namespace storage +} // namespace lbug diff --git a/src/include/storage/wal/record/update_sequence_record.h b/src/include/storage/wal/record/update_sequence_record.h index 86b0cf1b11..95559e8964 100644 --- a/src/include/storage/wal/record/update_sequence_record.h +++ b/src/include/storage/wal/record/update_sequence_record.h @@ -20,15 +20,13 @@ namespace storage { struct UpdateSequenceRecord final : WALRecord { common::sequence_id_t sequenceID; uint64_t kCount; - std::string sequenceName; UpdateSequenceRecord() : WALRecord{WALRecordType::UPDATE_SEQUENCE_RECORD}, sequenceID{0}, kCount{0} {} - UpdateSequenceRecord(common::sequence_id_t sequenceID, uint64_t kCount, - std::string sequenceName) - : WALRecord{WALRecordType::UPDATE_SEQUENCE_RECORD}, sequenceID{sequenceID}, kCount{kCount}, - sequenceName{sequenceName} {} + UpdateSequenceRecord(common::sequence_id_t sequenceID, uint64_t kCount) + : WALRecord{WALRecordType::UPDATE_SEQUENCE_RECORD}, sequenceID{sequenceID}, kCount{kCount} { + } void serialize(common::Serializer& serializer) const override; static std::unique_ptr deserialize(common::Deserializer& deserializer); diff --git a/src/include/storage/wal/record/wal_record_base.h b/src/include/storage/wal/record/wal_record_base.h index e0f55b741a..0ea29543b4 100644 --- a/src/include/storage/wal/record/wal_record_base.h +++ b/src/include/storage/wal/record/wal_record_base.h @@ -32,6 +32,7 @@ enum class WALRecordType : uint8_t { DROP_CATALOG_ENTRY_RECORD = 16, ALTER_TABLE_ENTRY_RECORD = 17, UPDATE_SEQUENCE_RECORD = 18, + UPDATE_SEQUENCE_NAMED_RECORD = 19, TABLE_INSERTION_RECORD = 30, NODE_DELETION_RECORD = 31, NODE_UPDATE_RECORD = 32, diff --git a/src/include/storage/wal/wal_record.h b/src/include/storage/wal/wal_record.h index 656012598f..40782526cf 100644 --- a/src/include/storage/wal/wal_record.h +++ b/src/include/storage/wal/wal_record.h @@ -15,5 +15,6 @@ #include "storage/wal/record/rel_detach_delete_record.h" #include "storage/wal/record/rel_update_record.h" #include "storage/wal/record/table_insertion_record.h" +#include "storage/wal/record/update_sequence_named_record.h" #include "storage/wal/record/update_sequence_record.h" #include "storage/wal/record/wal_record_base.h" diff --git a/src/include/storage/wal/wal_replayer.h b/src/include/storage/wal/wal_replayer.h index d2e028c374..c942f1ae0a 100644 --- a/src/include/storage/wal/wal_replayer.h +++ b/src/include/storage/wal/wal_replayer.h @@ -69,6 +69,9 @@ class WALReplayer { void replayRelUpdateRecord(const WALRecord& walRecord) const; void replayCopyTableRecord(const WALRecord& walRecord) const; void replayUpdateSequenceRecord(const WALRecord& walRecord) const; + void replayUpdateSequenceNamedRecord(const WALRecord& walRecord) const; + void replaySequenceRecord(common::sequence_id_t sequenceID, uint64_t kCount, + const std::string& sequenceName) const; void replayNodeTableInsertRecord(const WALRecord& walRecord) const; void replayRelTableInsertRecord(const WALRecord& walRecord) const; diff --git a/src/storage/wal/CMakeLists.txt b/src/storage/wal/CMakeLists.txt index 5c5cf135fe..b17d46b8ee 100644 --- a/src/storage/wal/CMakeLists.txt +++ b/src/storage/wal/CMakeLists.txt @@ -33,6 +33,7 @@ add_library(lbug_storage_wal records/alter_table_entry_record.cpp records/copy_table_record.cpp records/update_sequence_record.cpp + records/update_sequence_named_record.cpp records/table_insertion_record.cpp records/node_deletion_record.cpp records/node_update_record.cpp diff --git a/src/storage/wal/local_wal.cpp b/src/storage/wal/local_wal.cpp index e4de77d191..47853aefe5 100644 --- a/src/storage/wal/local_wal.cpp +++ b/src/storage/wal/local_wal.cpp @@ -101,7 +101,7 @@ void LocalWAL::logRelUpdate(const std::string& ownerCatalogName, table_id_t tabl void LocalWAL::logUpdateSequenceRecord(const std::string& ownerCatalogName, sequence_id_t sequenceID, uint64_t kCount, const std::string& sequenceName) { - UpdateSequenceRecord walRecord(sequenceID, kCount, sequenceName); + UpdateSequenceNamedRecord walRecord(sequenceID, kCount, sequenceName); walRecord.ownerCatalogName = ownerCatalogName; addNewWALRecord(walRecord); } diff --git a/src/storage/wal/records/update_sequence_named_record.cpp b/src/storage/wal/records/update_sequence_named_record.cpp new file mode 100644 index 0000000000..4ed1a1bfd7 --- /dev/null +++ b/src/storage/wal/records/update_sequence_named_record.cpp @@ -0,0 +1,41 @@ +//===----------------------------------------------------------------------===// +// This file is automatically generated by scripts/generate_wal_typespec.py. +// Do not edit this file manually. +//===----------------------------------------------------------------------===// + +#include "common/serializer/deserializer.h" +#include "common/serializer/serializer.h" +#include "storage/wal/wal_record.h" + +using namespace lbug::common; +namespace lbug { +namespace storage { + +void UpdateSequenceNamedRecord::serialize(Serializer& serializer) const { + WALRecord::serialize(serializer); + serializer.write(sequenceID); + serializer.write(kCount); + serializer.write(sequenceName); +} + +std::unique_ptr UpdateSequenceNamedRecord::deserialize( + Deserializer& deserializer) { + sequence_id_t sequenceID = 0; + if (deserializer.hasRemainingData()) { + deserializer.deserializeValue(sequenceID); + } + uint64_t kCount = 0; + if (deserializer.hasRemainingData()) { + deserializer.deserializeValue(kCount); + } + std::string sequenceName{}; + if (deserializer.hasRemainingData()) { + deserializer.deserializeValue(sequenceName); + } + + return std::make_unique(std::move(sequenceID), std::move(kCount), + std::move(sequenceName)); +} + +} // namespace storage +} // namespace lbug diff --git a/src/storage/wal/records/update_sequence_record.cpp b/src/storage/wal/records/update_sequence_record.cpp index a16358673b..75fc1b1ce3 100644 --- a/src/storage/wal/records/update_sequence_record.cpp +++ b/src/storage/wal/records/update_sequence_record.cpp @@ -15,7 +15,6 @@ void UpdateSequenceRecord::serialize(Serializer& serializer) const { WALRecord::serialize(serializer); serializer.write(sequenceID); serializer.write(kCount); - serializer.write(sequenceName); } std::unique_ptr UpdateSequenceRecord::deserialize( @@ -28,13 +27,8 @@ std::unique_ptr UpdateSequenceRecord::deserialize( if (deserializer.hasRemainingData()) { deserializer.deserializeValue(kCount); } - std::string sequenceName{}; - if (deserializer.hasRemainingData()) { - deserializer.deserializeValue(sequenceName); - } - return std::make_unique(std::move(sequenceID), std::move(kCount), - std::move(sequenceName)); + return std::make_unique(std::move(sequenceID), std::move(kCount)); } } // namespace storage diff --git a/src/storage/wal/records/update_sequence_record_replay.cpp b/src/storage/wal/records/update_sequence_record_replay.cpp index ba8d872a10..dde44c4cdf 100644 --- a/src/storage/wal/records/update_sequence_record_replay.cpp +++ b/src/storage/wal/records/update_sequence_record_replay.cpp @@ -12,27 +12,38 @@ namespace storage { void WALReplayer::replayUpdateSequenceRecord(const WALRecord& walRecord) const { auto& sequenceEntryRecord = walRecord.constCast(); + replaySequenceRecord(sequenceEntryRecord.sequenceID, sequenceEntryRecord.kCount, ""); +} + +void WALReplayer::replayUpdateSequenceNamedRecord(const WALRecord& walRecord) const { + auto& sequenceEntryRecord = walRecord.constCast(); + replaySequenceRecord(sequenceEntryRecord.sequenceID, sequenceEntryRecord.kCount, + sequenceEntryRecord.sequenceName); +} + +void WALReplayer::replaySequenceRecord(common::sequence_id_t sequenceID, uint64_t kCount, + const std::string& sequenceName) const { auto catalog = Catalog::Get(clientContext); auto transaction = transaction::Transaction::Get(clientContext); - // sequenceName is empty in records written before the field existed. The name is - // logged precisely because implicit serial sequences have no create record, so their - // entry IDs shift between logging and replay (e.g. ANY-graph infrastructure - // materialized before a standalone graph WAL replays). + // The named variant logs the sequence name precisely because implicit serial + // sequences have no create record, so their entry IDs shift between logging and + // replay (e.g. ANY-graph infrastructure materialized before a standalone graph WAL + // replays). Legacy UPDATE_SEQUENCE records carry no name and always resolve by + // translated entry ID. SequenceCatalogEntry* entry = nullptr; - if (!sequenceEntryRecord.sequenceName.empty()) { - for (auto* candidate : catalog->getSequenceEntries(transaction)) { - if (candidate->getName() == sequenceEntryRecord.sequenceName) { - entry = candidate; - break; - } + if (!sequenceName.empty() && catalog->containsSequence(transaction, sequenceName)) { + entry = catalog->getSequenceEntry(transaction, sequenceName, false); + // The catalog's name index is case-insensitive; accept only an exact match. + if (entry->getName() != sequenceName) { + entry = nullptr; } } if (entry == nullptr) { - const auto sequenceID = - getReplayedEntryID(CatalogEntryType::SEQUENCE_ENTRY, sequenceEntryRecord.sequenceID); - entry = catalog->getSequenceEntry(transaction, sequenceID); + const auto replayedSequenceID = + getReplayedEntryID(CatalogEntryType::SEQUENCE_ENTRY, sequenceID); + entry = catalog->getSequenceEntry(transaction, replayedSequenceID); } - entry->nextKVal(transaction, sequenceEntryRecord.kCount); + entry->nextKVal(transaction, kCount); } } // namespace storage diff --git a/src/storage/wal/typespec/records/update_sequence_named_record.tsp b/src/storage/wal/typespec/records/update_sequence_named_record.tsp new file mode 100644 index 0000000000..1931328731 --- /dev/null +++ b/src/storage/wal/typespec/records/update_sequence_named_record.tsp @@ -0,0 +1,13 @@ +// @record_type=UPDATE_SEQUENCE_NAMED_RECORD +// @debug_fields=false +import "../common.tsp"; +import ; +import ; +import ; +import "common/types/types.h"; + +model UpdateSequenceNamedRecord extends WalRecordBase { + sequenceID: sequence_id_t; + kCount: uint64_t; + sequenceName: string; +} diff --git a/src/storage/wal/typespec/records/update_sequence_record.tsp b/src/storage/wal/typespec/records/update_sequence_record.tsp index 43846a462d..e65314dfc9 100644 --- a/src/storage/wal/typespec/records/update_sequence_record.tsp +++ b/src/storage/wal/typespec/records/update_sequence_record.tsp @@ -8,5 +8,4 @@ import "common/types/types.h"; model UpdateSequenceRecord extends WalRecordBase { sequenceID: sequence_id_t; kCount: uint64_t; - sequenceName: string; } diff --git a/src/storage/wal/wal_record.cpp b/src/storage/wal/wal_record.cpp index 58f69a8dec..665bfcbcc5 100644 --- a/src/storage/wal/wal_record.cpp +++ b/src/storage/wal/wal_record.cpp @@ -85,6 +85,9 @@ std::unique_ptr WALRecord::deserialize(Deserializer& deserializer, case WALRecordType::UPDATE_SEQUENCE_RECORD: { walRecord = UpdateSequenceRecord::deserialize(deserializer); } break; + case WALRecordType::UPDATE_SEQUENCE_NAMED_RECORD: { + walRecord = UpdateSequenceNamedRecord::deserialize(deserializer); + } break; case WALRecordType::LOAD_EXTENSION_RECORD: { walRecord = LoadExtensionRecord::deserialize(deserializer); } break; diff --git a/src/storage/wal/wal_replayer.cpp b/src/storage/wal/wal_replayer.cpp index 013d99b6ab..1e39da39b6 100644 --- a/src/storage/wal/wal_replayer.cpp +++ b/src/storage/wal/wal_replayer.cpp @@ -892,6 +892,9 @@ void WALReplayer::replayWALRecord(WALRecord& walRecord) const { case WALRecordType::UPDATE_SEQUENCE_RECORD: { replayUpdateSequenceRecord(walRecord); } break; + case WALRecordType::UPDATE_SEQUENCE_NAMED_RECORD: { + replayUpdateSequenceNamedRecord(walRecord); + } break; case WALRecordType::LOAD_EXTENSION_RECORD: { replayLoadExtensionRecord(walRecord); } break; diff --git a/test/transaction/wal_test.cpp b/test/transaction/wal_test.cpp index 9ef4313885..11d26f69e1 100644 --- a/test/transaction/wal_test.cpp +++ b/test/transaction/wal_test.cpp @@ -396,6 +396,61 @@ TEST_F(WalTest, WALRecordDeserializeSkipsUnknownTrailingBytes) { EXPECT_TRUE(deserializer.finished()); } +// A pre-named-format binary wrote UPDATE_SEQUENCE records as [sequenceID][kCount] +// followed by the ownerCatalogName trailer that every record carries. The current +// reader must keep decoding that layout: the fields survive and the owner trailer is +// still routed as ownerCatalogName, never misread as a sequence name. +TEST_F(WalTest, WALRecordDeserializeKeepsLegacyUpdateSequenceOwnerIntact) { + auto recordBuffer = std::make_shared(); + Serializer recordSerializer{recordBuffer}; + lbug::storage::UpdateSequenceRecord record{7, 42}; + record.serialize(recordSerializer); + recordSerializer.writeDebuggingInfo("ownerCatalogName"); + recordSerializer.write("mygraph"); + + auto walBuffer = std::make_shared(); + Serializer walSerializer{walBuffer}; + walSerializer.write(recordBuffer->getSize()); + walSerializer.write(recordBuffer->getBlobData(), recordBuffer->getSize()); + + auto walData = walBuffer->getData(); + Deserializer deserializer{std::make_unique(walData.data.get(), walData.size)}; + auto deserialized = + lbug::storage::WALRecord::deserialize(deserializer, *conn->getClientContext()); + + ASSERT_EQ(deserialized->type, lbug::storage::WALRecordType::UPDATE_SEQUENCE_RECORD); + const auto& sequenceRecord = deserialized->constCast(); + EXPECT_EQ(sequenceRecord.sequenceID, 7); + EXPECT_EQ(sequenceRecord.kCount, 42); + EXPECT_EQ(deserialized->ownerCatalogName, "mygraph"); + EXPECT_TRUE(deserializer.finished()); +} + +// The named variant travels under its own record type, so its sequenceName field can +// never be confused with the legacy layout, and the owner trailer stays last. +TEST_F(WalTest, WALRecordDeserializeRoundTripsNamedUpdateSequenceRecord) { + lbug::storage::UpdateSequenceNamedRecord record{5, 3, "person_id_serial"}; + record.ownerCatalogName = "mygraph"; + + auto walBuffer = std::make_shared(); + Serializer walSerializer{walBuffer}; + lbug::storage::WALRecord::serializeWithLength(walSerializer, record); + + auto walData = walBuffer->getData(); + Deserializer deserializer{std::make_unique(walData.data.get(), walData.size)}; + auto deserialized = + lbug::storage::WALRecord::deserialize(deserializer, *conn->getClientContext()); + + ASSERT_EQ(deserialized->type, lbug::storage::WALRecordType::UPDATE_SEQUENCE_NAMED_RECORD); + const auto& sequenceRecord = + deserialized->constCast(); + EXPECT_EQ(sequenceRecord.sequenceID, 5); + EXPECT_EQ(sequenceRecord.kCount, 3); + EXPECT_EQ(sequenceRecord.sequenceName, "person_id_serial"); + EXPECT_EQ(deserialized->ownerCatalogName, "mygraph"); + EXPECT_TRUE(deserializer.finished()); +} + // Simulates the scenario where the declared record length is smaller than the actual serialized // record. This happens when an older writer (e.g. v42) wrote a WAL record with fewer fields than // the current reader (e.g. v43) expects. The deserializer must gracefully handle the size mismatch diff --git a/tools/wal_dump/main.cpp b/tools/wal_dump/main.cpp index d618dbbfa4..077dfcd93d 100644 --- a/tools/wal_dump/main.cpp +++ b/tools/wal_dump/main.cpp @@ -70,6 +70,8 @@ static std::string walRecordTypeToString(WALRecordType type) { return "ALTER_TABLE_ENTRY_RECORD"; case WALRecordType::UPDATE_SEQUENCE_RECORD: return "UPDATE_SEQUENCE_RECORD"; + case WALRecordType::UPDATE_SEQUENCE_NAMED_RECORD: + return "UPDATE_SEQUENCE_NAMED_RECORD"; case WALRecordType::TABLE_INSERTION_RECORD: return "TABLE_INSERTION_RECORD"; case WALRecordType::NODE_DELETION_RECORD: @@ -172,6 +174,14 @@ static void dumpRecord(const WALRecord& record) { std::cout << " KCount: " << seqRecord.kCount << "\n"; break; } + case WALRecordType::UPDATE_SEQUENCE_NAMED_RECORD: { + const auto& seqRecord = record.constCast(); + std::cout << " Type: UPDATE_SEQUENCE_NAMED\n"; + std::cout << " SequenceID: " << seqRecord.sequenceID << "\n"; + std::cout << " KCount: " << seqRecord.kCount << "\n"; + std::cout << " SequenceName: " << seqRecord.sequenceName << "\n"; + break; + } case WALRecordType::LOAD_EXTENSION_RECORD: { const auto& extRecord = record.constCast(); std::cout << " Type: LOAD_EXTENSION\n"; From aeb4185fcdfe5b9b5fba530e7c20422aa552d367 Mon Sep 17 00:00:00 2001 From: Aikins Laryea Date: Mon, 5 Oct 2026 02:40:44 +0000 Subject: [PATCH 10/17] fix(storage): key graph registry recursion and liveness checks per manager --- src/include/main/database_manager.h | 10 ++++++++-- src/main/database_manager.cpp | 23 +++++++++++++---------- 2 files changed, 21 insertions(+), 12 deletions(-) diff --git a/src/include/main/database_manager.h b/src/include/main/database_manager.h index 268c4aa7cc..c494f81d5f 100644 --- a/src/include/main/database_manager.h +++ b/src/include/main/database_manager.h @@ -4,6 +4,7 @@ #include #include #include +#include #include #include "attached_database.h" @@ -103,12 +104,17 @@ class DatabaseManager { std::vector> attachedDatabases; std::string defaultDatabase; std::vector> graphs; + // Mirror of graphs keyed by catalog identity for O(1) liveness checks; mutated only + // under graphsMutex alongside graphs. + std::unordered_set graphIdentities; mutable std::shared_mutex graphsMutex; // withGraphCatalog callbacks re-enter registry lookups on the same thread (e.g. // index initialization resolving the default graph catalog); std::shared_mutex is // not recursive, so nested same-thread acquisitions are counted instead of - // reacquired. - inline static thread_local uint64_t graphsSharedHolds = 0; + // reacquired. Counts are kept per manager: a callback nested over another + // manager's registry must still take that manager's lock. + inline static thread_local std::unordered_map + graphsSharedHolds; void acquireGraphsShared() const; void releaseGraphsShared() const; diff --git a/src/main/database_manager.cpp b/src/main/database_manager.cpp index ca2ed5f6e0..a59e405686 100644 --- a/src/main/database_manager.cpp +++ b/src/main/database_manager.cpp @@ -221,6 +221,7 @@ void DatabaseManager::createGraph(const std::string& graphName, { std::unique_lock lck{graphsMutex}; + graphIdentities.insert(catalog.get()); graphs.push_back(std::move(catalog)); } // NOTE: Do NOT set defaultGraph here. Setting defaultGraph before the transaction @@ -272,6 +273,7 @@ void DatabaseManager::dropGraph(const std::string& graphName, main::ClientContex if (hasAttachedDatabase(graphName)) { detachDatabase(graphName); } + graphIdentities.erase(it->get()); graphs.erase(it); lck.unlock(); @@ -429,6 +431,7 @@ bool DatabaseManager::loadGraph(main::ClientContext* clientContext, auto* graphStorageManager = catalog->getStorageManager(); { std::unique_lock lck{graphsMutex}; + graphIdentities.insert(catalog.get()); graphs.push_back(std::move(catalog)); } @@ -537,17 +540,19 @@ catalog::Catalog* DatabaseManager::getGraphCatalog(const std::string& graphName) } void DatabaseManager::acquireGraphsShared() const { - if (graphsSharedHolds > 0) { - ++graphsSharedHolds; + auto& holds = graphsSharedHolds[this]; + if (holds > 0) { + ++holds; return; } graphsMutex.lock_shared(); - ++graphsSharedHolds; + ++holds; } void DatabaseManager::releaseGraphsShared() const { - --graphsSharedHolds; - if (graphsSharedHolds == 0) { + auto& holds = graphsSharedHolds.at(this); + --holds; + if (holds == 0) { graphsMutex.unlock_shared(); } } @@ -572,11 +577,9 @@ bool DatabaseManager::withGraphCatalogIfAlive(catalog::Catalog* catalog, return false; } GraphsSharedLock lck{*this}; - for (auto& graph : graphs) { - if (graph.get() == catalog) { - action(); - return true; - } + if (graphIdentities.contains(catalog)) { + action(); + return true; } return false; } From 1c0c08c857e66606d0dd0c3451f409840fe2cc07 Mon Sep 17 00:00:00 2001 From: Aikins Laryea Date: Mon, 5 Oct 2026 04:01:22 +0000 Subject: [PATCH 11/17] fix(wal): cover legacy sequence record replay with hand-written WAL bytes --- test/transaction/checkpoint_test.cpp | 84 ++++++++++++++++++++++++++++ test/transaction/wal_test.cpp | 9 ++- 2 files changed, 91 insertions(+), 2 deletions(-) diff --git a/test/transaction/checkpoint_test.cpp b/test/transaction/checkpoint_test.cpp index 6227bce04c..8b2a65e195 100644 --- a/test/transaction/checkpoint_test.cpp +++ b/test/transaction/checkpoint_test.cpp @@ -8,6 +8,7 @@ #include #include #include +#include #include #include #include @@ -15,6 +16,7 @@ #include #include "api_test/private_api_test.h" +#include "binder/ddl/bound_create_sequence_info.h" #include "catalog/catalog.h" #include "catalog/catalog_entry/rel_group_catalog_entry.h" #include "catalog/catalog_entry/sequence_catalog_entry.h" @@ -25,6 +27,7 @@ #include "common/serializer/buffer_reader.h" #include "common/serializer/buffer_writer.h" #include "common/serializer/deserializer.h" +#include "common/serializer/in_mem_file_writer.h" #include "common/serializer/serializer.h" #include "common/vector/value_vector.h" #include "main/database_manager.h" @@ -42,6 +45,7 @@ #include "storage/table/rel_table.h" #include "storage/table/rel_table_data.h" #include "storage/table/string_chunk_data.h" +#include "storage/wal/checksum_writer.h" #include "storage/wal/wal.h" #include "test_env.h" #include "transaction/transaction_manager.h" @@ -3049,6 +3053,86 @@ TEST_F(FlakyCheckpointerTest, StandaloneSerialTableSurvivesParentReopen) { ASSERT_FALSE(idsResult->hasNext()); } +// A pre-named-format binary logged sequence advances as UPDATE_SEQUENCE records carrying no +// sequence name, so replay must resolve their recorded entry ID through the create-replay ID +// map. The crafted transaction appended below reproduces the shift that map exists for: its +// CREATE record claims the entry ID of the table's SERIAL-column sequence (implicit serials +// have no create record, so replay cannot re-derive their IDs), and replay assigns the new +// sequence the next ID. Reading the recorded ID raw would advance the serial instead; the +// translated path must advance the crafted sequence and leave the serial untouched. +TEST_F(FlakyCheckpointerTest, LegacyUpdateSequenceReplayResolvesRecordedEntryID) { + if (inMemMode || systemConfig->checkpointThreshold == 0) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + auto tableResult = conn->query("CREATE NODE TABLE T(id SERIAL, name STRING, PRIMARY KEY(id));"); + ASSERT_TRUE(tableResult->isSuccess()) << tableResult->getErrorMessage(); + tableResult.reset(); + + auto& context = *conn->getClientContext(); + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + const auto recordedSequenceID = + catalog::Catalog::Get(context) + ->getSequenceEntry(transaction::Transaction::Get(context), "T_id_serial") + ->getOID(); + ASSERT_TRUE(conn->query("COMMIT;")->isSuccess()); + + BinaryData craftedWAL; + { + auto* memoryManager = MemoryManager::Get(context); + auto inMemWriter = std::make_shared(*memoryManager); + std::shared_ptr writer; + if (systemConfig->enableChecksums) { + writer = std::make_shared(inMemWriter, *memoryManager); + } else { + writer = inMemWriter; + } + Serializer serializer{writer}; + const auto appendRecord = [&serializer, &writer](const WALRecord& record) { + writer->onObjectBegin(); + WALRecord::serializeWithLength(serializer, record); + writer->onObjectEnd(); + }; + appendRecord(BeginTransactionRecord{}); + { + binder::BoundCreateSequenceInfo sequenceInfo("s", 1 /* startWith */, 1 /* increment */, + 1 /* minValue */, std::numeric_limits::max(), false /* cycle */, + ConflictAction::ON_CONFLICT_THROW, false /* isInternal */); + catalog::SequenceCatalogEntry sequenceEntry(sequenceInfo); + sequenceEntry.setOID(recordedSequenceID); + CreateCatalogEntryRecord createRecord(&sequenceEntry, false); + appendRecord(createRecord); + } + appendRecord(UpdateSequenceRecord{recordedSequenceID, 42}); + appendRecord(CommitRecord{}); + auto bufferWriter = std::make_shared(); + inMemWriter->flush(*bufferWriter); + craftedWAL = bufferWriter->getData(); + } + conn.reset(); + database.reset(); + + std::ofstream walFile{StorageUtils::getWALFilePath(databasePath), + std::ios::binary | std::ios::app}; + ASSERT_TRUE(walFile.is_open()); + walFile.write(reinterpret_cast(craftedWAL.data.get()), + static_cast(craftedWAL.size)); + walFile.close(); + ASSERT_TRUE(walFile.good()); + + createDBAndConn(); + auto openResult = conn->query("RETURN 1;"); + ASSERT_TRUE(openResult->isSuccess()) << openResult->getErrorMessage(); + ASSERT_EQ(openResult->getNext()->getValue(0)->getValue(), 1); + auto serialResult = conn->query("RETURN nextval('T_id_serial');"); + ASSERT_TRUE(serialResult->isSuccess()) << serialResult->getErrorMessage(); + ASSERT_EQ(serialResult->getNext()->getValue(0)->getValue(), 0); + auto sequenceResult = conn->query("RETURN nextval('s');"); + ASSERT_TRUE(sequenceResult->isSuccess()) << sequenceResult->getErrorMessage(); + ASSERT_EQ(sequenceResult->getNext()->getValue(0)->getValue(), 43); +} + // A standalone session's graph WAL addresses rel data by the per-direction physical // rel-table ID and binds a rel group through its endpoint node-table IDs, all recorded // in that session's plain catalog, which lacks the ANY-graph infrastructure entries diff --git a/test/transaction/wal_test.cpp b/test/transaction/wal_test.cpp index 11d26f69e1..b2ef37e74b 100644 --- a/test/transaction/wal_test.cpp +++ b/test/transaction/wal_test.cpp @@ -403,8 +403,13 @@ TEST_F(WalTest, WALRecordDeserializeSkipsUnknownTrailingBytes) { TEST_F(WalTest, WALRecordDeserializeKeepsLegacyUpdateSequenceOwnerIntact) { auto recordBuffer = std::make_shared(); Serializer recordSerializer{recordBuffer}; - lbug::storage::UpdateSequenceRecord record{7, 42}; - record.serialize(recordSerializer); + // Hand-write the pre-named-format payload instead of delegating to the current writer, so a + // matching incompatible change to both writer and reader cannot still pass as legacy + // decoding. + recordSerializer.writeDebuggingInfo("type"); + recordSerializer.write(lbug::storage::WALRecordType::UPDATE_SEQUENCE_RECORD); + recordSerializer.write(7); + recordSerializer.write(42); recordSerializer.writeDebuggingInfo("ownerCatalogName"); recordSerializer.write("mygraph"); From d43609fb39ec30e4546a30b62d0daa588c60f0ce Mon Sep 17 00:00:00 2001 From: Aikins Laryea Date: Wed, 7 Oct 2026 05:39:10 +0000 Subject: [PATCH 12/17] fix(storage): retire dropped graph catalogs after their records are applied --- docs/checkpoint_recovery.md | 73 ++- src/include/catalog/catalog.h | 5 + src/include/catalog/catalog_set.h | 5 + src/include/main/database_manager.h | 7 + src/include/storage/undo_buffer.h | 7 +- src/include/storage/wal/wal_replayer.h | 14 + src/include/transaction/transaction.h | 4 + src/main/database_manager.cpp | 49 +- src/storage/undo_buffer.cpp | 98 ++-- .../create_catalog_entry_record_replay.cpp | 5 + .../drop_catalog_entry_record_replay.cpp | 55 ++ src/storage/wal/wal_replayer.cpp | 37 +- src/transaction/transaction.cpp | 4 + test/transaction/checkpoint_test.cpp | 471 ++++++++++++++++++ tools/wal_dump/main.cpp | 3 + 15 files changed, 761 insertions(+), 76 deletions(-) diff --git a/docs/checkpoint_recovery.md b/docs/checkpoint_recovery.md index 6be366e8cd..bed775f34d 100644 --- a/docs/checkpoint_recovery.md +++ b/docs/checkpoint_recovery.md @@ -34,19 +34,32 @@ and recovery finishes the checkpoint. ## Checkpoint bundle format -`CheckpointRecord::bundleFormatVersion` is 1 for checkpoints written by builds that include this -recovery format, including nightly builds made from it. Version 0 records come from older builds, -including 0.21.2 and every earlier release, and have no version field. Recovery rejects an -unsupported checkpoint version before applying that checkpoint's shadow pages; opening a database -may already have recovered other graphs before it encounters the unsupported version. - -A version 1 checkpoint stamps every shadow file header with -`ShadowFile::CHECKPOINT_BUNDLE_DATABASE_ID` instead of a real database ID. Recovery identifies the -data file through the last database-header page in the shadow, which must match the data file's -database ID before any page is written. +`CheckpointRecord::bundleFormatVersion` is 2 for checkpoints written by current builds, including +nightly builds made from them. Version 1 records come from the first builds of this recovery +format, and version 0 records come from older builds, including 0.21.2 and every earlier release, +and have no version field. Recovery rejects an unsupported checkpoint version before applying that +checkpoint's shadow pages; opening a database may already have recovered other graphs before it +encounters the unsupported version. + +Version 1 and version 2 share the bundle layout. A checkpoint of either version stamps every +shadow file header with `ShadowFile::CHECKPOINT_BUNDLE_DATABASE_ID` instead of a real database ID, +and recovery identifies the data file through the last database-header page in the shadow, which +must match the data file's database ID before any page is written. + +Version 2 changes the WAL record encoding. Every record a version 2 build writes ends with an +`ownerCatalogName` trailer naming the graph catalog the record replays against; the trailer is +empty for main-database records, and its absence in a version 1 record is detected from the +record's framed length. A sequence update that names its sequence is written as the distinct +`UPDATE_SEQUENCE_NAMED` record type — the name lets replay find the sequence by name, because an +implicit serial sequence has no create record, so its entry ID can shift between logging and +replay — while a version 1 build writes only the plain `UPDATE_SEQUENCE` record. Recovery decodes +both versions' records. An older build cannot: it reads a record type it does not know as an +invalid record type, and a record type it does know decodes with the trailer skipped and replays +against the main catalog, so a WAL written by a version 2 build must be replayed and checkpointed +by a version 2 build before an older build opens the database (see Downgrades). A graph data file opened on its own (outside the database it was created in) checkpoints through -the same format: its WAL ends in a version 1 `CHECKPOINT` record and its shadow carries the +the same format: its WAL ends in a version 2 `CHECKPOINT` record and its shadow carries the sentinel. Reopening the parent database recovers such a graph from that record — including a checkpoint interrupted after the record became durable — after validating the graph's WAL header against the graph data file and requiring the sentinel in its shadow header. @@ -65,18 +78,32 @@ or its committed checkpoint recovered when the parent database is reopened. ## Downgrades -Open a database with a build that writes version 1 if it was last closed while a version 1 -checkpoint was pending. Pending means `.wal` or `.wal.checkpoint` ends in a `CHECKPOINT` -record and `.shadow` exists. - -Older builds refuse such a database because the shadow header does not match the data file. Their -error message suggests deleting the shadow file. **Do not delete it.** After the commit point, the -shadow files hold the only copy of the committed pages, and deleting them loses committed data. -Reopen the database with a build that writes version 1, let recovery finish, and close it cleanly -before going back to an older build. - -A database without a pending checkpoint has the same on-disk format as before and opens with older -builds. +Before opening a database that a version 2 build has written with an older build, run `CHECKPOINT` +on a version 2 build and let it succeed. The checkpoint is the step that empties the WAL; a clean +close does not, because `force_checkpoint_on_close=false` leaves version 2 records in the WAL tail +and an auto-checkpoint fires only once the WAL grows past `checkpoint_threshold`. + +A pending version 2 checkpoint means `.wal` or `.wal.checkpoint` ends in a `CHECKPOINT` +record and `.shadow` exists. Older builds refuse such a database because the shadow header does +not match the data file. Their error message suggests deleting the shadow file. **Do not delete it.** +After the commit point, the shadow files hold the only copy of the committed pages, and deleting them +loses committed data. Reopen the database with a build that writes version 2, let recovery finish, +and checkpoint before going back to an older build. + +An ordinary version 2 WAL tail — owner trailers on the records but no pending checkpoint — is the +quieter hazard: an older build does not reliably refuse it. A record type the older build knows +decodes with the trailer skipped as unknown trailing bytes and replays against the main catalog, +silently misapplying records meant for a graph. A record type the older build does not know, such +as `UPDATE_SEQUENCE_NAMED`, is no safer: with the default non-throwing replay configuration the +decoding error is treated like a corrupt WAL tail — the older build replays the committed prefix +and truncates the WAL there, silently discarding that transaction and every transaction recorded +after it. Setting `throwOnWalReplayFailure` to true rejects an unknown record before that WAL +is applied, but it still does not make downgrading safe: a WAL containing only recognized +record types passes validation, and its owner trailers are ignored. This is why the successful +`CHECKPOINT`, not a clean close, is the downgrade gate. + +A database with neither a pending checkpoint nor any remaining version 2 records has the same +on-disk format as before and opens with older builds. ## Version 0 checkpoints diff --git a/src/include/catalog/catalog.h b/src/include/catalog/catalog.h index 563bcc3121..3bf11bfcf4 100644 --- a/src/include/catalog/catalog.h +++ b/src/include/catalog/catalog.h @@ -216,6 +216,11 @@ class LBUG_API Catalog { // Get all graph entries. std::vector getGraphEntries( const transaction::Transaction* transaction) const; + // The next graph-entry OID this catalog would assign. Replay compares recorded + // GRAPH_ENTRY IDs against the value from before the replay pass: IDs at or above + // it were assigned after the last checkpoint and shift during recovery, while IDs + // below it belong to persisted entries and stay stable. + common::oid_t peekNextGraphOID() const { return graphs->peekNextOID(); } // Create graph entry. void createGraph(transaction::Transaction* transaction, std::string name, bool isAnyGraph); diff --git a/src/include/catalog/catalog_set.h b/src/include/catalog/catalog_set.h index ef5997244f..8ad55b186f 100644 --- a/src/include/catalog/catalog_set.h +++ b/src/include/catalog/catalog_set.h @@ -61,6 +61,11 @@ class LBUG_API CatalogSet { common::oid_t getNextOIDNoLock() { return nextOID++; } + common::oid_t peekNextOID() const { + std::shared_lock lck{mtx}; + return nextOID; + } + private: bool containsEntryNoLock(const transaction::Transaction* transaction, const std::string& name) const; diff --git a/src/include/main/database_manager.h b/src/include/main/database_manager.h index c494f81d5f..9f63816df4 100644 --- a/src/include/main/database_manager.h +++ b/src/include/main/database_manager.h @@ -63,6 +63,13 @@ class DatabaseManager { void clearDefaultGraph(); bool hasGraph(const std::string& graphName); catalog::Catalog* getGraphCatalog(const std::string& graphName); + // Unloads the named graph's in-memory catalog and any graph-WAL replay queued for it, + // without file or database-header work (the runtime drop removed the dropped graph's + // files; at replay the files under the name can belong to a recreated graph). Returns + // ownership of the removed catalog so the caller can evict state keyed by its address + // and defer destruction until no active transaction can hold undo records into it. + // Returns nullptr when no loaded catalog holds the name. + std::unique_ptr unloadGraphCatalog(const std::string& graphName); // Runs action with the graph registry's shared lock held, so a concurrent DROP GRAPH // cannot destroy the catalog for the action's duration. Callers that dereference the // catalog after a getGraphCatalog lookup must use this instead; getGraphCatalog only diff --git a/src/include/storage/undo_buffer.h b/src/include/storage/undo_buffer.h index badb3498f0..656f2d2b57 100644 --- a/src/include/storage/undo_buffer.h +++ b/src/include/storage/undo_buffer.h @@ -105,11 +105,12 @@ class UndoBuffer { static void rollbackRecord(main::ClientContext* context, UndoRecordType recordType, const uint8_t* record); - static void commitCatalogEntryRecord(const uint8_t* record, common::transaction_t commitTS); - static void rollbackCatalogEntryRecord(const uint8_t* record); + static void commitCatalogEntryRecord(main::ClientContext* context, const uint8_t* record, + common::transaction_t commitTS); + static void rollbackCatalogEntryRecord(main::ClientContext* context, const uint8_t* record); static void commitSequenceEntry(uint8_t const* entry, common::transaction_t commitTS); - static void rollbackSequenceEntry(uint8_t const* entry); + static void rollbackSequenceEntry(main::ClientContext* context, uint8_t const* entry); static void commitVersionInfo(main::ClientContext* context, UndoRecordType recordType, const uint8_t* record, common::transaction_t commitTS); diff --git a/src/include/storage/wal/wal_replayer.h b/src/include/storage/wal/wal_replayer.h index c942f1ae0a..9c0b1a5af8 100644 --- a/src/include/storage/wal/wal_replayer.h +++ b/src/include/storage/wal/wal_replayer.h @@ -31,6 +31,7 @@ class WALReplayer { std::vector walReplayRanges; bool retireActiveWAL = false; bool retireFrozenWAL = false; + common::oid_t persistedGraphOIDFloor = 0; }; explicit WALReplayer(main::ClientContext& clientContext); @@ -57,6 +58,8 @@ class WALReplayer { common::oid_t replayedEntryID) const; common::oid_t getReplayedEntryID(catalog::CatalogEntryType entryType, common::oid_t recordedEntryID) const; + bool tryGetReplayedEntryID(catalog::CatalogEntryType entryType, common::oid_t recordedEntryID, + common::oid_t& replayedEntryID) const; void replayCreateCatalogEntryRecord(WALRecord& walRecord) const; void replayCreateIndexRecord(WALRecord& walRecord) const; void replayDropCatalogEntryRecord(const WALRecord& walRecord) const; @@ -129,6 +132,17 @@ class WALReplayer { std::unordered_map>> replayedEntryIDs; + // Graph-entry IDs this pass could have assigned, from the base catalog's graph-set + // counter captured when the pass applies its first record (the persisted state). + // GRAPH_ENTRY drops treat recorded IDs at or above this floor as shifting replay + // IDs rather than stable persisted ones. Captured once per pass; graph WAL passes + // reset it alongside replayedEntryIDs. + mutable common::oid_t graphOIDReplayFloor = 0; + mutable bool graphOIDReplayFloorValid = false; + // Catalogs a GRAPH_ENTRY drop unregistered mid-pass. The active recovery transaction + // can hold undo records pointing into them, so destruction waits until the replayer + // dies, after the pass and its transactions have completed. + mutable std::vector> retiredCatalogs; }; } // namespace storage diff --git a/src/include/transaction/transaction.h b/src/include/transaction/transaction.h index 1e2a3ee54f..e8a50a011b 100644 --- a/src/include/transaction/transaction.h +++ b/src/include/transaction/transaction.h @@ -152,6 +152,9 @@ class LBUG_API Transaction { const binder::BoundAlterInfo& alterInfo, bool skipLoggingToWAL = false); void pushSequenceChange(catalog::SequenceCatalogEntry* sequenceEntry, int64_t kCount, const catalog::SequenceRollbackData& data); + // The transaction's undo records may point into the parked catalog; destroying it + // before commit or rollback finishes would leave those records dangling. + void retireGraphCatalog(std::unique_ptr catalog); void pushInsertInfo(common::node_group_idx_t nodeGroupIdx, common::row_idx_t startRow, common::row_idx_t numRows, const storage::VersionRecordHandler* versionRecordHandler) const; void pushDeleteInfo(common::node_group_idx_t nodeGroupIdx, common::row_idx_t startRow, @@ -180,6 +183,7 @@ class LBUG_API Transaction { std::vector> commitCallbacks; std::vector> rollbackCallbacks; bool forceCheckpoint; + std::vector> retiredGraphCatalogs; std::unordered_set changedCatalogs; std::mutex changedCatalogsMutex; }; diff --git a/src/main/database_manager.cpp b/src/main/database_manager.cpp index a59e405686..88253811c5 100644 --- a/src/main/database_manager.cpp +++ b/src/main/database_manager.cpp @@ -273,9 +273,15 @@ void DatabaseManager::dropGraph(const std::string& graphName, main::ClientContex if (hasAttachedDatabase(graphName)) { detachDatabase(graphName); } - graphIdentities.erase(it->get()); + // The dropping transaction can hold undo records pointing into the catalog, + // so destruction waits until commit or rollback finishes. + auto retiredCatalog = std::move(*it); + graphIdentities.erase(retiredCatalog.get()); graphs.erase(it); lck.unlock(); + if (transaction != nullptr) { + transaction->retireGraphCatalog(std::move(retiredCatalog)); + } // Delete the physical graph files if (!graphPath.empty() && @@ -419,6 +425,11 @@ bool DatabaseManager::loadGraph(main::ClientContext* clientContext, persistedHeader->catalogPageRange.startPageIdx != common::INVALID_PAGE_IDX; storage::Checkpointer::readCheckpoint(clientContext, catalog.get(), storageManager.get()); } + // Capture the persisted graph-ID counter before any replay can insert into + // this catalog: it is the floor separating persisted entries from the ones + // the graph's own WAL pass assigns, and the pass may run after the main WAL + // pass has already replayed tagged records into it. + recoveryState.persistedGraphOIDFloor = catalog->peekNextGraphOID(); catalog->setStorageManager(std::move(storageManager)); if (graphEntry->isAnyGraphType() && !hasPersistedCatalog) { // A graph whose create-graph record replayed from the WAL (never checkpointed) @@ -539,6 +550,41 @@ catalog::Catalog* DatabaseManager::getGraphCatalog(const std::string& graphName) throw BinderException{std::format("No graph named {}.", graphName)}; } +std::unique_ptr DatabaseManager::unloadGraphCatalog( + const std::string& graphName) { + auto upperCaseName = StringUtils::getUpper(graphName); + std::unique_ptr unloadedCatalog; + storage::StorageManager* unloadedStorageManager = nullptr; + { + std::unique_lock lck{graphsMutex}; + for (auto it = graphs.begin(); it != graphs.end(); ++it) { + auto graphNameUpper = StringUtils::getUpper((*it)->getCatalogName()); + if (graphNameUpper != upperCaseName) { + continue; + } + if (defaultGraph != "" && StringUtils::getUpper(defaultGraph) == upperCaseName) { + defaultGraph = ""; + } + unloadedStorageManager = (*it)->getStorageManager(); + if (unloadedStorageManager != nullptr) { + unloadedStorageManager->closeFileHandle(); + } + unloadedCatalog = std::move(*it); + graphIdentities.erase(unloadedCatalog.get()); + graphs.erase(it); + break; + } + } + if (unloadedCatalog == nullptr) { + return nullptr; + } + std::erase_if(pendingGraphWALReplays, + [&unloadedStorageManager](const std::unique_ptr& request) { + return request->storageManager == unloadedStorageManager; + }); + return unloadedCatalog; +} + void DatabaseManager::acquireGraphsShared() const { auto& holds = graphsSharedHolds[this]; if (holds > 0) { @@ -554,6 +600,7 @@ void DatabaseManager::releaseGraphsShared() const { --holds; if (holds == 0) { graphsMutex.unlock_shared(); + graphsSharedHolds.erase(this); } } diff --git a/src/storage/undo_buffer.cpp b/src/storage/undo_buffer.cpp index 8f6faf1e3d..655ca7ff06 100644 --- a/src/storage/undo_buffer.cpp +++ b/src/storage/undo_buffer.cpp @@ -29,11 +29,13 @@ struct UndoRecordHeader { struct CatalogEntryRecord { CatalogSet* catalogSet; CatalogEntry* catalogEntry; + catalog::Catalog* ownerCatalog; }; struct SequenceEntryRecord { SequenceCatalogEntry* sequenceEntry; SequenceRollbackData sequenceRollbackData; + catalog::Catalog* ownerCatalog; }; struct NodeBatchInsertRecord { @@ -99,7 +101,8 @@ void UndoBuffer::createCatalogEntry(CatalogSet& catalogSet, CatalogEntry& catalo const UndoRecordHeader recordHeader{UndoRecordType::CATALOG_ENTRY, sizeof(CatalogEntryRecord)}; *reinterpret_cast(buffer) = recordHeader; buffer += sizeof(UndoRecordHeader); - const CatalogEntryRecord catalogEntryRecord{&catalogSet, &catalogEntry}; + const CatalogEntryRecord catalogEntryRecord{&catalogSet, &catalogEntry, + catalogSet.getOwnerCatalogName().empty() ? nullptr : catalogSet.getCatalog()}; *reinterpret_cast(buffer) = catalogEntryRecord; } @@ -110,7 +113,8 @@ void UndoBuffer::createSequenceChange(SequenceCatalogEntry& sequenceEntry, sizeof(SequenceEntryRecord)}; *reinterpret_cast(buffer) = recordHeader; buffer += sizeof(UndoRecordHeader); - const SequenceEntryRecord sequenceEntryRecord{&sequenceEntry, data}; + const SequenceEntryRecord sequenceEntryRecord{&sequenceEntry, data, + sequenceEntry.getOwningCatalogName().empty() ? nullptr : sequenceEntry.getOwningCatalog()}; *reinterpret_cast(buffer) = sequenceEntryRecord; } @@ -165,22 +169,23 @@ uint8_t* UndoBuffer::createUndoRecord(const uint64_t size) { } namespace { -// A version record stores the owning catalog captured when it was pushed. If that -// catalog is no longer in the graph registry, a concurrent DROP GRAPH has destroyed the -// table's storage, so the record's handler dangles and there is nothing to apply: the -// graph's data, the only reader of this version info, died with it. Liveness is decided -// by pointer identity, so a graph recreated under the same name cannot adopt an old -// transaction's records. The apply runs under the registry shared lock, so a DROP GRAPH -// that has not won the race yet cannot destroy the storage mid-apply. -void applyVersionInfoIfOwnerGraphAlive(ClientContext* context, catalog::Catalog* ownerCatalog, - const std::function& applyVersionInfo) { +// An undo record stores the owning catalog captured when it was pushed; main and attached +// catalogs, which are never registered graphs, store nullptr instead. If a stored catalog +// is no longer in the graph registry, a DROP GRAPH has destroyed what the record points +// into, so the record dangles and there is nothing to apply: the graph's catalog and data, +// the only readers, died with it. Liveness is decided by pointer identity, so a graph +// recreated under the same name cannot adopt an old transaction's records. The apply runs +// under the registry shared lock, so a DROP GRAPH that has not won the race yet cannot +// destroy the storage mid-apply. +void applyIfOwnerGraphAlive(ClientContext* context, catalog::Catalog* ownerCatalog, + const std::function& apply) { if (ownerCatalog == nullptr) { - applyVersionInfo(); + apply(); return; } auto* dbManager = DatabaseManager::Get(*context); if (dbManager != nullptr) { - dbManager->withGraphCatalogIfAlive(ownerCatalog, applyVersionInfo); + dbManager->withGraphCatalogIfAlive(ownerCatalog, apply); } } } // namespace @@ -203,7 +208,7 @@ void UndoBuffer::commitRecord(ClientContext* context, UndoRecordType recordType, const uint8_t* record, transaction_t commitTS) { switch (recordType) { case UndoRecordType::CATALOG_ENTRY: { - commitCatalogEntryRecord(record, commitTS); + commitCatalogEntryRecord(context, record, commitTS); } break; case UndoRecordType::SEQUENCE_ENTRY: { commitSequenceEntry(record, commitTS); @@ -220,17 +225,20 @@ void UndoBuffer::commitRecord(ClientContext* context, UndoRecordType recordType, } } -void UndoBuffer::commitCatalogEntryRecord(const uint8_t* record, const transaction_t commitTS) { - const auto& [_, catalogEntry] = *reinterpret_cast(record); - const auto newCatalogEntry = catalogEntry->getNext(); - DASSERT(newCatalogEntry); - newCatalogEntry->setTimestamp(commitTS); +void UndoBuffer::commitCatalogEntryRecord(ClientContext* context, const uint8_t* record, + const transaction_t commitTS) { + const auto& entryRecord = *reinterpret_cast(record); + applyIfOwnerGraphAlive(context, entryRecord.ownerCatalog, [&]() { + const auto newCatalogEntry = entryRecord.catalogEntry->getNext(); + DASSERT(newCatalogEntry); + newCatalogEntry->setTimestamp(commitTS); + }); } void UndoBuffer::commitVersionInfo(ClientContext* context, UndoRecordType recordType, const uint8_t* record, transaction_t commitTS) { const auto& undoRecord = *reinterpret_cast(record); - applyVersionInfoIfOwnerGraphAlive(context, undoRecord.ownerCatalog, [&]() { + applyIfOwnerGraphAlive(context, undoRecord.ownerCatalog, [&]() { switch (recordType) { case UndoRecordType::INSERT_INFO: { undoRecord.versionRecordHandler->applyFuncToChunkedGroups( @@ -260,10 +268,10 @@ void UndoBuffer::rollbackRecord(ClientContext* context, const UndoRecordType rec const uint8_t* record) { switch (recordType) { case UndoRecordType::CATALOG_ENTRY: { - rollbackCatalogEntryRecord(record); + rollbackCatalogEntryRecord(context, record); } break; case UndoRecordType::SEQUENCE_ENTRY: { - rollbackSequenceEntry(record); + rollbackSequenceEntry(context, record); } break; case UndoRecordType::INSERT_INFO: case UndoRecordType::DELETE_INFO: { @@ -278,40 +286,44 @@ void UndoBuffer::rollbackRecord(ClientContext* context, const UndoRecordType rec } } -void UndoBuffer::rollbackCatalogEntryRecord(const uint8_t* record) { - const auto& [catalogSet, catalogEntry] = *reinterpret_cast(record); - const auto entryToRollback = catalogEntry->getNext(); - DASSERT(entryToRollback); - if (entryToRollback->getNext()) { - // If entryToRollback has a newer entry (next) in the version chain. Simple remove - // entryToRollback from the chain. - const auto newerEntry = entryToRollback->getNext(); - newerEntry->setPrev(entryToRollback->movePrev()); - } else { - // This is the beginning of the version chain. - auto olderEntry = entryToRollback->movePrev(); - catalogSet->eraseNoLock(catalogEntry->getName()); - if (olderEntry) { - catalogSet->emplaceNoLock(std::move(olderEntry)); +void UndoBuffer::rollbackCatalogEntryRecord(ClientContext* context, const uint8_t* record) { + const auto& entryRecord = *reinterpret_cast(record); + applyIfOwnerGraphAlive(context, entryRecord.ownerCatalog, [&]() { + const auto entryToRollback = entryRecord.catalogEntry->getNext(); + DASSERT(entryToRollback); + if (entryToRollback->getNext()) { + // If entryToRollback has a newer entry (next) in the version chain. Simple remove + // entryToRollback from the chain. + const auto newerEntry = entryToRollback->getNext(); + newerEntry->setPrev(entryToRollback->movePrev()); + } else { + // This is the beginning of the version chain. + auto olderEntry = entryToRollback->movePrev(); + entryRecord.catalogSet->eraseNoLock(entryRecord.catalogEntry->getName()); + if (olderEntry) { + entryRecord.catalogSet->emplaceNoLock(std::move(olderEntry)); + } } - } + }); } void UndoBuffer::commitSequenceEntry(const uint8_t*, transaction_t) { // DO NOTHING. } -void UndoBuffer::rollbackSequenceEntry(const uint8_t* entry) { +void UndoBuffer::rollbackSequenceEntry(ClientContext* context, const uint8_t* entry) { const auto& sequenceRecord = *reinterpret_cast(entry); - const auto sequenceEntry = sequenceRecord.sequenceEntry; - const auto& data = sequenceRecord.sequenceRollbackData; - sequenceEntry->rollbackVal(data.usageCount, data.currVal); + applyIfOwnerGraphAlive(context, sequenceRecord.ownerCatalog, [&]() { + const auto sequenceEntry = sequenceRecord.sequenceEntry; + const auto& data = sequenceRecord.sequenceRollbackData; + sequenceEntry->rollbackVal(data.usageCount, data.currVal); + }); } void UndoBuffer::rollbackVersionInfo(ClientContext* context, UndoRecordType recordType, const uint8_t* record) { auto& undoRecord = *reinterpret_cast(record); - applyVersionInfoIfOwnerGraphAlive(context, undoRecord.ownerCatalog, [&]() { + applyIfOwnerGraphAlive(context, undoRecord.ownerCatalog, [&]() { switch (recordType) { case UndoRecordType::INSERT_INFO: { undoRecord.versionRecordHandler->rollbackInsert(context, undoRecord.nodeGroupIdx, diff --git a/src/storage/wal/records/create_catalog_entry_record_replay.cpp b/src/storage/wal/records/create_catalog_entry_record_replay.cpp index 77e82ff708..8d7691bfa9 100644 --- a/src/storage/wal/records/create_catalog_entry_record_replay.cpp +++ b/src/storage/wal/records/create_catalog_entry_record_replay.cpp @@ -114,6 +114,11 @@ void WALReplayer::replayCreateCatalogEntryRecord(WALRecord& walRecord) const { case CatalogEntryType::GRAPH_ENTRY: { auto& graphEntry = record.ownedCatalogEntry->constCast(); catalog->createGraph(transaction, graphEntry.getName(), graphEntry.isAnyGraphType()); + // Graph-entry IDs shift when rolled-back DDL consumed IDs between logging and + // replay, so a later GRAPH_ENTRY drop must translate through this map like every + // other entry type. + recordReplayedEntryID(CatalogEntryType::GRAPH_ENTRY, graphEntry.getOID(), + catalog->getGraphEntry(transaction, graphEntry.getName())->getOID()); } break; default: { UNREACHABLE_CODE; diff --git a/src/storage/wal/records/drop_catalog_entry_record_replay.cpp b/src/storage/wal/records/drop_catalog_entry_record_replay.cpp index 0686b7615d..d0d019dfe4 100644 --- a/src/storage/wal/records/drop_catalog_entry_record_replay.cpp +++ b/src/storage/wal/records/drop_catalog_entry_record_replay.cpp @@ -1,5 +1,8 @@ #include "catalog/catalog.h" #include "catalog/catalog_entry/catalog_entry.h" +#include "main/client_context.h" +#include "main/database.h" +#include "main/database_manager.h" #include "storage/storage_manager.h" #include "storage/wal/wal_replayer.h" @@ -31,6 +34,58 @@ void WALReplayer::replayDropCatalogEntryRecord(const WALRecord& walRecord) const catalog->dropMacroEntry(transaction, entryID); } break; case CatalogEntryType::GRAPH_ENTRY: { + // Covers DROP GRAPH and the implicit subgraph entry a DROP TABLE or RENAME + // TABLE removes. Replay mirrors the in-memory half of the runtime drop: the + // entry goes so a later CREATE GRAPH of the same name replays, and a loaded + // catalog unloads so the name cannot keep routing records to a dropped graph + // and replayedEntryIDs cannot keep a sub-map keyed by it. Files stay + // untouched: DROP GRAPH removed the dropped graph's files at runtime, and at + // replay any files under the name belong to a recreated graph. + // + // A subgraph's create is never WAL-logged (createNodeTableSubgraph skips it), + // so its recorded ID has no recorded->replayed translation, and the table or + // ALTER record's own replay already removed the subgraph entry: those records + // must stay no-ops. Replay assigns fresh IDs to everything it creates, so an + // untranslated ID can collide with another replay-created graph's ID, and + // acting on such a match would drop a graph the user never dropped. Only two + // matches are safe: the translation a CREATE GRAPH replay recorded (a DROP + // GRAPH of a graph created in this WAL), and a recorded ID below the ID floor + // captured when the pass began (a persisted entry, whose ID cannot shift). + // Tagged records are always subgraph drops from a session on the graph, and + // any other untranslated ID at or above the floor is one too. + if (!walRecord.ownerCatalogName.empty()) { + break; + } + common::oid_t translatedEntryID = 0; + const auto translated = tryGetReplayedEntryID(CatalogEntryType::GRAPH_ENTRY, + dropEntryRecord.entryID, translatedEntryID); + if (!translated && dropEntryRecord.entryID >= graphOIDReplayFloor) { + break; + } + catalog::GraphCatalogEntry* droppedEntry = nullptr; + for (auto* graphEntry : catalog->getGraphEntries(transaction)) { + if (graphEntry->getOID() == (translated ? translatedEntryID : entryID)) { + droppedEntry = graphEntry; + break; + } + } + if (droppedEntry != nullptr) { + auto graphName = droppedEntry->getName(); + auto* dbManager = main::DatabaseManager::Get(clientContext); + // Only the main catalog's graphs are registered in the manager. A graph + // dropped from another graph's own catalog (a standalone session's nested + // create) must not unload the outer, registered graph of the same name. + if (dbManager != nullptr && catalog == clientContext.getDatabase()->getCatalog()) { + auto unloadedCatalog = dbManager->unloadGraphCatalog(graphName); + if (unloadedCatalog != nullptr) { + // The recovery transaction can hold undo records pointing into + // the catalog, so destruction waits until the replayer dies. + replayedEntryIDs.erase(unloadedCatalog.get()); + retiredCatalogs.push_back(std::move(unloadedCatalog)); + } + } + catalog->dropGraph(transaction, graphName); + } } break; default: { UNREACHABLE_CODE; diff --git a/src/storage/wal/wal_replayer.cpp b/src/storage/wal/wal_replayer.cpp index 1e39da39b6..058f3bd6bc 100644 --- a/src/storage/wal/wal_replayer.cpp +++ b/src/storage/wal/wal_replayer.cpp @@ -13,6 +13,7 @@ #include "common/file_system/file_system.h" #include "common/file_system/virtual_file_system.h" #include "common/serializer/buffered_file.h" +#include "common/string_utils.h" #include "common/system_message.h" #include "common/type_utils.h" #include "common/types/types.h" @@ -498,6 +499,11 @@ WALReplayer::GraphRecoveryState WALReplayer::prepareGraphCheckpoint(StorageManag void WALReplayer::replayGraphWAL(StorageManager& storageManager, const GraphRecoveryState& recoveryState) const { replayedEntryIDs.clear(); + // A graph's own WAL pass may run after the main pass has already inserted + // entries into its catalog, so the floor cannot be captured lazily here: + // it must be the persisted ID counter captured when the catalog loaded. + graphOIDReplayFloor = recoveryState.persistedGraphOIDFloor; + graphOIDReplayFloorValid = true; try { for (const auto& range : recoveryState.walReplayRanges) { auto flags = FileFlags::READ_ONLY; @@ -541,19 +547,29 @@ void WALReplayer::recordReplayedEntryID(catalog::CatalogEntryType entryType, common::oid_t WALReplayer::getReplayedEntryID(catalog::CatalogEntryType entryType, common::oid_t recordedEntryID) const { + common::oid_t replayedEntryID; + if (!tryGetReplayedEntryID(entryType, recordedEntryID, replayedEntryID)) { + return recordedEntryID; + } + return replayedEntryID; +} + +bool WALReplayer::tryGetReplayedEntryID(catalog::CatalogEntryType entryType, + common::oid_t recordedEntryID, common::oid_t& replayedEntryID) const { const auto catalogIt = replayedEntryIDs.find(catalog::Catalog::Get(clientContext)); if (catalogIt == replayedEntryIDs.end()) { - return recordedEntryID; + return false; } const auto typeIt = catalogIt->second.find(entryType); if (typeIt == catalogIt->second.end()) { - return recordedEntryID; + return false; } const auto idIt = typeIt->second.find(recordedEntryID); if (idIt == typeIt->second.end()) { - return recordedEntryID; + return false; } - return idIt->second; + replayedEntryID = idIt->second; + return true; } void WALReplayer::retireGraphCheckpointWALs(StorageManager& storageManager, @@ -811,11 +827,20 @@ WALReplayer::WALReplayInfo WALReplayer::dryReplay(FileInfo& fileInfo, bool throw } void WALReplayer::replayWALRecord(WALRecord& walRecord) const { + if (!graphOIDReplayFloorValid) { + // The main-WAL pass captures the floor before applying any record, when + // its catalog is in the persisted state: every ID replay assigns is at + // or above it, and persisted entries are below it. A graph pass seeds + // the floor from its recovery state instead (see replayGraphWAL). + graphOIDReplayFloor = catalog::Catalog::Get(clientContext)->peekNextGraphOID(); + graphOIDReplayFloorValid = true; + } std::optional replayOwnerScope; if (!walRecord.ownerCatalogName.empty()) { auto dbManager = main::DatabaseManager::Get(clientContext); + const auto upperOwnerName = StringUtils::getUpper(walRecord.ownerCatalogName); if (dbManager == nullptr || !dbManager->hasGraph(walRecord.ownerCatalogName)) { - if (dbManager != nullptr && !failedOwnerNames.contains(walRecord.ownerCatalogName)) { + if (dbManager != nullptr && !failedOwnerNames.contains(upperOwnerName)) { // A graph created earlier in this same WAL is not registered yet: its // create-graph record only adds the catalog entry. Materialize that one // graph (not the full graph-recovery pass) before routing this record into @@ -827,7 +852,7 @@ void WALReplayer::replayWALRecord(WALRecord& walRecord) const { dbManager->loadGraphFromCatalog(MemoryManager::Get(clientContext), &clientContext, walRecord.ownerCatalogName); if (!dbManager->hasGraph(walRecord.ownerCatalogName)) { - failedOwnerNames.insert(walRecord.ownerCatalogName); + failedOwnerNames.insert(upperOwnerName); } } } diff --git a/src/transaction/transaction.cpp b/src/transaction/transaction.cpp index 519b48c803..471073439e 100644 --- a/src/transaction/transaction.cpp +++ b/src/transaction/transaction.cpp @@ -257,6 +257,10 @@ void Transaction::pushSequenceChange(SequenceCatalogEntry* sequenceEntry, int64_ } } +void Transaction::retireGraphCatalog(std::unique_ptr catalog) { + retiredGraphCatalogs.push_back(std::move(catalog)); +} + void Transaction::pushInsertInfo(common::node_group_idx_t nodeGroupIdx, common::row_idx_t startRow, common::row_idx_t numRows, const storage::VersionRecordHandler* versionRecordHandler) const { undoBuffer->createInsertInfo(nodeGroupIdx, startRow, numRows, versionRecordHandler); diff --git a/test/transaction/checkpoint_test.cpp b/test/transaction/checkpoint_test.cpp index 8b2a65e195..5a5dce386f 100644 --- a/test/transaction/checkpoint_test.cpp +++ b/test/transaction/checkpoint_test.cpp @@ -3255,6 +3255,136 @@ TEST_F(FlakyCheckpointerTest, StandaloneIndexReplayDoesNotDropParentIndex) { ASSERT_TRUE(dropPersonResult->isSuccess()) << dropPersonResult->getErrorMessage(); } +// A standalone session on a graph file is that file's main session: its records go to +// the graph's own WAL untagged, and the graph's WAL pass runs after the main pass has +// already replayed tagged records into the graph's catalog. If that pass captured its +// graph-ID floor lazily at that point, an untagged subgraph drop recorded at or above +// the persisted counter — but below the post-main-pass counter — would identity-match +// a main-pass-created subgraph and drop it. The floor must come from the persisted +// counter captured when the graph's catalog loaded. The graph is never checkpointed, +// so the main recovery pass materializes it mid-replay and defers its WAL pass. +TEST_F(FlakyCheckpointerTest, DeferredGraphWALFloorSkipsReplayCreatedSubgraphs) { + if (inMemMode || systemConfig->checkpointThreshold == 0) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH floor_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH floor_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE NODE TABLE ExtraT(id INT64 PRIMARY KEY);")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(conn->query("COMMIT;")->isSuccess()); + + const auto graphPath = StorageUtils::getGraphPath(databasePath, "floor_graph"); + conn.reset(); + database.reset(); + + auto graphConfig = *systemConfig; + graphConfig.autoCheckpoint = false; + graphConfig.forceCheckpointOnClose = false; + auto graphDatabase = std::make_unique(graphPath, graphConfig); + auto graphConnection = std::make_unique(graphDatabase.get()); + auto ddl1 = graphConnection->query("CREATE NODE TABLE t(id INT64 PRIMARY KEY);"); + ASSERT_TRUE(ddl1->isSuccess()) << ddl1->getErrorMessage(); + auto ddl2 = graphConnection->query("DROP TABLE t;"); + ASSERT_TRUE(ddl2->isSuccess()) << ddl2->getErrorMessage(); + ddl1.reset(); + ddl2.reset(); + graphConnection.reset(); + graphDatabase.reset(); + + createDBAndConn(); + auto openResult = conn->query("RETURN 1;"); + ASSERT_TRUE(openResult->isSuccess()) << openResult->getErrorMessage(); + ASSERT_TRUE(conn->query("USE GRAPH floor_graph;")->isSuccess()); + auto extra = conn->query("MATCH (n:ExtraT) RETURN count(n);"); + ASSERT_TRUE(extra->isSuccess()) << extra->getErrorMessage(); + ASSERT_EQ(extra->getNext()->getValue(0)->getValue(), 0); + extra.reset(); + // Query binding resolves node tables from the catalog's table set, so the wrongly + // dropped entry is only observable at the graph-entry level: the main-pass-created + // subgraph must remain in the graph's own catalog, and the dropped table's must not. + auto* context = getClientContext(*conn); + std::unordered_set subgraphs; + main::DatabaseManager::Get(*context)->withGraphCatalog("floor_graph", + [&subgraphs](catalog::Catalog* catalog) { + for (auto* entry : + catalog->getGraphEntries(&transaction::DUMMY_CHECKPOINT_TRANSACTION)) { + subgraphs.insert(entry->getName()); + } + }); + EXPECT_TRUE(subgraphs.contains("ExtraT")) + << "the deferred graph-WAL pass dropped the main-pass-created ExtraT subgraph"; + EXPECT_FALSE(subgraphs.contains("t")); +} + +// The same standalone setup as above, but the untagged records are a nested +// CREATE GRAPH and DROP GRAPH under the outer graph's own name. Replaying the drop +// must remove the nested entry from the graph's catalog without unloading the outer, +// registered graph through the manager — otherwise the rest of the pass (and any +// later records) route to the main catalog instead of the graph being replayed. +TEST_F(FlakyCheckpointerTest, NestedGraphDropInGraphWALDoesNotUnloadOuterGraph) { + if (inMemMode || systemConfig->checkpointThreshold == 0) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH nested_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH nested_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE NODE TABLE Pin(id INT64 PRIMARY KEY);")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(conn->query("COMMIT;")->isSuccess()); + + const auto graphPath = StorageUtils::getGraphPath(databasePath, "nested_graph"); + conn.reset(); + database.reset(); + + auto graphConfig = *systemConfig; + graphConfig.autoCheckpoint = false; + graphConfig.forceCheckpointOnClose = false; + auto graphDatabase = std::make_unique(graphPath, graphConfig); + auto graphConnection = std::make_unique(graphDatabase.get()); + auto ddl1 = graphConnection->query("CREATE GRAPH nested_graph;"); + ASSERT_TRUE(ddl1->isSuccess()) << ddl1->getErrorMessage(); + auto ddl2 = graphConnection->query("DROP GRAPH nested_graph;"); + ASSERT_TRUE(ddl2->isSuccess()) << ddl2->getErrorMessage(); + auto ddl3 = graphConnection->query("CREATE NODE TABLE Survivor(id INT64 PRIMARY KEY);"); + ASSERT_TRUE(ddl3->isSuccess()) << ddl3->getErrorMessage(); + ddl1.reset(); + ddl2.reset(); + ddl3.reset(); + graphConnection.reset(); + graphDatabase.reset(); + + createDBAndConn(); + auto openResult = conn->query("RETURN 1;"); + ASSERT_TRUE(openResult->isSuccess()) << openResult->getErrorMessage(); + ASSERT_TRUE(conn->query("USE GRAPH nested_graph;")->isSuccess()); + auto survivor = conn->query("MATCH (n:Survivor) RETURN count(n);"); + ASSERT_TRUE(survivor->isSuccess()) << survivor->getErrorMessage(); + ASSERT_EQ(survivor->getNext()->getValue(0)->getValue(), 0); + survivor.reset(); + auto pin = conn->query("MATCH (n:Pin) RETURN count(n);"); + ASSERT_TRUE(pin->isSuccess()) << pin->getErrorMessage(); + ASSERT_EQ(pin->getNext()->getValue(0)->getValue(), 0); + pin.reset(); + + std::unordered_set outerGraphEntries; + main::DatabaseManager::Get(*getClientContext(*conn)) + ->withGraphCatalog("nested_graph", [&outerGraphEntries](catalog::Catalog* catalog) { + for (auto* entry : + catalog->getGraphEntries(&transaction::DUMMY_CHECKPOINT_TRANSACTION)) { + outerGraphEntries.insert(entry->getName()); + } + }); + EXPECT_FALSE(outerGraphEntries.contains("nested_graph")) + << "the replayed nested drop left its entry in the outer graph's catalog"; + EXPECT_TRUE(outerGraphEntries.contains("Pin")) + << "the outer graph lost the Pin table's implicit subgraph entry"; +} + // A staged rel insert into a graph-owned table resolves its owning catalog under the registry // lock at commit time. When DROP GRAPH wins the race the commit must fail cleanly with the // missing-graph binder error and leave the connection usable; when the commit wins it must @@ -4935,6 +5065,347 @@ TEST_F(ReviewFixesTest, SubgraphCatalogPersistsAfterCheckpointWithPreExistingTab ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); } +// Regression test for the #1113 review: replaying a DROP GRAPH record was a no-op, so a +// graph dropped and recreated within the same WAL tail replayed its second CREATE onto +// the stale main-catalog entry and wedged recovery. The dropped graph's materialized +// catalog also stayed loaded, keeping replay state keyed to it. Replay must mirror the +// in-memory half of DROP GRAPH: drop the catalog entry and unload the materialized +// catalog (which also evicts any WAL-replay request still queued for it). +TEST_F(ReviewFixesTest, DroppedGraphRecreatedInSameWALTailReplaysCleanly) { + if (inMemMode) { + GTEST_SKIP(); + } + + conn->query("CALL auto_checkpoint=false;"); + conn->query("CALL force_checkpoint_on_close=false;"); + // Consume a graph entry ID in a rolled-back transaction: the recorded entry IDs now + // diverge from the ones replay assigns, so the DROP record can only find its entry + // through the translation recorded at CREATE replay. The results are scoped so they + // are destroyed before the database below is torn down. + { + auto beginTx = conn->query("BEGIN TRANSACTION;"); + ASSERT_TRUE(beginTx->isSuccess()) << beginTx->getErrorMessage(); + ASSERT_TRUE(conn->query("CREATE GRAPH oid_burner;")->isSuccess()); + auto rb = conn->query("ROLLBACK;"); + ASSERT_TRUE(rb->isSuccess()) << rb->getErrorMessage(); + } + ASSERT_TRUE(conn->query("CREATE GRAPH churn_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH churn_graph;")->isSuccess()); + ASSERT_TRUE( + conn->query("CREATE NODE TABLE Gen1(id INT64 PRIMARY KEY, name STRING);")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE (:Gen1 {id: 1, name: 'one'});")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(conn->query("DROP GRAPH churn_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH churn_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH churn_graph;")->isSuccess()); + ASSERT_TRUE( + conn->query("CREATE NODE TABLE Gen2(id INT64 PRIMARY KEY, name STRING);")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE (:Gen2 {id: 2, name: 'two'});")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + + createDBAndConn(); + + ASSERT_TRUE(conn->query("USE GRAPH churn_graph;")->isSuccess()); + auto result = conn->query("MATCH (n:Gen2 {id: 2}) RETURN n.name;"); + ASSERT_TRUE(result->isSuccess()) << result->getErrorMessage(); + ASSERT_TRUE(result->hasNext()); + EXPECT_EQ(result->getNext()->getValue(0)->getValue(), "two"); + EXPECT_FALSE(result->hasNext()); + // The first generation must be gone with the drop, not resurrected by replay. + auto gen1 = conn->query("MATCH (n:Gen1) RETURN COUNT(n);"); + EXPECT_FALSE(gen1->isSuccess()); +} + +// A DROP TABLE logs a GRAPH_ENTRY drop for the table's implicit subgraph. The subgraph's +// create is never WAL-logged, so replay cannot translate its recorded ID, and replay +// assigns that ID range fresh: the record can collide with another replay-created graph's +// ID. Acting on that collision drops a graph the user never dropped, so the record must +// stay a no-op (the table record's own replay already removed the subgraph). +TEST_F(ReviewFixesTest, ImplicitSubgraphDropRecordSkipsUnrelatedGraphs) { + if (inMemMode) { + GTEST_SKIP(); + } + + conn->query("CALL auto_checkpoint=false;"); + conn->query("CALL force_checkpoint_on_close=false;"); + { + auto beginTx = conn->query("BEGIN TRANSACTION;"); + ASSERT_TRUE(beginTx->isSuccess()) << beginTx->getErrorMessage(); + ASSERT_TRUE(conn->query("CREATE GRAPH oid_burner;")->isSuccess()); + auto rb = conn->query("ROLLBACK;"); + ASSERT_TRUE(rb->isSuccess()) << rb->getErrorMessage(); + } + ASSERT_TRUE( + conn->query("CREATE NODE TABLE t(id INT64 PRIMARY KEY, name STRING);")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH survivor;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH survivor;")->isSuccess()); + ASSERT_TRUE( + conn->query("CREATE NODE TABLE S(id INT64 PRIMARY KEY, name STRING);")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE (:S {id: 7, name: 'seven'});")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(conn->query("DROP TABLE t;")->isSuccess()); + + createDBAndConn(); + + auto droppedTable = conn->query("MATCH (n:t) RETURN COUNT(n);"); + EXPECT_FALSE(droppedTable->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH survivor;")->isSuccess()); + auto result = conn->query("MATCH (n:S {id: 7}) RETURN n.name;"); + ASSERT_TRUE(result->isSuccess()) << result->getErrorMessage(); + ASSERT_TRUE(result->hasNext()); + EXPECT_EQ(result->getNext()->getValue(0)->getValue(), "seven"); + EXPECT_FALSE(result->hasNext()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); +} + +// A DROP GRAPH whose recorded entry ID is below the graph-ID floor targets a graph that +// the checkpoint persisted into the catalog before the session dropped it. The floor +// guards against replay-created ID collisions, not against persisted drops: replay +// must identity-apply the recorded ID so the entry is gone after recovery. +TEST_F(ReviewFixesTest, PersistedGraphDropBelowOIDFloorAppliesByIdentity) { + if (inMemMode) { + GTEST_SKIP(); + } + + conn->query("CALL auto_checkpoint=false;"); + conn->query("CALL force_checkpoint_on_close=false;"); + ASSERT_TRUE(conn->query("CREATE GRAPH doomed_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CHECKPOINT;")->isSuccess()); + ASSERT_TRUE(conn->query("DROP GRAPH doomed_graph;")->isSuccess()); + + createDBAndConn(); + + auto graphs = conn->query("CALL show_graphs() RETURN name ORDER BY name;"); + ASSERT_TRUE(graphs->isSuccess()) << graphs->getErrorMessage(); + while (graphs->hasNext()) { + const auto name = graphs->getNext()->getValue(0)->getValue(); + EXPECT_NE(name, "doomed_graph") << "drop of a persisted graph was skipped by replay"; + } + graphs.reset(); + auto recreate = conn->query("CREATE GRAPH doomed_graph;"); + ASSERT_TRUE(recreate->isSuccess()) << recreate->getErrorMessage(); +} + +// A GRAPH_ENTRY drop tagged with an owning graph replays against that graph's catalog, +// where replay assigns subgraph IDs from the graph's own ID counter. The main pass's +// OID floor counts the main catalog's entries, so the tagged record's untranslated ID +// must skip the floor fallback's identity lookup: a match there is a replay-created +// subgraph of the same graph that happens to carry the recorded ID, not the entry the +// runtime dropped. The rolled-back CREATE below shifts the recorded IDs by one so the +// dropped table's tagged record collides with another table's replay-assigned subgraph +// exactly when the tag guard is missing. +TEST_F(ReviewFixesTest, TaggedSubgraphDropSkipsMainFloorIdentityLookup) { + if (inMemMode) { + GTEST_SKIP(); + } + + conn->query("CALL auto_checkpoint=false;"); + conn->query("CALL force_checkpoint_on_close=false;"); + // Two persisted graphs put the main pass's graph-ID floor at 2. + ASSERT_TRUE(conn->query("CREATE GRAPH floor_one;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH drop_owner;")->isSuccess()); + ASSERT_TRUE(conn->query("CHECKPOINT;")->isSuccess()); + + ASSERT_TRUE(conn->query("USE GRAPH drop_owner;")->isSuccess()); + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE NODE TABLE consumed(id INT64 PRIMARY KEY);")->isSuccess()); + ASSERT_TRUE(conn->query("ROLLBACK;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE NODE TABLE t(id INT64 PRIMARY KEY);")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE NODE TABLE survivor(id INT64 PRIMARY KEY);")->isSuccess()); + ASSERT_TRUE(conn->query("DROP TABLE t;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + + createDBAndConn(); + auto openResult = conn->query("RETURN 1;"); + ASSERT_TRUE(openResult->isSuccess()) << openResult->getErrorMessage(); + + std::unordered_set subgraphs; + main::DatabaseManager::Get(*getClientContext(*conn)) + ->withGraphCatalog("drop_owner", [&subgraphs](catalog::Catalog* catalog) { + for (auto* entry : + catalog->getGraphEntries(&transaction::DUMMY_CHECKPOINT_TRANSACTION)) { + subgraphs.insert(entry->getName()); + } + }); + EXPECT_TRUE(subgraphs.contains("survivor")) + << "the tagged subgraph drop removed the surviving table's subgraph"; + EXPECT_FALSE(subgraphs.contains("t")); +} + +// The whole lifecycle in one explicit transaction makes recovery replay the drop inside +// the same recovery transaction that created the dropped graph's catalog entries. The +// transaction's undo records point into that catalog, so replay must unregister it +// without destroying it until the pass ends. The runtime drop faces the same constraint: +// it parks the catalog on the transaction so the commit below cannot apply its records +// to freed memory, no matter what the later same-name recreation allocates there. +TEST_F(ReviewFixesTest, DroppedGraphInSingleTransactionReplaysCleanly) { + if (inMemMode) { + GTEST_SKIP(); + } + + conn->query("CALL auto_checkpoint=false;"); + conn->query("CALL force_checkpoint_on_close=false;"); + { + auto beginTx = conn->query("BEGIN TRANSACTION;"); + ASSERT_TRUE(beginTx->isSuccess()) << beginTx->getErrorMessage(); + ASSERT_TRUE(conn->query("CREATE GRAPH churn_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH churn_graph;")->isSuccess()); + ASSERT_TRUE( + conn->query("CREATE NODE TABLE Gen1(id INT64 PRIMARY KEY, name STRING);")->isSuccess()); + // Gen1 is DDL-only on purpose: staging rows into it before the drop would hit + // the pre-existing local-storage generation conflation (drop + recreate in one + // transaction), which is tracked as a separate follow-up. + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(conn->query("DROP GRAPH churn_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH churn_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH churn_graph;")->isSuccess()); + ASSERT_TRUE( + conn->query("CREATE NODE TABLE Gen2(id INT64 PRIMARY KEY, name STRING);")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE (:Gen2 {id: 2, name: 'two'});")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + auto commitTx = conn->query("COMMIT;"); + ASSERT_TRUE(commitTx->isSuccess()) << commitTx->getErrorMessage(); + ASSERT_TRUE(conn->query("USE GRAPH churn_graph;")->isSuccess()); + auto committedIds = conn->query("MATCH (n:Gen2) RETURN n.id ORDER BY n.id;"); + ASSERT_TRUE(committedIds->isSuccess()) << committedIds->getErrorMessage(); + std::vector committed; + while (committedIds->hasNext()) { + committed.push_back(committedIds->getNext()->getValue(0)->getValue()); + } + EXPECT_EQ(committed, (std::vector{2})); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + } + + createDBAndConn(); + + ASSERT_TRUE(conn->query("USE GRAPH churn_graph;")->isSuccess()); + auto result = conn->query("MATCH (n:Gen2 {id: 2}) RETURN n.name;"); + ASSERT_TRUE(result->isSuccess()) << result->getErrorMessage(); + ASSERT_TRUE(result->hasNext()); + EXPECT_EQ(result->getNext()->getValue(0)->getValue(), "two"); + EXPECT_FALSE(result->hasNext()); + auto replayedIds = conn->query("MATCH (n:Gen2) RETURN n.id ORDER BY n.id;"); + ASSERT_TRUE(replayedIds->isSuccess()) << replayedIds->getErrorMessage(); + std::vector replayed; + while (replayedIds->hasNext()) { + replayed.push_back(replayedIds->getNext()->getValue(0)->getValue()); + } + EXPECT_EQ(replayed, (std::vector{2})); + auto gen1 = conn->query("MATCH (n:Gen1) RETURN COUNT(n);"); + EXPECT_FALSE(gen1->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); +} + +// DROP GRAPH removes the catalog's liveness identity from the manager immediately, even +// though the catalog itself stays parked on the dropping transaction until commit. The +// undo-buffer liveness guard relies on that identity: records owned by a dropped graph +// must be rejected while the catalog is parked, not admitted into retired storage, and +// the registry must not keep a stale identity after the transaction frees the catalog. +TEST_F(ReviewFixesTest, DroppedGraphLivenessIdentityRemovedAtDrop) { + conn->query("CALL auto_checkpoint=false;"); + conn->query("CALL force_checkpoint_on_close=false;"); + ASSERT_TRUE(conn->query("CREATE GRAPH gone_graph;")->isSuccess()); + catalog::Catalog* droppedCatalog = nullptr; + main::DatabaseManager::Get(*getClientContext(*conn)) + ->withGraphCatalog("gone_graph", + [&droppedCatalog](catalog::Catalog* catalog) { droppedCatalog = catalog; }); + ASSERT_NE(droppedCatalog, nullptr); + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + ASSERT_TRUE(conn->query("DROP GRAPH gone_graph;")->isSuccess()); + auto invoked = false; + EXPECT_FALSE(main::DatabaseManager::Get(*getClientContext(*conn)) + ->withGraphCatalogIfAlive(droppedCatalog, [&invoked] { invoked = true; })); + EXPECT_FALSE(invoked); + ASSERT_TRUE(conn->query("COMMIT;")->isSuccess()); +} + +// A transaction holding a graph-owned catalog-entry undo record must skip the record once +// another connection's committed DROP GRAPH has retired the owning catalog: the version chain +// and the catalog set the record points into dangle with it, so applying the rollback would +// write through freed memory instead of restoring anything a reader can still reach. The +// live-owner row asserts the same rollback path still removes an uncommitted entry when the +// owning graph survives. +TEST_F(ReviewFixesTest, RollbackSkipsDroppedGraphCatalogEntryUndo) { + conn->query("CALL auto_checkpoint=false;"); + conn->query("CALL force_checkpoint_on_close=false;"); + conn->query("CALL debug_enable_multi_writes=true;"); + ASSERT_TRUE(conn->query("CREATE GRAPH live_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH doomed_graph;")->isSuccess()); + + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH live_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE NODE TABLE t(id INT64 PRIMARY KEY);")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(conn->query("ROLLBACK;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH live_graph;")->isSuccess()); + auto uncommitted = conn->query("MATCH (n:t) RETURN count(n);"); + EXPECT_FALSE(uncommitted->isSuccess()) << "rollback left the uncommitted table resolvable"; + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + + main::Connection dropConn(database.get()); + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH doomed_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE NODE TABLE t(id INT64 PRIMARY KEY);")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(dropConn.query("DROP GRAPH doomed_graph;")->isSuccess()); + auto rollback = conn->query("ROLLBACK;"); + ASSERT_TRUE(rollback->isSuccess()) << rollback->getErrorMessage(); + auto probe = conn->query("RETURN 1;"); + ASSERT_TRUE(probe->isSuccess()) << probe->getErrorMessage(); + ASSERT_EQ(probe->getNext()->getValue(0)->getValue(), 1); + auto graphs = conn->query("CALL show_graphs() RETURN name;"); + ASSERT_TRUE(graphs->isSuccess()) << graphs->getErrorMessage(); + while (graphs->hasNext()) { + const auto name = graphs->getNext()->getValue(0)->getValue(); + EXPECT_NE(name, "doomed_graph") << "the skipped rollback resurrected the dropped graph"; + } +} + +// The sequence counterpart of the rollback above: advancing a graph-owned sequence inside a +// transaction records the owning catalog with the restore data. After another connection's +// committed DROP GRAPH retires that catalog, the rollback must skip restoring the dead +// sequence; with the owner alive it must still restore the sequence state the transaction +// consumed. +TEST_F(ReviewFixesTest, RollbackSkipsDroppedGraphSequenceUndo) { + conn->query("CALL auto_checkpoint=false;"); + conn->query("CALL force_checkpoint_on_close=false;"); + conn->query("CALL debug_enable_multi_writes=true;"); + ASSERT_TRUE(conn->query("CREATE GRAPH live_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH doomed_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH live_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE SEQUENCE live_seq;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH doomed_graph;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE SEQUENCE doomed_seq;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH live_graph;")->isSuccess()); + auto consumed = conn->query("RETURN nextval('live_seq');"); + ASSERT_TRUE(consumed->isSuccess()) << consumed->getErrorMessage(); + const auto consumedVal = consumed->getNext()->getValue(0)->getValue(); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(conn->query("ROLLBACK;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH live_graph;")->isSuccess()); + auto restored = conn->query("RETURN nextval('live_seq');"); + ASSERT_TRUE(restored->isSuccess()) << restored->getErrorMessage(); + EXPECT_EQ(restored->getNext()->getValue(0)->getValue(), consumedVal) + << "rollback did not restore the sequence state the transaction consumed"; + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + + main::Connection dropConn(database.get()); + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + ASSERT_TRUE(conn->query("USE GRAPH doomed_graph;")->isSuccess()); + auto doomedUse = conn->query("RETURN nextval('doomed_seq');"); + ASSERT_TRUE(doomedUse->isSuccess()) << doomedUse->getErrorMessage(); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + ASSERT_TRUE(dropConn.query("DROP GRAPH doomed_graph;")->isSuccess()); + auto rollback = conn->query("ROLLBACK;"); + ASSERT_TRUE(rollback->isSuccess()) << rollback->getErrorMessage(); + auto probe = conn->query("RETURN 1;"); + ASSERT_TRUE(probe->isSuccess()) << probe->getErrorMessage(); + ASSERT_EQ(probe->getNext()->getValue(0)->getValue(), 1); +} + // Regression tests for #1050: a checkpoint that fails after the PK index storage phase must // leave lookups, uniqueness and reopen behaving as if the checkpoint had not run. The PK // index used to publish read headers and clear its local storage mid-checkpoint, which diff --git a/tools/wal_dump/main.cpp b/tools/wal_dump/main.cpp index 077dfcd93d..02910a1038 100644 --- a/tools/wal_dump/main.cpp +++ b/tools/wal_dump/main.cpp @@ -113,6 +113,9 @@ static void printValueVector(const ValueVector& vector, uint64_t numRows) { } static void dumpRecord(const WALRecord& record) { + if (!record.ownerCatalogName.empty()) { + std::cout << " OwnerCatalog: " << record.ownerCatalogName << "\n"; + } switch (record.type) { case WALRecordType::BEGIN_TRANSACTION_RECORD: std::cout << " Type: BEGIN_TRANSACTION\n"; From 29e1db754bacc0b01b75daabfededfa0b20a7ee6 Mon Sep 17 00:00:00 2001 From: Aikins Laryea Date: Wed, 7 Oct 2026 22:54:49 +0000 Subject: [PATCH 13/17] fix(wal): record and translate ALTER connection endpoints across replay --- src/catalog/catalog_set.cpp | 8 +++++++- src/include/common/serializer/deserializer.h | 6 ++++++ src/include/storage/wal/local_wal.h | 3 ++- .../storage/wal/record/alter_table_entry_record.h | 7 +++++-- src/include/transaction/transaction.h | 3 ++- src/storage/wal/local_wal.cpp | 4 ++-- src/storage/wal/records/alter_table_entry_record.cpp | 4 ++++ .../wal/records/alter_table_entry_record_replay.cpp | 12 ++++++++++++ .../typespec/records/alter_table_entry_record.tsp | 4 ++++ .../wal/typespec/templates/wal_record_header.h.j2 | 7 +++++-- .../wal/typespec/templates/wal_record_source.cpp.j2 | 4 ++++ src/transaction/transaction.cpp | 6 ++++-- 12 files changed, 57 insertions(+), 11 deletions(-) diff --git a/src/catalog/catalog_set.cpp b/src/catalog/catalog_set.cpp index a6aa569a6b..6058310ebf 100644 --- a/src/catalog/catalog_set.cpp +++ b/src/catalog/catalog_set.cpp @@ -5,6 +5,7 @@ #include "binder/ddl/bound_alter_info.h" #include "catalog/catalog.h" #include "catalog/catalog_entry/dummy_catalog_entry.h" +#include "catalog/catalog_entry/rel_group_catalog_entry.h" #include "catalog/catalog_entry/table_catalog_entry.h" #include "common/assert.h" #include "common/exception/catalog.h" @@ -206,9 +207,14 @@ void CatalogSet::alterTableEntry(Transaction* transaction, const binder::BoundAl case AlterType::SET_SORTED_BY: case AlterType::ADD_FROM_TO_CONNECTION: case AlterType::DROP_FROM_TO_CONNECTION: { + auto addedRelTableOID = common::INVALID_TABLE_ID; + if (alterInfo.alterType == AlterType::ADD_FROM_TO_CONNECTION) { + addedRelTableOID = + newEntry->ptrCast()->getRelEntryInfos().back().oid; + } emplaceNoLock(std::move(newEntry)); if (transaction->shouldAppendToUndoBuffer()) { - transaction->pushAlterCatalogEntry(*this, *entry, alterInfo); + transaction->pushAlterCatalogEntry(*this, *entry, alterInfo, false, addedRelTableOID); } } break; default: { diff --git a/src/include/common/serializer/deserializer.h b/src/include/common/serializer/deserializer.h index c4d916e64d..a99dfb0786 100644 --- a/src/include/common/serializer/deserializer.h +++ b/src/include/common/serializer/deserializer.h @@ -53,6 +53,12 @@ class LBUG_API Deserializer { uint64_t getReadOffset() const { return reader->getReadOffset(); } void beginReadLimit(uint64_t size) { readLimit = getReadOffset() + size; } + uint64_t getRemainingReadLimit() const { + if (!readLimit.has_value()) { + return std::numeric_limits::max(); + } + return *readLimit > getReadOffset() ? *readLimit - getReadOffset() : 0; + } bool hasRemainingData() const { return !readLimit.has_value() || getReadOffset() < *readLimit; } void skipReadLimit() { if (!readLimit.has_value()) { diff --git a/src/include/storage/wal/local_wal.h b/src/include/storage/wal/local_wal.h index 416653a77f..04dd0f9969 100644 --- a/src/include/storage/wal/local_wal.h +++ b/src/include/storage/wal/local_wal.h @@ -31,7 +31,8 @@ class LocalWAL { void logDropCatalogEntryRecord(const std::string& ownerCatalogName, uint64_t tableID, catalog::CatalogEntryType type); void logAlterCatalogEntryRecord(const std::string& ownerCatalogName, - const binder::BoundAlterInfo* alterInfo); + const binder::BoundAlterInfo* alterInfo, + common::table_id_t addedRelTableOID = common::INVALID_TABLE_ID); void logUpdateSequenceRecord(const std::string& ownerCatalogName, common::sequence_id_t sequenceID, uint64_t kCount, const std::string& sequenceName); diff --git a/src/include/storage/wal/record/alter_table_entry_record.h b/src/include/storage/wal/record/alter_table_entry_record.h index c547830456..53cba85558 100644 --- a/src/include/storage/wal/record/alter_table_entry_record.h +++ b/src/include/storage/wal/record/alter_table_entry_record.h @@ -18,11 +18,14 @@ namespace storage { struct AlterTableEntryRecord final : WALRecord { const binder::BoundAlterInfo* alterInfo; std::unique_ptr ownedAlterInfo; + common::table_id_t addedRelTableOID = common::INVALID_TABLE_ID; AlterTableEntryRecord() : WALRecord{WALRecordType::ALTER_TABLE_ENTRY_RECORD}, alterInfo{nullptr} {} - explicit AlterTableEntryRecord(const binder::BoundAlterInfo* alterInfo) - : WALRecord{WALRecordType::ALTER_TABLE_ENTRY_RECORD}, alterInfo{alterInfo} {} + explicit AlterTableEntryRecord(const binder::BoundAlterInfo* alterInfo, + common::table_id_t addedRelTableOID = common::INVALID_TABLE_ID) + : WALRecord{WALRecordType::ALTER_TABLE_ENTRY_RECORD}, alterInfo{alterInfo}, + addedRelTableOID{addedRelTableOID} {} void serialize(common::Serializer& serializer) const override; static std::unique_ptr deserialize(common::Deserializer& deserializer); diff --git a/src/include/transaction/transaction.h b/src/include/transaction/transaction.h index e8a50a011b..6870014c5f 100644 --- a/src/include/transaction/transaction.h +++ b/src/include/transaction/transaction.h @@ -149,7 +149,8 @@ class LBUG_API Transaction { void pushCreateDropCatalogEntry(catalog::CatalogSet& catalogSet, catalog::CatalogEntry& catalogEntry, bool isInternal, bool skipLoggingToWAL = false); void pushAlterCatalogEntry(catalog::CatalogSet& catalogSet, catalog::CatalogEntry& catalogEntry, - const binder::BoundAlterInfo& alterInfo, bool skipLoggingToWAL = false); + const binder::BoundAlterInfo& alterInfo, bool skipLoggingToWAL = false, + common::table_id_t addedRelTableOID = common::INVALID_TABLE_ID); void pushSequenceChange(catalog::SequenceCatalogEntry* sequenceEntry, int64_t kCount, const catalog::SequenceRollbackData& data); // The transaction's undo records may point into the parked catalog; destroying it diff --git a/src/storage/wal/local_wal.cpp b/src/storage/wal/local_wal.cpp index 47853aefe5..533dc41111 100644 --- a/src/storage/wal/local_wal.cpp +++ b/src/storage/wal/local_wal.cpp @@ -49,8 +49,8 @@ void LocalWAL::logDropCatalogEntryRecord(const std::string& ownerCatalogName, ta } void LocalWAL::logAlterCatalogEntryRecord(const std::string& ownerCatalogName, - const BoundAlterInfo* alterInfo) { - AlterTableEntryRecord walRecord(alterInfo); + const BoundAlterInfo* alterInfo, table_id_t addedRelTableOID) { + AlterTableEntryRecord walRecord(alterInfo, addedRelTableOID); walRecord.ownerCatalogName = ownerCatalogName; addNewWALRecord(walRecord); } diff --git a/src/storage/wal/records/alter_table_entry_record.cpp b/src/storage/wal/records/alter_table_entry_record.cpp index 8a96878461..9dfef9638b 100644 --- a/src/storage/wal/records/alter_table_entry_record.cpp +++ b/src/storage/wal/records/alter_table_entry_record.cpp @@ -135,6 +135,7 @@ static decltype(auto) deserializeAlterRecord(Deserializer& deserializer) { void AlterTableEntryRecord::serialize(Serializer& serializer) const { WALRecord::serialize(serializer); serializeAlterExtraInfo(serializer, alterInfo); + serializer.write(addedRelTableOID); } std::unique_ptr AlterTableEntryRecord::deserialize( @@ -143,6 +144,9 @@ std::unique_ptr AlterTableEntryRecord::deserialize( auto retval = std::make_unique(); retval->ownedAlterInfo = std::make_unique(alterType, tableName, std::move(extraInfo)); + if (deserializer.getRemainingReadLimit() >= sizeof(uint64_t) * 2) { + deserializer.deserializeValue(retval->addedRelTableOID); + } return retval; } diff --git a/src/storage/wal/records/alter_table_entry_record_replay.cpp b/src/storage/wal/records/alter_table_entry_record_replay.cpp index ff90e2a294..a048115ac8 100644 --- a/src/storage/wal/records/alter_table_entry_record_replay.cpp +++ b/src/storage/wal/records/alter_table_entry_record_replay.cpp @@ -24,6 +24,14 @@ void WALReplayer::replayAlterTableEntryRecord(const WALRecord& walRecord) const auto transaction = transaction::Transaction::Get(clientContext); auto storageManager = StorageManager::Get(clientContext); auto ownedAlterInfo = alterEntryRecord.ownedAlterInfo.get(); + if (ownedAlterInfo->alterType == AlterType::ADD_FROM_TO_CONNECTION || + ownedAlterInfo->alterType == AlterType::DROP_FROM_TO_CONNECTION) { + auto& connectionInfo = ownedAlterInfo->extraInfo->cast(); + connectionInfo.fromTableID = + getReplayedEntryID(CatalogEntryType::NODE_TABLE_ENTRY, connectionInfo.fromTableID); + connectionInfo.toTableID = + getReplayedEntryID(CatalogEntryType::NODE_TABLE_ENTRY, connectionInfo.toTableID); + } catalog->alterTableEntry(transaction, *ownedAlterInfo); auto& pageAllocator = *PageManager::Get(clientContext); switch (ownedAlterInfo->alterType) { @@ -62,6 +70,10 @@ void WALReplayer::replayAlterTableEntryRecord(const WALRecord& walRecord) const auto relEntryInfo = relGroupEntry->getRelEntryInfo(extraInfo->fromTableID, extraInfo->toTableID); storageManager->addRelTable(relGroupEntry, *relEntryInfo, &clientContext); + if (alterEntryRecord.addedRelTableOID != INVALID_TABLE_ID) { + recordReplayedEntryID(CatalogEntryType::REL_GROUP_ENTRY, + alterEntryRecord.addedRelTableOID, relEntryInfo->oid); + } } break; default: break; diff --git a/src/storage/wal/typespec/records/alter_table_entry_record.tsp b/src/storage/wal/typespec/records/alter_table_entry_record.tsp index 6222ed6517..e1afa7187f 100644 --- a/src/storage/wal/typespec/records/alter_table_entry_record.tsp +++ b/src/storage/wal/typespec/records/alter_table_entry_record.tsp @@ -67,4 +67,8 @@ model DropFromToConnectionAlterPayload { model AlterTableEntryRecord extends WalRecordBase { alterInfo: BoundAlterInfoPayload; + // Physical rel-table OID allocated for an added connection; INVALID_TABLE_ID otherwise. + // Lets replay map runtime data records to the replay-added connection's physical table + // when the recorded and replayed ID spaces diverge. + addedRelTableOID: table_id_t; } diff --git a/src/storage/wal/typespec/templates/wal_record_header.h.j2 b/src/storage/wal/typespec/templates/wal_record_header.h.j2 index d392f8a3d0..133f3e9982 100644 --- a/src/storage/wal/typespec/templates/wal_record_header.h.j2 +++ b/src/storage/wal/typespec/templates/wal_record_header.h.j2 @@ -57,11 +57,14 @@ namespace storage { struct AlterTableEntryRecord final : WALRecord { const binder::BoundAlterInfo* alterInfo; std::unique_ptr ownedAlterInfo; + common::table_id_t addedRelTableOID = common::INVALID_TABLE_ID; AlterTableEntryRecord() : WALRecord{WALRecordType::{{ record_type }}}, alterInfo{nullptr} {} - explicit AlterTableEntryRecord(const binder::BoundAlterInfo* alterInfo) - : WALRecord{WALRecordType::{{ record_type }}}, alterInfo{alterInfo} {} + explicit AlterTableEntryRecord(const binder::BoundAlterInfo* alterInfo, + common::table_id_t addedRelTableOID = common::INVALID_TABLE_ID) + : WALRecord{WALRecordType::{{ record_type }}}, alterInfo{alterInfo}, + addedRelTableOID{addedRelTableOID} {} void serialize(common::Serializer& serializer) const override; static std::unique_ptr deserialize(common::Deserializer& deserializer); diff --git a/src/storage/wal/typespec/templates/wal_record_source.cpp.j2 b/src/storage/wal/typespec/templates/wal_record_source.cpp.j2 index 55d58b3b21..c99575e81e 100644 --- a/src/storage/wal/typespec/templates/wal_record_source.cpp.j2 +++ b/src/storage/wal/typespec/templates/wal_record_source.cpp.j2 @@ -195,6 +195,7 @@ static decltype(auto) deserializeAlterRecord(Deserializer& deserializer) { void AlterTableEntryRecord::serialize(Serializer& serializer) const { WALRecord::serialize(serializer); serializeAlterExtraInfo(serializer, alterInfo); + serializer.write(addedRelTableOID); } std::unique_ptr AlterTableEntryRecord::deserialize( @@ -203,6 +204,9 @@ std::unique_ptr AlterTableEntryRecord::deserialize( auto retval = std::make_unique(); retval->ownedAlterInfo = std::make_unique(alterType, tableName, std::move(extraInfo)); + if (deserializer.getRemainingReadLimit() >= sizeof(uint64_t) * 2) { + deserializer.deserializeValue(retval->addedRelTableOID); + } return retval; } {% elif dc.name == "CreateCatalogEntryRecord" %} diff --git a/src/transaction/transaction.cpp b/src/transaction/transaction.cpp index 471073439e..961144341b 100644 --- a/src/transaction/transaction.cpp +++ b/src/transaction/transaction.cpp @@ -237,12 +237,14 @@ void Transaction::pushCreateDropCatalogEntry(CatalogSet& catalogSet, CatalogEntr } void Transaction::pushAlterCatalogEntry(CatalogSet& catalogSet, CatalogEntry& catalogEntry, - const binder::BoundAlterInfo& alterInfo, bool skipLoggingToWAL) { + const binder::BoundAlterInfo& alterInfo, bool skipLoggingToWAL, + common::table_id_t addedRelTableOID) { undoBuffer->createCatalogEntry(catalogSet, catalogEntry); recordCatalogChange(catalogSet.getCatalog()); if (shouldLogToWAL() && !skipLoggingToWAL) { DASSERT(localWAL); - localWAL->logAlterCatalogEntryRecord(catalogSet.getOwnerCatalogName(), &alterInfo); + localWAL->logAlterCatalogEntryRecord(catalogSet.getOwnerCatalogName(), &alterInfo, + addedRelTableOID); } } From cd36e9d1fc84f504f883d7bb7f13de1620e37549 Mon Sep 17 00:00:00 2001 From: Aikins Laryea Date: Wed, 7 Oct 2026 22:55:00 +0000 Subject: [PATCH 14/17] fix(wal): translate recovered rel-insert endpoints through replayed entry IDs --- .../records/table_insertion_record_replay.cpp | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/src/storage/wal/records/table_insertion_record_replay.cpp b/src/storage/wal/records/table_insertion_record_replay.cpp index 1179e56144..c3c36b64c9 100644 --- a/src/storage/wal/records/table_insertion_record_replay.cpp +++ b/src/storage/wal/records/table_insertion_record_replay.cpp @@ -82,6 +82,23 @@ void WALReplayer::replayRelTableInsertRecord(const WALRecord& walRecord) const { const auto anchorState = insertionRecord.ownedVectors[0]->state; const auto numRels = anchorState->getSelVector().getSelSize(); DASSERT(insertionRecord.numRows == numRels); + // Recovered rows must reference the replayed catalog's node-table IDs; when the recovery ID + // space shifted, the recorded endpoint IDs are stale and must be translated before insert. + for (auto columnID : {LOCAL_BOUND_NODE_ID_COLUMN_ID, LOCAL_NBR_NODE_ID_COLUMN_ID}) { + auto& nodeIDVector = *insertionRecord.ownedVectors[columnID]; + for (auto i = 0u; i < numRels; i++) { + const auto pos = anchorState->getSelVector()[i]; + if (nodeIDVector.isNull(pos)) { + continue; + } + const auto nodeID = nodeIDVector.getValue(pos); + const auto replayedTableID = + getReplayedEntryID(catalog::CatalogEntryType::NODE_TABLE_ENTRY, nodeID.tableID); + if (replayedTableID != nodeID.tableID) { + nodeIDVector.setValue(pos, nodeID_t{nodeID.offset, replayedTableID}); + } + } + } anchorState->getSelVectorUnsafe().setToFiltered(1); for (auto i = 0u; i < insertionRecord.ownedVectors.size(); i++) { insertionRecord.ownedVectors[i]->setState(anchorState); From 6c4fdd0d36be454afde9ee006f994042fe60da37 Mon Sep 17 00:00:00 2001 From: Aikins Laryea Date: Wed, 7 Oct 2026 22:55:00 +0000 Subject: [PATCH 15/17] fix(wal): reject torn owner trailers instead of misrouting them --- src/storage/wal/wal_record.cpp | 33 +++++++++++++++++++++++++++++++-- 1 file changed, 31 insertions(+), 2 deletions(-) diff --git a/src/storage/wal/wal_record.cpp b/src/storage/wal/wal_record.cpp index 665bfcbcc5..59708e5bc9 100644 --- a/src/storage/wal/wal_record.cpp +++ b/src/storage/wal/wal_record.cpp @@ -1,5 +1,7 @@ #include "storage/wal/wal_record.h" +#include + #include "common/exception/runtime.h" #include "common/serializer/buffer_writer.h" #include "common/serializer/deserializer.h" @@ -95,13 +97,40 @@ std::unique_ptr WALRecord::deserialize(Deserializer& deserializer, throw RuntimeException("Corrupted wal file. Read out invalid WAL record type."); } } - if (deserializer.hasRemainingData()) { + bool hasOwnerTrailer = deserializer.hasRemainingData(); + bool hasCompleteLengthField = false; + uint64_t declaredOwnerNameLength = 0; + uint64_t decodedOwnerNameLength = 0; + bool hasTrailingInFrame = false; + if (hasOwnerTrailer) { deserializer.validateDebuggingInfo(key, "ownerCatalogName"); - deserializer.deserializeValue(walRecord->ownerCatalogName); + hasCompleteLengthField = deserializer.getRemainingReadLimit() >= sizeof(uint64_t); + deserializer.deserializeValue(declaredOwnerNameLength); + decodedOwnerNameLength = + std::min(declaredOwnerNameLength, deserializer.getRemainingReadLimit()); + auto& ownerCatalogName = walRecord->ownerCatalogName; + ownerCatalogName.resize(decodedOwnerNameLength); + deserializer.read(reinterpret_cast(ownerCatalogName.data()), + decodedOwnerNameLength); + hasTrailingInFrame = deserializer.hasRemainingData(); } walRecord->type = type; deserializer.skipReadLimit(); deserializer.getReader()->onObjectEnd(); + if (hasOwnerTrailer) { + if (!hasCompleteLengthField || declaredOwnerNameLength != decodedOwnerNameLength) { + throw RuntimeException( + "Corrupted wal file. Owner catalog name length overflows the record boundary."); + } + const auto& ownerCatalogName = walRecord->ownerCatalogName; + if (ownerCatalogName.find('\0') != std::string::npos) { + throw RuntimeException("Corrupted wal file. Owner catalog name contains a null byte."); + } + if (hasTrailingInFrame) { + throw RuntimeException( + "Corrupted wal file. Trailing bytes after the owner catalog name."); + } + } return walRecord; } From dbeba1afe9c9edbd67a7e4cf198b545c76393f91 Mon Sep 17 00:00:00 2001 From: Aikins Laryea Date: Wed, 7 Oct 2026 22:55:00 +0000 Subject: [PATCH 16/17] fix(storage): reject null bytes in graph names --- src/main/database_manager.cpp | 3 +++ 1 file changed, 3 insertions(+) diff --git a/src/main/database_manager.cpp b/src/main/database_manager.cpp index 88253811c5..069e76053f 100644 --- a/src/main/database_manager.cpp +++ b/src/main/database_manager.cpp @@ -175,6 +175,9 @@ static void createAnyGraphTables(catalog::Catalog& catalog) { void DatabaseManager::createGraph(const std::string& graphName, storage::MemoryManager* memoryManager, main::ClientContext* clientContext, bool isAnyGraph) { + if (graphName.find('\0') != std::string::npos) { + throw RuntimeException{"Graph names must not contain null bytes."}; + } if (StringUtils::caseInsensitiveEquals(graphName, "main")) { throw RuntimeException{"MAIN is a reserved graph name."}; } From 27970586bff80aea0c5591f28059ce3e7cb4cd4a Mon Sep 17 00:00:00 2001 From: Aikins Laryea Date: Wed, 7 Oct 2026 22:55:00 +0000 Subject: [PATCH 17/17] test(transaction): pin crafted recovery bytes for endpoints, torn trailers, and reopen --- test/transaction/checkpoint_test.cpp | 780 +++++++++++++++++++++++++++ test/transaction/wal_test.cpp | 191 ++++++- 2 files changed, 949 insertions(+), 22 deletions(-) diff --git a/test/transaction/checkpoint_test.cpp b/test/transaction/checkpoint_test.cpp index 5a5dce386f..538c33c15e 100644 --- a/test/transaction/checkpoint_test.cpp +++ b/test/transaction/checkpoint_test.cpp @@ -16,6 +16,7 @@ #include #include "api_test/private_api_test.h" +#include "binder/ddl/bound_alter_info.h" #include "binder/ddl/bound_create_sequence_info.h" #include "catalog/catalog.h" #include "catalog/catalog_entry/rel_group_catalog_entry.h" @@ -3133,6 +3134,341 @@ TEST_F(FlakyCheckpointerTest, LegacyUpdateSequenceReplayResolvesRecordedEntryID) ASSERT_EQ(sequenceResult->getNext()->getValue(0)->getValue(), 43); } +// The committed sentinel appended after the frame under test. A scanner that accepts +// the preceding frame reaches this commit and replays the sequence it creates; a +// scanner that rejects the frame truncates recovery before it and the sequence stays +// absent. +static BinaryData craftPostFrameSentinelWAL(main::ClientContext& context) { + auto* memoryManager = MemoryManager::Get(context); + auto inMemWriter = std::make_shared(*memoryManager); + std::shared_ptr writer = inMemWriter; + Serializer serializer{writer}; + const auto appendRecord = [&serializer, &writer](const WALRecord& record) { + writer->onObjectBegin(); + WALRecord::serializeWithLength(serializer, record); + writer->onObjectEnd(); + }; + appendRecord(BeginTransactionRecord{}); + { + binder::BoundCreateSequenceInfo sequenceInfo("post_torn", 1 /* startWith */, + 1 /* increment */, 1 /* minValue */, std::numeric_limits::max(), + false /* cycle */, ConflictAction::ON_CONFLICT_THROW, false /* isInternal */); + catalog::SequenceCatalogEntry sequenceEntry(sequenceInfo); + sequenceEntry.setOID(101); + CreateCatalogEntryRecord createRecord(&sequenceEntry, false); + appendRecord(createRecord); + } + appendRecord(CommitRecord{}); + auto bufferWriter = std::make_shared(); + inMemWriter->flush(*bufferWriter); + return bufferWriter->getData(); +} + +// The owner trailer's declared length must stay inside the record's WAL frame. A torn +// trailer whose length field overruns the frame used to zero-fill the name silently, so +// the record replayed into whichever catalog the garbage name selected. The strict +// length check must reject the record so the scan truncates at the last committed +// record instead. Checksummed WALs reject any physical tear at the checksum layer +// before this check can fire, so this exercises the checksum-less configuration. +TEST_F(FlakyCheckpointerTest, TornOwnerTrailerOverrunTruncatesAtLastCommit) { + if (inMemMode || systemConfig->checkpointThreshold == 0 || systemConfig->enableChecksums) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + // A committed record leaves a WAL file with a database-ID header to append to. + auto tableResult = conn->query("CREATE NODE TABLE T(id INT64 PRIMARY KEY);"); + ASSERT_TRUE(tableResult->isSuccess()) << tableResult->getErrorMessage(); + + auto& context = *conn->getClientContext(); + BinaryData craftedWAL; + { + auto* memoryManager = MemoryManager::Get(context); + auto inMemWriter = std::make_shared(*memoryManager); + std::shared_ptr writer = inMemWriter; + Serializer serializer{writer}; + const auto appendRecord = [&serializer, &writer](const WALRecord& record) { + writer->onObjectBegin(); + WALRecord::serializeWithLength(serializer, record); + writer->onObjectEnd(); + }; + appendRecord(BeginTransactionRecord{}); + { + binder::BoundCreateSequenceInfo sequenceInfo("s", 1 /* startWith */, 1 /* increment */, + 1 /* minValue */, std::numeric_limits::max(), false /* cycle */, + ConflictAction::ON_CONFLICT_THROW, false /* isInternal */); + catalog::SequenceCatalogEntry sequenceEntry(sequenceInfo); + sequenceEntry.setOID(100); + CreateCatalogEntryRecord createRecord(&sequenceEntry, false); + appendRecord(createRecord); + } + appendRecord(CommitRecord{}); + UpdateSequenceRecord tornRecord{100, 43}; + tornRecord.ownerCatalogName = "ABCD"; + appendRecord(tornRecord); + auto bufferWriter = std::make_shared(); + inMemWriter->flush(*bufferWriter); + craftedWAL = bufferWriter->getData(); + } + // The trailer's 8-byte length field sits directly before the 4-byte name. + std::memset(craftedWAL.data.get() + craftedWAL.size - 12, 0xFF, 8); + auto sentinelWAL = craftPostFrameSentinelWAL(context); + + conn.reset(); + database.reset(); + std::ofstream walFile{StorageUtils::getWALFilePath(databasePath), + std::ios::binary | std::ios::app}; + ASSERT_TRUE(walFile.is_open()); + walFile.write(reinterpret_cast(craftedWAL.data.get()), + static_cast(craftedWAL.size)); + walFile.write(reinterpret_cast(sentinelWAL.data.get()), + static_cast(sentinelWAL.size)); + walFile.close(); + ASSERT_TRUE(walFile.good()); + + createDBAndConn(); + auto openResult = conn->query("RETURN 1;"); + ASSERT_TRUE(openResult->isSuccess()) << openResult->getErrorMessage(); + auto sequenceResult = conn->query("RETURN nextval('s');"); + ASSERT_TRUE(sequenceResult->isSuccess()) << sequenceResult->getErrorMessage(); + ASSERT_EQ(sequenceResult->getNext()->getValue(0)->getValue(), 1); + // The scan must stop at the torn frame: the committed sentinel behind it never + // replays, so the sequence it creates stays absent. + auto sentinelResult = conn->query("RETURN nextval('post_torn');"); + EXPECT_FALSE(sentinelResult->isSuccess()) << sentinelResult->getErrorMessage(); +} + +// A frame that still has bytes left after a fully decoded owner trailer is torn the +// same way: the old reader skipped the leftovers, so a record could carry a valid +// name plus trailing garbage and still replay. The trailer must consume the frame +// exactly or the record is rejected. +TEST_F(FlakyCheckpointerTest, TornOwnerTrailerTrailingBytesTruncatesAtLastCommit) { + if (inMemMode || systemConfig->checkpointThreshold == 0 || systemConfig->enableChecksums) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + // A committed record leaves a WAL file with a database-ID header to append to. + auto tableResult = conn->query("CREATE NODE TABLE T(id INT64 PRIMARY KEY);"); + ASSERT_TRUE(tableResult->isSuccess()) << tableResult->getErrorMessage(); + + auto& context = *conn->getClientContext(); + BinaryData goodWAL, tornWAL; + { + auto* memoryManager = MemoryManager::Get(context); + auto inMemWriter = std::make_shared(*memoryManager); + std::shared_ptr writer = inMemWriter; + Serializer serializer{writer}; + const auto appendRecord = [&serializer, &writer](const WALRecord& record) { + writer->onObjectBegin(); + WALRecord::serializeWithLength(serializer, record); + writer->onObjectEnd(); + }; + appendRecord(BeginTransactionRecord{}); + { + binder::BoundCreateSequenceInfo sequenceInfo("s", 1 /* startWith */, 1 /* increment */, + 1 /* minValue */, std::numeric_limits::max(), false /* cycle */, + ConflictAction::ON_CONFLICT_THROW, false /* isInternal */); + catalog::SequenceCatalogEntry sequenceEntry(sequenceInfo); + sequenceEntry.setOID(100); + CreateCatalogEntryRecord createRecord(&sequenceEntry, false); + appendRecord(createRecord); + } + appendRecord(CommitRecord{}); + auto bufferWriter = std::make_shared(); + inMemWriter->flush(*bufferWriter); + goodWAL = bufferWriter->getData(); + } + { + auto* memoryManager = MemoryManager::Get(context); + auto inMemWriter = std::make_shared(*memoryManager); + std::shared_ptr writer = inMemWriter; + Serializer serializer{writer}; + writer->onObjectBegin(); + UpdateSequenceRecord tornRecord{100, 43}; + tornRecord.ownerCatalogName = "ABCD"; + WALRecord::serializeWithLength(serializer, tornRecord); + writer->onObjectEnd(); + auto bufferWriter = std::make_shared(); + inMemWriter->flush(*bufferWriter); + tornWAL = bufferWriter->getData(); + } + // Grow the torn record's frame by 8 so it covers 8 bytes of trailing garbage + // written after the (empty) trailer. + uint64_t frameLength = 0; + std::memcpy(&frameLength, tornWAL.data.get(), sizeof(frameLength)); + frameLength += 8; + std::memcpy(tornWAL.data.get(), &frameLength, sizeof(frameLength)); + auto sentinelWAL = craftPostFrameSentinelWAL(context); + + conn.reset(); + database.reset(); + std::ofstream walFile{StorageUtils::getWALFilePath(databasePath), + std::ios::binary | std::ios::app}; + ASSERT_TRUE(walFile.is_open()); + walFile.write(reinterpret_cast(goodWAL.data.get()), + static_cast(goodWAL.size)); + walFile.write(reinterpret_cast(tornWAL.data.get()), + static_cast(tornWAL.size)); + const uint8_t trailing[8] = {0xEE, 0xEE, 0xEE, 0xEE, 0xEE, 0xEE, 0xEE, 0xEE}; + walFile.write(reinterpret_cast(trailing), sizeof(trailing)); + walFile.write(reinterpret_cast(sentinelWAL.data.get()), + static_cast(sentinelWAL.size)); + walFile.close(); + ASSERT_TRUE(walFile.good()); + + createDBAndConn(); + auto openResult = conn->query("RETURN 1;"); + ASSERT_TRUE(openResult->isSuccess()) << openResult->getErrorMessage(); + auto sequenceResult = conn->query("RETURN nextval('s');"); + ASSERT_TRUE(sequenceResult->isSuccess()) << sequenceResult->getErrorMessage(); + ASSERT_EQ(sequenceResult->getNext()->getValue(0)->getValue(), 1); + // The scan must stop at the torn frame: the committed sentinel behind it never + // replays, so the sequence it creates stays absent. + auto sentinelResult = conn->query("RETURN nextval('post_torn');"); + EXPECT_FALSE(sentinelResult->isSuccess()) << sentinelResult->getErrorMessage(); +} + +// A zero byte inside a decoded owner name is the signature of a payload torn from +// zeroed file blocks: real catalog names are identifiers and never contain NUL. The +// strict decode must reject such a record rather than route it to a bogus owner. +TEST_F(FlakyCheckpointerTest, TornOwnerTrailerNullByteTruncatesAtLastCommit) { + if (inMemMode || systemConfig->checkpointThreshold == 0 || systemConfig->enableChecksums) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + // A committed record leaves a WAL file with a database-ID header to append to. + auto tableResult = conn->query("CREATE NODE TABLE T(id INT64 PRIMARY KEY);"); + ASSERT_TRUE(tableResult->isSuccess()) << tableResult->getErrorMessage(); + + auto& context = *conn->getClientContext(); + BinaryData craftedWAL; + { + auto* memoryManager = MemoryManager::Get(context); + auto inMemWriter = std::make_shared(*memoryManager); + std::shared_ptr writer = inMemWriter; + Serializer serializer{writer}; + const auto appendRecord = [&serializer, &writer](const WALRecord& record) { + writer->onObjectBegin(); + WALRecord::serializeWithLength(serializer, record); + writer->onObjectEnd(); + }; + appendRecord(BeginTransactionRecord{}); + { + binder::BoundCreateSequenceInfo sequenceInfo("s", 1 /* startWith */, 1 /* increment */, + 1 /* minValue */, std::numeric_limits::max(), false /* cycle */, + ConflictAction::ON_CONFLICT_THROW, false /* isInternal */); + catalog::SequenceCatalogEntry sequenceEntry(sequenceInfo); + sequenceEntry.setOID(100); + CreateCatalogEntryRecord createRecord(&sequenceEntry, false); + appendRecord(createRecord); + } + appendRecord(CommitRecord{}); + UpdateSequenceRecord tornRecord{100, 43}; + tornRecord.ownerCatalogName = "ABCD"; + appendRecord(tornRecord); + auto bufferWriter = std::make_shared(); + inMemWriter->flush(*bufferWriter); + craftedWAL = bufferWriter->getData(); + } + craftedWAL.data.get()[craftedWAL.size - 4 + 1] = 0x00; + auto sentinelWAL = craftPostFrameSentinelWAL(context); + + conn.reset(); + database.reset(); + std::ofstream walFile{StorageUtils::getWALFilePath(databasePath), + std::ios::binary | std::ios::app}; + ASSERT_TRUE(walFile.is_open()); + walFile.write(reinterpret_cast(craftedWAL.data.get()), + static_cast(craftedWAL.size)); + walFile.write(reinterpret_cast(sentinelWAL.data.get()), + static_cast(sentinelWAL.size)); + walFile.close(); + ASSERT_TRUE(walFile.good()); + + createDBAndConn(); + auto openResult = conn->query("RETURN 1;"); + ASSERT_TRUE(openResult->isSuccess()) << openResult->getErrorMessage(); + auto sequenceResult = conn->query("RETURN nextval('s');"); + ASSERT_TRUE(sequenceResult->isSuccess()) << sequenceResult->getErrorMessage(); + ASSERT_EQ(sequenceResult->getNext()->getValue(0)->getValue(), 1); + // The scan must stop at the torn frame: the committed sentinel behind it never + // replays, so the sequence it creates stays absent. + auto sentinelResult = conn->query("RETURN nextval('post_torn');"); + EXPECT_FALSE(sentinelResult->isSuccess()) << sentinelResult->getErrorMessage(); +} + +// The control for the three torn-trailer tests: the identical frame left well formed +// decodes and the scanner walks on to the committed sentinel behind it, so the sentinel +// sequence replays. Its absence in the torn tests is therefore the torn frames' doing. +TEST_F(FlakyCheckpointerTest, WellFormedOwnerTrailerFrameReachesLaterCommit) { + if (inMemMode || systemConfig->checkpointThreshold == 0 || systemConfig->enableChecksums) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + // A committed record leaves a WAL file with a database-ID header to append to. + auto tableResult = conn->query("CREATE NODE TABLE T(id INT64 PRIMARY KEY);"); + ASSERT_TRUE(tableResult->isSuccess()) << tableResult->getErrorMessage(); + + auto& context = *conn->getClientContext(); + BinaryData craftedWAL; + { + auto* memoryManager = MemoryManager::Get(context); + auto inMemWriter = std::make_shared(*memoryManager); + std::shared_ptr writer = inMemWriter; + Serializer serializer{writer}; + const auto appendRecord = [&serializer, &writer](const WALRecord& record) { + writer->onObjectBegin(); + WALRecord::serializeWithLength(serializer, record); + writer->onObjectEnd(); + }; + appendRecord(BeginTransactionRecord{}); + { + binder::BoundCreateSequenceInfo sequenceInfo("s", 1 /* startWith */, 1 /* increment */, + 1 /* minValue */, std::numeric_limits::max(), false /* cycle */, + ConflictAction::ON_CONFLICT_THROW, false /* isInternal */); + catalog::SequenceCatalogEntry sequenceEntry(sequenceInfo); + sequenceEntry.setOID(100); + CreateCatalogEntryRecord createRecord(&sequenceEntry, false); + appendRecord(createRecord); + } + appendRecord(CommitRecord{}); + UpdateSequenceRecord updateRecord{100, 43}; + updateRecord.ownerCatalogName = "ABCD"; + appendRecord(updateRecord); + auto bufferWriter = std::make_shared(); + inMemWriter->flush(*bufferWriter); + craftedWAL = bufferWriter->getData(); + } + auto sentinelWAL = craftPostFrameSentinelWAL(context); + + conn.reset(); + database.reset(); + std::ofstream walFile{StorageUtils::getWALFilePath(databasePath), + std::ios::binary | std::ios::app}; + ASSERT_TRUE(walFile.is_open()); + walFile.write(reinterpret_cast(craftedWAL.data.get()), + static_cast(craftedWAL.size)); + walFile.write(reinterpret_cast(sentinelWAL.data.get()), + static_cast(sentinelWAL.size)); + walFile.close(); + ASSERT_TRUE(walFile.good()); + + createDBAndConn(); + auto openResult = conn->query("RETURN 1;"); + ASSERT_TRUE(openResult->isSuccess()) << openResult->getErrorMessage(); + auto sequenceResult = conn->query("RETURN nextval('s');"); + ASSERT_TRUE(sequenceResult->isSuccess()) << sequenceResult->getErrorMessage(); + ASSERT_EQ(sequenceResult->getNext()->getValue(0)->getValue(), 1); + // The well-formed frame lets the scan reach the sentinel commit; the update itself + // routes to the unknown ABCD graph, so sequence s stays untouched. + auto sentinelResult = conn->query("RETURN nextval('post_torn');"); + ASSERT_TRUE(sentinelResult->isSuccess()) << sentinelResult->getErrorMessage(); + ASSERT_EQ(sentinelResult->getNext()->getValue(0)->getValue(), 1); +} + // A standalone session's graph WAL addresses rel data by the per-direction physical // rel-table ID and binds a rel group through its endpoint node-table IDs, all recorded // in that session's plain catalog, which lacks the ANY-graph infrastructure entries @@ -3188,6 +3524,64 @@ TEST_F(FlakyCheckpointerTest, StandaloneRelTableReplayDoesNotPoisonRecovery) { ASSERT_EQ(relResult->getNext()->getValue(0)->getValue(), 1); } +// One graph written from two namespaces: the main session logs owner-tagged records +// into the main WAL, then a standalone session on the graph file logs plain records +// into the graph's own WAL. Both WALs replay in a single reopen — the main pass first, +// then the graph pass — and both namespaces recorded their table IDs against the same +// persisted catalog counter, so their recorded IDs overlap. The rolled-back DDL in the +// main session exercises the counter rewind in the same WAL. Each table's data must +// still land in that table after the reopen. +TEST_F(FlakyCheckpointerTest, TwoNamespaceReopenKeepsPerTableData) { + if (inMemMode || systemConfig->checkpointThreshold == 0) { + GTEST_SKIP(); + } + ASSERT_TRUE(conn->query("CALL force_checkpoint_on_close=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CALL auto_checkpoint=false;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE GRAPH two_ns;")->isSuccess()); + ASSERT_TRUE(conn->query("CHECKPOINT;")->isSuccess()); + + ASSERT_TRUE(conn->query("USE GRAPH two_ns;")->isSuccess()); + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE NODE TABLE consumed(id INT64 PRIMARY KEY);")->isSuccess()); + ASSERT_TRUE(conn->query("ROLLBACK;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE NODE TABLE Person(id INT64 PRIMARY KEY);")->isSuccess()); + for (auto id = 1; id <= 2; id++) { + auto insert = conn->query(std::format("CREATE (:Person {{id: {}}});", id)); + ASSERT_TRUE(insert->isSuccess()) << insert->getErrorMessage(); + } + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + conn.reset(); + database.reset(); + + const auto graphPath = StorageUtils::getGraphPath(databasePath, "two_ns"); + auto graphConfig = *systemConfig; + graphConfig.autoCheckpoint = false; + graphConfig.forceCheckpointOnClose = false; + auto graphDatabase = std::make_unique(graphPath, graphConfig); + auto graphConnection = std::make_unique(graphDatabase.get()); + auto ddl = graphConnection->query("CREATE NODE TABLE Bookstand(id INT64 PRIMARY KEY);"); + ASSERT_TRUE(ddl->isSuccess()) << ddl->getErrorMessage(); + for (auto id = 10; id <= 12; id++) { + auto insert = graphConnection->query(std::format("CREATE (:Bookstand {{id: {}}});", id)); + ASSERT_TRUE(insert->isSuccess()) << insert->getErrorMessage(); + } + ddl.reset(); + graphConnection.reset(); + graphDatabase.reset(); + + createDBAndConn(); + auto openResult = conn->query("RETURN 1;"); + ASSERT_TRUE(openResult->isSuccess()) << openResult->getErrorMessage(); + ASSERT_EQ(openResult->getNext()->getValue(0)->getValue(), 1); + ASSERT_TRUE(conn->query("USE GRAPH two_ns;")->isSuccess()); + auto personResult = conn->query("MATCH (n:Person) RETURN COUNT(n);"); + ASSERT_TRUE(personResult->isSuccess()) << personResult->getErrorMessage(); + ASSERT_EQ(personResult->getNext()->getValue(0)->getValue(), 2); + auto bookstandResult = conn->query("MATCH (n:Bookstand) RETURN COUNT(n);"); + ASSERT_TRUE(bookstandResult->isSuccess()) << bookstandResult->getErrorMessage(); + ASSERT_EQ(bookstandResult->getNext()->getValue(0)->getValue(), 3); +} + // A standalone session's graph WAL records index operations in the standalone catalog's own // index-ID space, which starts empty because the parent session's graph-owned records live in // the main WAL. Parent recovery installs those parent-created indexes first, so replaying the @@ -5232,6 +5626,392 @@ TEST_F(ReviewFixesTest, TaggedSubgraphDropSkipsMainFloorIdentityLookup) { EXPECT_FALSE(subgraphs.contains("t")); } +// ALTER ADD/DROP FROM TO records carry the endpoint node-table IDs as bound at log time, +// and the ALTER-added connection is recorded under its physical rel-table OID. Natural +// graph-session recovery is identity-preserving (a rolled-back DDL rewinds the catalog +// counter), so this test guards the whole ALTER + replay lifecycle end to end under +// identity; CraftedAlterFromToEndpointsTranslateRecordedEntryIDs below forces a +// recorded/replayed ID gap and proves the translations themselves. +TEST_F(ReviewFixesTest, AlterFromToEndpointsResolveReplayedNodeIDs) { + if (inMemMode) { + GTEST_SKIP(); + } + + conn->query("CALL auto_checkpoint=false;"); + conn->query("CALL force_checkpoint_on_close=false;"); + ASSERT_TRUE(conn->query("CREATE GRAPH alter_replay;")->isSuccess()); + ASSERT_TRUE(conn->query("CHECKPOINT;")->isSuccess()); + + { + ASSERT_TRUE(conn->query("USE GRAPH alter_replay;")->isSuccess()); + ASSERT_TRUE(conn->query("BEGIN TRANSACTION;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE NODE TABLE consumed(id INT64 PRIMARY KEY);")->isSuccess()); + ASSERT_TRUE(conn->query("ROLLBACK;")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE NODE TABLE Person(id INT64 PRIMARY KEY);")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE NODE TABLE Hobby(id INT64 PRIMARY KEY);")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE REL TABLE Knows(FROM Person TO Person);")->isSuccess()); + for (auto id = 1; id <= 2; id++) { + auto insert = conn->query(std::format("CREATE (:Person {{id: {}}});", id)); + ASSERT_TRUE(insert->isSuccess()) << insert->getErrorMessage(); + } + auto initialRel = + conn->query("MATCH (a:Person {id: 1}), (b:Person {id: 2}) CREATE (a)-[:Knows]->(b);"); + ASSERT_TRUE(initialRel->isSuccess()) << initialRel->getErrorMessage(); + auto hobbyInsert = conn->query("CREATE (:Hobby {id: 9});"); + ASSERT_TRUE(hobbyInsert->isSuccess()) << hobbyInsert->getErrorMessage(); + auto alterResult = conn->query("ALTER TABLE Knows ADD FROM Hobby TO Person;"); + ASSERT_TRUE(alterResult->isSuccess()) << alterResult->getErrorMessage(); + auto addedRelResult = + conn->query("MATCH (h:Hobby {id: 9}), (p:Person {id: 1}) CREATE (h)-[:Knows]->(p);"); + ASSERT_TRUE(addedRelResult->isSuccess()) << addedRelResult->getErrorMessage(); + ASSERT_TRUE(conn->query("USE GRAPH main;")->isSuccess()); + } + + createDBAndConn(); + auto openResult = conn->query("RETURN 1;"); + ASSERT_TRUE(openResult->isSuccess()) << openResult->getErrorMessage(); + ASSERT_TRUE(conn->query("USE GRAPH alter_replay;")->isSuccess()); + auto personResult = conn->query("MATCH (n:Person) RETURN COUNT(n);"); + ASSERT_TRUE(personResult->isSuccess()) << personResult->getErrorMessage(); + ASSERT_EQ(personResult->getNext()->getValue(0)->getValue(), 2); + auto hobbyResult = conn->query("MATCH (n:Hobby) RETURN COUNT(n);"); + ASSERT_TRUE(hobbyResult->isSuccess()) << hobbyResult->getErrorMessage(); + ASSERT_EQ(hobbyResult->getNext()->getValue(0)->getValue(), 1); + auto initialRelResult = conn->query("MATCH (:Person)-[k:Knows]->(:Person) RETURN COUNT(k);"); + ASSERT_TRUE(initialRelResult->isSuccess()) << initialRelResult->getErrorMessage(); + ASSERT_EQ(initialRelResult->getNext()->getValue(0)->getValue(), 1); + auto addedRelCount = conn->query("MATCH (:Hobby)-[k:Knows]->(:Person) RETURN COUNT(k);"); + ASSERT_TRUE(addedRelCount->isSuccess()) << addedRelCount->getErrorMessage(); + ASSERT_EQ(addedRelCount->getNext()->getValue(0)->getValue(), 1); +} + +// The crafted WAL below appends owner-tagged records whose recorded entry IDs (node tables +// 100/200, group 300, per-connection rel-table OIDs 350/400) deliberately disagree with the +// IDs replay assigns in graph craft's empty catalog. CREATE replay translates the group's +// relTableInfos endpoints, ALTER ADD FROM TO translates its from/to IDs and re-registers the +// added connection under its recorded physical rel-table OID, and node/rel data records +// resolve their own table IDs and rel endpoints through the same map. Reading any of these +// raw binds the added connection to missing entries and misroutes the crafted rows. +TEST_F(ReviewFixesTest, CraftedAlterFromToEndpointsTranslateRecordedEntryID) { + if (inMemMode) { + GTEST_SKIP(); + } + + conn->query("CALL auto_checkpoint=false;"); + conn->query("CALL force_checkpoint_on_close=false;"); + { + ASSERT_TRUE(conn->query("CREATE GRAPH craft;")->isSuccess()); + ASSERT_TRUE(conn->query("CHECKPOINT;")->isSuccess()); + // Source entries for the crafted records; these tables replay into main's catalog. + ASSERT_TRUE(conn->query("CREATE NODE TABLE Person(id INT64 PRIMARY KEY);")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE NODE TABLE Hobby(id INT64 PRIMARY KEY);")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE REL TABLE Knows(FROM Person TO Person);")->isSuccess()); + } + + auto& context = *conn->getClientContext(); + BinaryData craftedWAL; + { + auto* memoryManager = MemoryManager::Get(context); + auto inMemWriter = std::make_shared(*memoryManager); + std::shared_ptr writer; + if (systemConfig->enableChecksums) { + writer = std::make_shared(inMemWriter, *memoryManager); + } else { + writer = inMemWriter; + } + Serializer serializer{writer}; + const auto appendRecord = [&serializer, &writer](const WALRecord& record) { + writer->onObjectBegin(); + WALRecord::serializeWithLength(serializer, record); + writer->onObjectEnd(); + }; + const auto appendCraftRecord = [&appendRecord](WALRecord& record) { + record.ownerCatalogName = "craft"; + appendRecord(record); + }; + + auto catalog = catalog::Catalog::Get(context); + auto personCopy = + catalog->getTableCatalogEntry(&transaction::DUMMY_CHECKPOINT_TRANSACTION, "Person") + ->copy(); + auto hobbyCopy = + catalog->getTableCatalogEntry(&transaction::DUMMY_CHECKPOINT_TRANSACTION, "Hobby") + ->copy(); + auto knowsCopy = + catalog->getTableCatalogEntry(&transaction::DUMMY_CHECKPOINT_TRANSACTION, "Knows") + ->copy(); + auto& knowsGroup = knowsCopy->cast(); + auto recordedInfos = knowsGroup.getRelEntryInfos(); + for (const auto& recordedInfo : recordedInfos) { + knowsGroup.dropFromToConnection(recordedInfo.nodePair.srcTableID, + recordedInfo.nodePair.dstTableID); + } + personCopy->setOID(100); + hobbyCopy->setOID(200); + knowsGroup.addFromToConnection(100, 100, 350); + knowsGroup.setOID(300); + + appendRecord(BeginTransactionRecord{}); + { + CreateCatalogEntryRecord personCreate(personCopy.get(), false); + appendCraftRecord(personCreate); + CreateCatalogEntryRecord hobbyCreate(hobbyCopy.get(), false); + appendCraftRecord(hobbyCreate); + CreateCatalogEntryRecord knowsCreate(knowsCopy.get(), false); + appendCraftRecord(knowsCreate); + binder::BoundAlterInfo alterInfo(AlterType::ADD_FROM_TO_CONNECTION, "Knows", + std::make_unique(200, 100)); + AlterTableEntryRecord alterRecord(&alterInfo, 400); + appendCraftRecord(alterRecord); + } + appendRecord(CommitRecord{}); + + appendRecord(BeginTransactionRecord{}); + { + const auto craftNodeInsert = [&appendCraftRecord, memoryManager]( + common::table_id_t recordedTableID, int64_t key) { + auto keyVector = std::make_unique(LogicalTypeID::INT64, memoryManager); + keyVector->setState(DataChunkState::getSingleValueDataChunkState()); + keyVector->setValue(0, key); + keyVector->setNull(0, false); + TableInsertionRecord insertRecord(recordedTableID, TableType::NODE, 1, + {keyVector.get()}); + appendCraftRecord(insertRecord); + }; + const auto craftRelInsert = [&appendCraftRecord, + memoryManager](common::table_id_t recordedTableID, + nodeID_t src, nodeID_t dst) { + auto srcVector = + std::make_unique(LogicalTypeID::INTERNAL_ID, memoryManager); + srcVector->setState(DataChunkState::getSingleValueDataChunkState()); + srcVector->setValue(0, src); + srcVector->setNull(0, false); + auto dstVector = + std::make_unique(LogicalTypeID::INTERNAL_ID, memoryManager); + dstVector->setState(DataChunkState::getSingleValueDataChunkState()); + dstVector->setValue(0, dst); + dstVector->setNull(0, false); + auto relIDVector = + std::make_unique(LogicalTypeID::INTERNAL_ID, memoryManager); + relIDVector->setState(DataChunkState::getSingleValueDataChunkState()); + relIDVector->setValue(0, nodeID_t{0, recordedTableID}); + relIDVector->setNull(0, false); + TableInsertionRecord insertRecord(recordedTableID, TableType::REL, 1, + {srcVector.get(), dstVector.get(), relIDVector.get()}); + appendCraftRecord(insertRecord); + }; + craftNodeInsert(100, 1); + craftNodeInsert(100, 2); + craftNodeInsert(200, 9); + craftRelInsert(350, nodeID_t{0, 100}, nodeID_t{1, 100}); + craftRelInsert(400, nodeID_t{0, 200}, nodeID_t{0, 100}); + } + appendRecord(CommitRecord{}); + + auto bufferWriter = std::make_shared(); + inMemWriter->flush(*bufferWriter); + craftedWAL = bufferWriter->getData(); + } + conn.reset(); + database.reset(); + + std::ofstream walFile{StorageUtils::getWALFilePath(databasePath), + std::ios::binary | std::ios::app}; + ASSERT_TRUE(walFile.is_open()); + walFile.write(reinterpret_cast(craftedWAL.data.get()), + static_cast(craftedWAL.size)); + walFile.close(); + ASSERT_TRUE(walFile.good()); + + createDBAndConn(); + auto openResult = conn->query("RETURN 1;"); + ASSERT_TRUE(openResult->isSuccess()) << openResult->getErrorMessage(); + ASSERT_TRUE(conn->query("USE GRAPH craft;")->isSuccess()); + auto personResult = conn->query("MATCH (n:Person) RETURN COUNT(n);"); + ASSERT_TRUE(personResult->isSuccess()) << personResult->getErrorMessage(); + ASSERT_EQ(personResult->getNext()->getValue(0)->getValue(), 2); + auto hobbyResult = conn->query("MATCH (n:Hobby) RETURN COUNT(n);"); + ASSERT_TRUE(hobbyResult->isSuccess()) << hobbyResult->getErrorMessage(); + ASSERT_EQ(hobbyResult->getNext()->getValue(0)->getValue(), 1); + auto initialRelResult = conn->query("MATCH (:Person)-[k:Knows]->(:Person) RETURN COUNT(k);"); + ASSERT_TRUE(initialRelResult->isSuccess()) << initialRelResult->getErrorMessage(); + ASSERT_EQ(initialRelResult->getNext()->getValue(0)->getValue(), 1); + auto addedRelResult = conn->query("MATCH (:Hobby)-[k:Knows]->(:Person) RETURN COUNT(k);"); + ASSERT_TRUE(addedRelResult->isSuccess()) << addedRelResult->getErrorMessage(); + ASSERT_EQ(addedRelResult->getNext()->getValue(0)->getValue(), 1); + // The neighbor reads resolve the stored dst node IDs by table ID, so they fail when a + // replayed row keeps a stale recorded endpoint. + auto initialNbrResult = + conn->query("MATCH (:Person {id: 1})-[k:Knows]->(p:Person) RETURN p.id;"); + ASSERT_TRUE(initialNbrResult->isSuccess()) << initialNbrResult->getErrorMessage(); + ASSERT_TRUE(initialNbrResult->hasNext()); + ASSERT_EQ(initialNbrResult->getNext()->getValue(0)->getValue(), 2); + auto addedNbrResult = conn->query("MATCH (:Hobby {id: 9})-[k:Knows]->(p:Person) RETURN p.id;"); + ASSERT_TRUE(addedNbrResult->isSuccess()) << addedNbrResult->getErrorMessage(); + ASSERT_TRUE(addedNbrResult->hasNext()); + ASSERT_EQ(addedNbrResult->getNext()->getValue(0)->getValue(), 1); + + common::table_id_t personID = common::INVALID_TABLE_ID; + common::table_id_t hobbyID = common::INVALID_TABLE_ID; + std::vector recoveredRelEntryInfos; + main::DatabaseManager::Get(*conn->getClientContext()) + ->withGraphCatalog("craft", [&](catalog::Catalog* graphCatalog) { + personID = + graphCatalog + ->getTableCatalogEntry(&transaction::DUMMY_CHECKPOINT_TRANSACTION, "Person") + ->getTableID(); + hobbyID = + graphCatalog + ->getTableCatalogEntry(&transaction::DUMMY_CHECKPOINT_TRANSACTION, "Hobby") + ->getTableID(); + recoveredRelEntryInfos = + graphCatalog + ->getTableCatalogEntry(&transaction::DUMMY_CHECKPOINT_TRANSACTION, "Knows") + ->constCast() + .getRelEntryInfos(); + }); + ASSERT_EQ(recoveredRelEntryInfos.size(), 2); + const auto hasPair = [&](common::table_id_t src, common::table_id_t dst) { + return std::any_of(recoveredRelEntryInfos.begin(), recoveredRelEntryInfos.end(), + [&](const catalog::RelTableCatalogInfo& info) { + return info.nodePair.srcTableID == src && info.nodePair.dstTableID == dst; + }); + }; + ASSERT_TRUE(hasPair(personID, personID)); + ASSERT_TRUE(hasPair(hobbyID, personID)); +} + +// The DROP arm of the ALTER endpoint translation. The crafted WAL logs a committed +// DROP_FROM_TO record whose recorded endpoints (Hobby 200 → Person 100) disagree with +// the IDs replay assigns in graph craftdrop's empty catalog. dropFromToConnection +// silently preserves a pair it cannot match, so replaying the stale recorded pair +// would keep both connections; with the translation, exactly the Hobby→Person +// connection disappears while Person→Person survives under its recovered IDs. +TEST_F(ReviewFixesTest, CraftedAlterDropFromToEndpointsTranslateRecordedEntryID) { + if (inMemMode) { + GTEST_SKIP(); + } + + conn->query("CALL auto_checkpoint=false;"); + conn->query("CALL force_checkpoint_on_close=false;"); + { + ASSERT_TRUE(conn->query("CREATE GRAPH craftdrop;")->isSuccess()); + ASSERT_TRUE(conn->query("CHECKPOINT;")->isSuccess()); + // Source entries for the crafted records; these tables replay into main's catalog. + ASSERT_TRUE(conn->query("CREATE NODE TABLE Person(id INT64 PRIMARY KEY);")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE NODE TABLE Hobby(id INT64 PRIMARY KEY);")->isSuccess()); + ASSERT_TRUE(conn->query("CREATE REL TABLE Knows(FROM Person TO Person);")->isSuccess()); + } + + auto& context = *conn->getClientContext(); + BinaryData craftedWAL; + { + auto* memoryManager = MemoryManager::Get(context); + auto inMemWriter = std::make_shared(*memoryManager); + std::shared_ptr writer; + if (systemConfig->enableChecksums) { + writer = std::make_shared(inMemWriter, *memoryManager); + } else { + writer = inMemWriter; + } + Serializer serializer{writer}; + const auto appendRecord = [&serializer, &writer](const WALRecord& record) { + writer->onObjectBegin(); + WALRecord::serializeWithLength(serializer, record); + writer->onObjectEnd(); + }; + const auto appendCraftRecord = [&appendRecord](WALRecord& record) { + record.ownerCatalogName = "craftdrop"; + appendRecord(record); + }; + + auto catalog = catalog::Catalog::Get(context); + auto personCopy = + catalog->getTableCatalogEntry(&transaction::DUMMY_CHECKPOINT_TRANSACTION, "Person") + ->copy(); + auto hobbyCopy = + catalog->getTableCatalogEntry(&transaction::DUMMY_CHECKPOINT_TRANSACTION, "Hobby") + ->copy(); + auto knowsCopy = + catalog->getTableCatalogEntry(&transaction::DUMMY_CHECKPOINT_TRANSACTION, "Knows") + ->copy(); + auto& knowsGroup = knowsCopy->cast(); + auto recordedInfos = knowsGroup.getRelEntryInfos(); + for (const auto& recordedInfo : recordedInfos) { + knowsGroup.dropFromToConnection(recordedInfo.nodePair.srcTableID, + recordedInfo.nodePair.dstTableID); + } + personCopy->setOID(100); + hobbyCopy->setOID(200); + knowsGroup.addFromToConnection(100, 100, 350); + knowsGroup.addFromToConnection(200, 100, 400); + knowsGroup.setOID(300); + + appendRecord(BeginTransactionRecord{}); + { + CreateCatalogEntryRecord personCreate(personCopy.get(), false); + appendCraftRecord(personCreate); + CreateCatalogEntryRecord hobbyCreate(hobbyCopy.get(), false); + appendCraftRecord(hobbyCreate); + CreateCatalogEntryRecord knowsCreate(knowsCopy.get(), false); + appendCraftRecord(knowsCreate); + binder::BoundAlterInfo alterInfo(AlterType::DROP_FROM_TO_CONNECTION, "Knows", + std::make_unique(200, 100)); + AlterTableEntryRecord alterRecord(&alterInfo); + appendCraftRecord(alterRecord); + } + appendRecord(CommitRecord{}); + + auto bufferWriter = std::make_shared(); + inMemWriter->flush(*bufferWriter); + craftedWAL = bufferWriter->getData(); + } + conn.reset(); + database.reset(); + + std::ofstream walFile{StorageUtils::getWALFilePath(databasePath), + std::ios::binary | std::ios::app}; + ASSERT_TRUE(walFile.is_open()); + walFile.write(reinterpret_cast(craftedWAL.data.get()), + static_cast(craftedWAL.size)); + walFile.close(); + ASSERT_TRUE(walFile.good()); + + createDBAndConn(); + auto openResult = conn->query("RETURN 1;"); + ASSERT_TRUE(openResult->isSuccess()) << openResult->getErrorMessage(); + ASSERT_TRUE(conn->query("USE GRAPH craftdrop;")->isSuccess()); + + common::table_id_t personID = common::INVALID_TABLE_ID; + common::table_id_t hobbyID = common::INVALID_TABLE_ID; + std::vector recoveredRelEntryInfos; + main::DatabaseManager::Get(*conn->getClientContext()) + ->withGraphCatalog("craftdrop", [&](catalog::Catalog* graphCatalog) { + personID = + graphCatalog + ->getTableCatalogEntry(&transaction::DUMMY_CHECKPOINT_TRANSACTION, "Person") + ->getTableID(); + hobbyID = + graphCatalog + ->getTableCatalogEntry(&transaction::DUMMY_CHECKPOINT_TRANSACTION, "Hobby") + ->getTableID(); + recoveredRelEntryInfos = + graphCatalog + ->getTableCatalogEntry(&transaction::DUMMY_CHECKPOINT_TRANSACTION, "Knows") + ->constCast() + .getRelEntryInfos(); + }); + ASSERT_EQ(recoveredRelEntryInfos.size(), 1); + const auto hasPair = [&](common::table_id_t src, common::table_id_t dst) { + return std::any_of(recoveredRelEntryInfos.begin(), recoveredRelEntryInfos.end(), + [&](const catalog::RelTableCatalogInfo& info) { + return info.nodePair.srcTableID == src && info.nodePair.dstTableID == dst; + }); + }; + ASSERT_TRUE(hasPair(personID, personID)); + ASSERT_FALSE(hasPair(hobbyID, personID)); +} + // The whole lifecycle in one explicit transaction makes recovery replay the drop inside // the same recovery transaction that created the dropped graph's catalog entries. The // transaction's undo records point into that catalog, so replay must unregister it diff --git a/test/transaction/wal_test.cpp b/test/transaction/wal_test.cpp index b2ef37e74b..7a6940dc43 100644 --- a/test/transaction/wal_test.cpp +++ b/test/transaction/wal_test.cpp @@ -4,6 +4,7 @@ #include #include "api_test/api_test.h" +#include "binder/ddl/bound_alter_info.h" #include "common/exception/io.h" #include "common/exception/runtime.h" #include "common/exception/storage.h" @@ -16,6 +17,7 @@ #include "storage/buffer_manager/memory_manager.h" #include "storage/storage_utils.h" #include "storage/wal/local_wal.h" +#include "storage/wal/record/alter_table_entry_record.h" #include "storage/wal/wal.h" #include @@ -374,26 +376,166 @@ TEST_F(WalTest, FrozenWALRemovalFailurePoisonsWALAfterDurableTruncate) { EXPECT_EQ(walCommitSequence, 0); } -TEST_F(WalTest, WALRecordDeserializeSkipsUnknownTrailingBytes) { - auto recordBuffer = std::make_shared(); - Serializer recordSerializer{recordBuffer}; - lbug::storage::CopyTableRecord record{123}; - record.serialize(recordSerializer); - recordSerializer.write(456); +// A record that ends exactly at its body (pre-owner-format writer) decodes with an empty +// owner name. Unknown bytes trailing the body are indistinguishable from a torn owner +// trailer, so they must be rejected instead of being absorbed into a garbage name. +TEST_F(WalTest, WALRecordDeserializeRejectsUnknownTrailingBytes) { + { + auto recordBuffer = std::make_shared(); + Serializer recordSerializer{recordBuffer}; + lbug::storage::CopyTableRecord record{123}; + record.serialize(recordSerializer); + + auto walBuffer = std::make_shared(); + Serializer walSerializer{walBuffer}; + walSerializer.write(recordBuffer->getSize()); + walSerializer.write(recordBuffer->getBlobData(), recordBuffer->getSize()); + + auto walData = walBuffer->getData(); + Deserializer deserializer{std::make_unique(walData.data.get(), walData.size)}; + auto deserialized = + lbug::storage::WALRecord::deserialize(deserializer, *conn->getClientContext()); + + ASSERT_EQ(deserialized->type, lbug::storage::WALRecordType::COPY_TABLE_RECORD); + EXPECT_EQ(deserialized->constCast().tableID, 123); + EXPECT_EQ(deserialized->ownerCatalogName, ""); + EXPECT_TRUE(deserializer.finished()); + } + { + auto recordBuffer = std::make_shared(); + Serializer recordSerializer{recordBuffer}; + lbug::storage::CopyTableRecord record{123}; + record.serialize(recordSerializer); + recordSerializer.write(456); + + auto walBuffer = std::make_shared(); + Serializer walSerializer{walBuffer}; + walSerializer.write(recordBuffer->getSize()); + walSerializer.write(recordBuffer->getBlobData(), recordBuffer->getSize()); + + auto walData = walBuffer->getData(); + Deserializer deserializer{std::make_unique(walData.data.get(), walData.size)}; + EXPECT_THROW(lbug::storage::WALRecord::deserialize(deserializer, *conn->getClientContext()), + lbug::common::RuntimeException); + } +} - auto walBuffer = std::make_shared(); - Serializer walSerializer{walBuffer}; - walSerializer.write(recordBuffer->getSize()); - walSerializer.write(recordBuffer->getBlobData(), recordBuffer->getSize()); +// Hand-crafted owner trailers pin the strict validation: a torn trailer must never +// decode into a name the replayer would route, while complete trailers — including +// empty ones — must keep decoding. +TEST_F(WalTest, WALRecordDeserializeValidatesOwnerTrailer) { + const auto deserializeWithTrailer = [&](const std::function& appendTrailer) { + auto recordBuffer = std::make_shared(); + Serializer recordSerializer{recordBuffer}; + lbug::storage::CopyTableRecord record{123}; + record.serialize(recordSerializer); + appendTrailer(recordSerializer); + + auto walBuffer = std::make_shared(); + Serializer walSerializer{walBuffer}; + walSerializer.write(recordBuffer->getSize()); + walSerializer.write(recordBuffer->getBlobData(), recordBuffer->getSize()); + + auto walData = walBuffer->getData(); + Deserializer deserializer{std::make_unique(walData.data.get(), walData.size)}; + return lbug::storage::WALRecord::deserialize(deserializer, *conn->getClientContext()); + }; - auto walData = walBuffer->getData(); - Deserializer deserializer{std::make_unique(walData.data.get(), walData.size)}; - auto deserialized = - lbug::storage::WALRecord::deserialize(deserializer, *conn->getClientContext()); + { + auto deserialized = deserializeWithTrailer( + [](Serializer& serializer) { serializer.write("craft"); }); + EXPECT_EQ(deserialized->ownerCatalogName, "craft"); + } + { + auto deserialized = + deserializeWithTrailer([](Serializer& serializer) { serializer.write(0); }); + EXPECT_EQ(deserialized->ownerCatalogName, ""); + } + { + EXPECT_THROW(deserializeWithTrailer([](Serializer& serializer) { + serializer.write(std::numeric_limits::max()); + const std::string name = "ABCD"; + serializer.write((const uint8_t*)name.data(), name.size()); + }), + lbug::common::RuntimeException); + } + { + EXPECT_THROW(deserializeWithTrailer([](Serializer& serializer) { + serializer.write(0); + const uint8_t garbage[8] = {0xEE, 0xEE, 0xEE, 0xEE, 0xEE, 0xEE, 0xEE, 0xEE}; + serializer.write(garbage, sizeof(garbage)); + }), + lbug::common::RuntimeException); + } + { + EXPECT_THROW(deserializeWithTrailer([](Serializer& serializer) { + serializer.write(std::string("AB\0D", 4)); + }), + lbug::common::RuntimeException); + } + { + EXPECT_THROW(deserializeWithTrailer([](Serializer& serializer) { + const uint8_t tornZeros[3] = {0x00, 0x00, 0x00}; + serializer.write(tornZeros, sizeof(tornZeros)); + }), + lbug::common::RuntimeException); + } +} - ASSERT_EQ(deserialized->type, lbug::storage::WALRecordType::COPY_TABLE_RECORD); - EXPECT_EQ(deserialized->constCast().tableID, 123); - EXPECT_TRUE(deserializer.finished()); +// A WAL written by the pre-extension revision carries ALTER records as +// [body][owner trailer] with no addedRelTableOID field. Short owners keep decoding +// exactly, with the field left at its INVALID_TABLE_ID default; an owner of eight or +// more bytes is indistinguishable from the extension, so the record is rejected and +// replay truncates at the last commit instead of replaying with a misread field. +TEST_F(WalTest, WALRecordDeserializeDecodesHeadFormatAlterTableEntry) { + const auto deserializeHeadFormat = [&](const std::string& ownerName) { + lbug::binder::BoundAlterInfo alterInfo(AlterType::ADD_FROM_TO_CONNECTION, "Knows", + std::make_unique(200, 100)); + lbug::storage::AlterTableEntryRecord record(&alterInfo); + record.ownerCatalogName = ownerName; + + auto walBuffer = std::make_shared(); + Serializer walSerializer{walBuffer}; + lbug::storage::WALRecord::serializeWithLength(walSerializer, record); + auto walData = walBuffer->getData(); + + uint64_t recordLength = 0; + std::memcpy(&recordLength, walData.data.get(), sizeof(recordLength)); + const auto bodyEnd = + recordLength - sizeof(uint64_t) * 2 - static_cast(ownerName.size()); + auto headBuffer = std::make_shared(); + Serializer headSerializer{headBuffer}; + headSerializer.write(recordLength - sizeof(uint64_t)); + headSerializer.write(walData.data.get() + sizeof(recordLength), bodyEnd); + headSerializer.write(walData.data.get() + sizeof(recordLength) + bodyEnd + sizeof(uint64_t), + recordLength - bodyEnd - sizeof(uint64_t)); + + auto headData = headBuffer->getData(); + Deserializer deserializer{ + std::make_unique(headData.data.get(), headData.size)}; + return lbug::storage::WALRecord::deserialize(deserializer, *conn->getClientContext()); + }; + + { + auto deserialized = deserializeHeadFormat(""); + const auto& alterRecord = deserialized->constCast(); + EXPECT_EQ(alterRecord.addedRelTableOID, lbug::common::INVALID_TABLE_ID); + EXPECT_EQ(deserialized->ownerCatalogName, ""); + } + { + auto deserialized = deserializeHeadFormat("graph"); + const auto& alterRecord = deserialized->constCast(); + EXPECT_EQ(alterRecord.addedRelTableOID, lbug::common::INVALID_TABLE_ID); + EXPECT_EQ(deserialized->ownerCatalogName, "graph"); + } + { EXPECT_THROW(deserializeHeadFormat("12345678"), lbug::common::RuntimeException); } +} + +TEST_F(WalTest, GraphNameContainingNullByteIsRejected) { + const std::string query = "CREATE GRAPH `" + std::string(1, '\0') + "`;"; + auto result = conn->query(query); + EXPECT_FALSE(result->isSuccess()); + EXPECT_THAT(result->getErrorMessage(), testing::HasSubstr("null byte")); } // A pre-named-format binary wrote UPDATE_SEQUENCE records as [sequenceID][kCount] @@ -461,18 +603,20 @@ TEST_F(WalTest, WALRecordDeserializeRoundTripsNamedUpdateSequenceRecord) { // the current reader (e.g. v43) expects. The deserializer must gracefully handle the size mismatch // by truncating reads at the declared record boundary and zero-filling any remaining bytes in the // new field. This allows silent forward-compatible migration without corrupting the next record. +// Declared lengths that reach past the body into a partial owner trailer are torn frames and are +// rejected by the strict trailer validation instead. TEST_F(WalTest, WALRecordDeserializeHandlesSizeMismatch) { auto recordBuffer = std::make_shared(); Serializer recordSerializer{recordBuffer}; lbug::storage::CopyTableRecord record{123}; record.serialize(recordSerializer); - recordSerializer.write(456); - // Declare a record length one byte shorter than the actual serialized size, simulating - // an older version that wrote a smaller struct. + // Declare a record length 4 bytes shorter than the actual serialized size, simulating + // an older version that wrote a smaller struct. The high 4 bytes of tableID fall past + // the declared boundary and must be zero-filled. const auto realRecordSize = recordBuffer->getSize(); - ASSERT_GT(realRecordSize, 1u); - const auto truncatedLength = realRecordSize - 1; + ASSERT_GT(realRecordSize, 4u); + const auto truncatedLength = realRecordSize - 4; auto walBuffer = std::make_shared(); Serializer walSerializer{walBuffer}; @@ -486,8 +630,11 @@ TEST_F(WalTest, WALRecordDeserializeHandlesSizeMismatch) { auto deserialized = lbug::storage::WALRecord::deserialize(deserializer, *conn->getClientContext()); ASSERT_EQ(deserialized->type, lbug::storage::WALRecordType::COPY_TABLE_RECORD); + // tableID 123 fits entirely in its low 4 bytes, so zero-filling preserves its value. EXPECT_EQ(deserialized->constCast().tableID, 123); + EXPECT_EQ(deserialized->ownerCatalogName, ""); // The stream should be positioned at the declared record boundary, not past it. + EXPECT_EQ(deserializer.getReadOffset(), 8u + truncatedLength); EXPECT_FALSE(deserializer.finished()); }