diff --git a/.clang-tidy b/.clang-tidy index 6f24252a8d0..3ff20bd1f31 100644 --- a/.clang-tidy +++ b/.clang-tidy @@ -6,12 +6,16 @@ Checks: > bugprone-chained-comparison, bugprone-copy-constructor-init, bugprone-dangling-handle, + bugprone-fold-init-type, + bugprone-forwarding-reference-overload, bugprone-implicit-widening-of-multiplication-result, bugprone-inaccurate-erase, bugprone-infinite-loop, + bugprone-integer-division, bugprone-macro-repeated-side-effects, bugprone-misplaced-widening-cast, bugprone-move-forwarding-reference, + bugprone-posix-return, bugprone-redundant-branch-condition, bugprone-return-const-ref-from-parameter, bugprone-shared-ptr-array-mismatch, @@ -29,6 +33,7 @@ Checks: > bugprone-swapped-arguments, bugprone-too-small-loop-variable, bugprone-undefined-memory-manipulation, + bugprone-unhandled-self-assignment, bugprone-unique-ptr-array-mismatch, bugprone-use-after-move, bugprone-virtual-near-miss, diff --git a/.github/workflows/tidy.yml b/.github/workflows/tidy.yml index 62cbf2cdf00..eccacfd0a8b 100644 --- a/.github/workflows/tidy.yml +++ b/.github/workflows/tidy.yml @@ -46,6 +46,7 @@ jobs: fdboptions \ ProtocolVersion \ fdb_c_generated \ + fdb_c_options \ fdb-java # all protobuf headers diff --git a/CMakeLists.txt b/CMakeLists.txt index 184e28a9c4c..c7bb603fe48 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -40,6 +40,9 @@ if("${PROJECT_SOURCE_DIR}" STREQUAL "${PROJECT_BINARY_DIR}") message(FATAL_ERROR "In-source builds are forbidden") endif() +set(FDB_RELEASE OFF CACHE BOOL "This is a building of a final release") +set(FDB_RELEASE_CANDIDATE OFF CACHE BOOL "This is a building of a release candidate") + set(OPEN_FOR_IDE OFF CACHE BOOL "Open this in an IDE (won't compile/link)") @@ -54,23 +57,18 @@ if(NOT CMAKE_BUILD_TYPE AND NOT CMAKE_CONFIGURATION_TYPES) Debug CACHE STRING "Choose the type of build" FORCE) else() - message(STATUS "Setting build type to 'Release' as none was specified") + message(STATUS "Setting build type to 'Release'") set(CMAKE_BUILD_TYPE Release CACHE STRING "Choose the type of build" FORCE) - set_property(CACHE CMAKE_BUILD_TYPE PROPERTY STRINGS "Debug" "Release" - "MinSizeRel" "RelWithDebInfo") endif() + set_property(CACHE CMAKE_BUILD_TYPE PROPERTY STRINGS "Debug" "Release" + "MinSizeRel" "RelWithDebInfo") endif() set(EXECUTABLE_OUTPUT_PATH ${PROJECT_BINARY_DIR}/bin) set(LIBRARY_OUTPUT_PATH ${PROJECT_BINARY_DIR}/lib) -option(USE_SCCACHE "Use sccache if found" ON) -if(USE_SCCACHE) - find_package(sccache) -endif() - option(FDB_MEMORY_TRACKER "Compile in the sampled per-call-site memory tracker (flow/MemoryTracker)" ON) if(NOT FDB_MEMORY_TRACKER) message(STATUS "Building FoundationDB with the memory tracker compiled out") diff --git a/bindings/c/CMakeLists.txt b/bindings/c/CMakeLists.txt index e5b6a8ec0fb..2969bcd90e2 100644 --- a/bindings/c/CMakeLists.txt +++ b/bindings/c/CMakeLists.txt @@ -354,15 +354,17 @@ if(NOT WIN32) ) endforeach() - add_python_venv_test(NAME fdb_c_upgrade_to_future_version - COMMAND python -m fdb_test_runner.upgrade_test - --build-dir ${CMAKE_BINARY_DIR} - --test-file ${CMAKE_SOURCE_DIR}/bindings/c/test/apitester/tests/upgrade/MixedApiWorkloadMultiThr.toml - --upgrade-path "${FDB_CURRENT_VERSION}" "${FDB_FUTURE_VERSION}" "${FDB_CURRENT_VERSION}" - --process-number 3 - ) + if(CMAKE_SYSTEM_NAME STREQUAL "Linux") + add_python_venv_test(NAME fdb_c_upgrade_to_future_version + COMMAND python -m fdb_test_runner.upgrade_test + --build-dir ${CMAKE_BINARY_DIR} + --test-file ${CMAKE_SOURCE_DIR}/bindings/c/test/apitester/tests/upgrade/MixedApiWorkloadMultiThr.toml + --upgrade-path "${FDB_CURRENT_VERSION}" "${FDB_FUTURE_VERSION}" "${FDB_CURRENT_VERSION}" + --process-number 3 + ) + endif() - if(CMAKE_SYSTEM_PROCESSOR STREQUAL "x86_64" AND NOT USE_SANITIZER) + if(CMAKE_SYSTEM_NAME STREQUAL "Linux" AND CMAKE_SYSTEM_PROCESSOR STREQUAL "x86_64" AND NOT USE_SANITIZER) add_python_venv_test(NAME fdb_c_client_config_tests COMMAND python ${CMAKE_CURRENT_SOURCE_DIR}/test/fdb_c_client_config_tests.py --build-dir ${CMAKE_BINARY_DIR} diff --git a/bindings/c/fdb_c.cpp b/bindings/c/fdb_c.cpp index 041efd33dd3..a0293d92809 100644 --- a/bindings/c/fdb_c.cpp +++ b/bindings/c/fdb_c.cpp @@ -95,6 +95,29 @@ FDBKeyRange copyNativeCdcKeyRange(Arena& arena, KeyRangeRef source) { return FDBKeyRange{ begin.begin(), begin.size(), end.begin(), end.size() }; } +std::vector copyNativeCdcRanges(FDBKeyRange const* ranges, int rangeCount) { + if (rangeCount <= 0 || rangeCount > NATIVE_CDC_MAX_RANGES || ranges == nullptr) { + throw client_invalid_operation(); + } + std::vector result; + result.reserve(rangeCount); + for (int i = 0; i < rangeCount; ++i) { + auto const& range = ranges[i]; + if (range.begin_key_length < 0 || range.end_key_length < 0 || + (range.begin_key_length > 0 && range.begin_key == nullptr) || + (range.end_key_length > 0 && range.end_key == nullptr)) { + throw client_invalid_operation(); + } + KeyRef begin(range.begin_key, range.begin_key_length); + KeyRef end(range.end_key, range.end_key_length); + if (begin >= end) { + throw client_invalid_operation(); + } + result.emplace_back(KeyRangeRef(begin, end)); + } + return result; +} + CNativeCdcStreamInfoArray makeCNativeCdcStreamInfoArray(std::vector const& source) { CNativeCdcStreamInfoArray result; result.streams.reserve(result.arena, source.size()); @@ -102,7 +125,12 @@ CNativeCdcStreamInfoArray makeCNativeCdcStreamInfoArray(std::vectorregisterNativeCdcStream( - KeyRef(name, name_length), - KeyRangeRef(KeyRef(begin_key, begin_key_length), KeyRef(end_key, end_key_length))) - .extractPtr());); + if (name_length <= 0 || name == nullptr) { throw client_invalid_operation(); } auto rangesCopy = + copyNativeCdcRanges(ranges, range_count); + return (FDBFuture*)(DB(db)->registerNativeCdcStream(KeyRef(name, name_length), rangesCopy).extractPtr());); } extern "C" DLLEXPORT FDBFuture* fdb_database_remove_cdc_stream(FDBDatabase* db, uint8_t const* name, int name_length) { diff --git a/bindings/c/foundationdb/CppWorkload.h b/bindings/c/foundationdb/CppWorkload.h index b0a430bfbba..ad91ea24648 100644 --- a/bindings/c/foundationdb/CppWorkload.h +++ b/bindings/c/foundationdb/CppWorkload.h @@ -25,6 +25,7 @@ #include #include #include +#include #ifndef DLLEXPORT #if defined(_MSC_VER) @@ -81,7 +82,8 @@ class GenericPromise { std::shared_ptr impl; public: - template + template , Ptr&&>::value, int>::type = 0> explicit GenericPromise(Ptr&& impl) : impl(std::forward(impl)) {} void send(T val) { impl->send(&val); } }; diff --git a/bindings/c/foundationdb/fdb_c.h b/bindings/c/foundationdb/fdb_c.h index cf344d51df3..05c1d045bcb 100644 --- a/bindings/c/foundationdb/fdb_c.h +++ b/bindings/c/foundationdb/fdb_c.h @@ -213,7 +213,8 @@ typedef enum { typedef struct cdc_stream_info { FDBKey name; uint64_t stream_id; - FDBKeyRange key_range; + const FDBKeyRange* ranges; + int range_count; int64_t min_version; } FDBCdcStreamInfo; @@ -437,10 +438,8 @@ DLLEXPORT WARN_UNUSED_RESULT fdb_error_t fdb_database_create_transaction(FDBData DLLEXPORT WARN_UNUSED_RESULT FDBFuture* fdb_database_register_cdc_stream(FDBDatabase* db, uint8_t const* name, int name_length, - uint8_t const* begin_key, - int begin_key_length, - uint8_t const* end_key, - int end_key_length); + FDBKeyRange const* ranges, + int range_count); DLLEXPORT WARN_UNUSED_RESULT FDBFuture* fdb_database_remove_cdc_stream(FDBDatabase* db, uint8_t const* name, diff --git a/bindings/c/test/fdb_api.hpp b/bindings/c/test/fdb_api.hpp index f2f9e4235fe..d9ef6659171 100644 --- a/bindings/c/test/fdb_api.hpp +++ b/bindings/c/test/fdb_api.hpp @@ -139,6 +139,8 @@ CharsRef toCharsRef(const std::optional>& s) noexcept { [[maybe_unused]] constexpr const bool OverflowCheck = false; +// clang-tidy does not recognize size() through intSize(); call-site suppressions mark +// byte buffers whose lengths are passed explicitly to the C API or KeySelector. inline int intSize(size_t size) { if constexpr (OverflowCheck) { if (size > static_cast(std::numeric_limits::max())) @@ -278,6 +280,7 @@ inline int maxApiVersion() { namespace network { inline Error setOptionNothrow(FDBNetworkOption option, BytesRef str) noexcept { + // NOLINTNEXTLINE(bugprone-suspicious-stringview-data-usage) return Error(native::fdb_network_set_option(option, str.data(), intSize(str.size()))); } @@ -345,7 +348,7 @@ class Result { friend class Transaction; std::shared_ptr r; - Result(native::FDBResult* result) { + explicit Result(native::FDBResult* result) { if (result) r = std::shared_ptr(result, &native::fdb_result_destroy); } @@ -377,7 +380,7 @@ class Future { friend std::hash; std::shared_ptr f; - Future(native::FDBFuture* future) { + explicit(false) Future(native::FDBFuture* future) { if (future) f = std::shared_ptr(future, &native::fdb_future_destroy); } @@ -473,7 +476,7 @@ class TypedFuture : public Future { using Future::get; using Future::getNothrow; using Future::then; - TypedFuture(const Future& f) noexcept : Future(f) {} + explicit TypedFuture(const Future& f) noexcept : Future(f) {} public: using ContainedType = typename VarTraits::Type; @@ -500,18 +503,22 @@ struct KeySelector { namespace key_select { inline KeySelector firstGreaterThan(KeyRef key, int offset = 0) { + // NOLINTNEXTLINE(bugprone-suspicious-stringview-data-usage) return KeySelector{ FDB_KEYSEL_FIRST_GREATER_THAN(key.data(), intSize(key.size())) + offset }; } inline KeySelector firstGreaterOrEqual(KeyRef key, int offset = 0) { + // NOLINTNEXTLINE(bugprone-suspicious-stringview-data-usage) return KeySelector{ FDB_KEYSEL_FIRST_GREATER_OR_EQUAL(key.data(), intSize(key.size())) + offset }; } inline KeySelector lastLessThan(KeyRef key, int offset = 0) { + // NOLINTNEXTLINE(bugprone-suspicious-stringview-data-usage) return KeySelector{ FDB_KEYSEL_LAST_LESS_THAN(key.data(), intSize(key.size())) + offset }; } inline KeySelector lastLessOrEqual(KeyRef key, int offset = 0) { + // NOLINTNEXTLINE(bugprone-suspicious-stringview-data-usage) return KeySelector{ FDB_KEYSEL_LAST_LESS_OR_EQUAL(key.data(), intSize(key.size())) + offset }; } @@ -549,6 +556,7 @@ class Transaction { } Error setOptionNothrow(FDBTransactionOption option, BytesRef str) noexcept { + // NOLINTNEXTLINE(bugprone-suspicious-stringview-data-usage) return Error(native::fdb_transaction_set_option(tr.get(), option, str.data(), intSize(str.size()))); } @@ -600,6 +608,7 @@ class Transaction { } TypedFuture get(KeyRef key, bool snapshot) { + // NOLINTNEXTLINE(bugprone-suspicious-stringview-data-usage) return native::fdb_transaction_get(tr.get(), key.data(), intSize(key.size()), snapshot); } @@ -631,6 +640,7 @@ class Transaction { } TypedFuture watch(KeyRef key) { + // NOLINTNEXTLINE(bugprone-suspicious-stringview-data-usage) return native::fdb_transaction_watch(tr.get(), key.data(), intSize(key.size())); } @@ -643,26 +653,36 @@ class Transaction { void cancel() { return native::fdb_transaction_cancel(tr.get()); } void set(KeyRef key, ValueRef value) { + // NOLINTNEXTLINE(bugprone-suspicious-stringview-data-usage) native::fdb_transaction_set(tr.get(), key.data(), intSize(key.size()), value.data(), intSize(value.size())); } void atomicOp(KeyRef key, ValueRef param, FDBMutationType operationType) { + // NOLINTBEGIN(bugprone-suspicious-stringview-data-usage) native::fdb_transaction_atomic_op( tr.get(), key.data(), intSize(key.size()), param.data(), intSize(param.size()), operationType); + // NOLINTEND(bugprone-suspicious-stringview-data-usage) } - void clear(KeyRef key) { native::fdb_transaction_clear(tr.get(), key.data(), intSize(key.size())); } + void clear(KeyRef key) { + // NOLINTNEXTLINE(bugprone-suspicious-stringview-data-usage) + native::fdb_transaction_clear(tr.get(), key.data(), intSize(key.size())); + } void clearRange(KeyRef begin, KeyRef end) { + // NOLINTBEGIN(bugprone-suspicious-stringview-data-usage) native::fdb_transaction_clear_range( tr.get(), begin.data(), intSize(begin.size()), end.data(), intSize(end.size())); + // NOLINTEND(bugprone-suspicious-stringview-data-usage) } void addConflictRange(KeyRef begin, KeyRef end, FDBConflictRangeType rangeType) { + // NOLINTBEGIN(bugprone-suspicious-stringview-data-usage) if (auto err = Error(native::fdb_transaction_add_conflict_range( tr.get(), begin.data(), intSize(begin.size()), end.data(), intSize(end.size()), rangeType))) { throwError("fdb_transaction_add_conflict_range returned error: ", err); } + // NOLINTEND(bugprone-suspicious-stringview-data-usage) } void addReadConflictRange(KeyRef begin, KeyRef end) { addConflictRange(begin, end, FDB_CONFLICT_RANGE_TYPE_READ); } @@ -686,7 +706,7 @@ class Database : public IDatabaseOps { public: Database(const Database&) noexcept = default; Database& operator=(const Database&) noexcept = default; - Database(const std::string& cluster_file_path) : db(nullptr) { + explicit Database(const std::string& cluster_file_path) : db(nullptr) { auto db_raw = static_cast(nullptr); if (auto err = Error(native::fdb_create_database(cluster_file_path.c_str(), &db_raw))) throwError(fmt::format("Failed to create database with '{}': ", cluster_file_path), err); @@ -708,6 +728,7 @@ class Database : public IDatabaseOps { } Error setOptionNothrow(FDBDatabaseOption option, BytesRef str) noexcept { + // NOLINTNEXTLINE(bugprone-suspicious-stringview-data-usage) return Error(native::fdb_database_set_option(db.get(), option, str.data(), intSize(str.size()))); } diff --git a/bindings/c/test/mako/admin_server.hpp b/bindings/c/test/mako/admin_server.hpp index b9b819fb6ca..e5ba4bdd9f3 100644 --- a/bindings/c/test/mako/admin_server.hpp +++ b/bindings/c/test/mako/admin_server.hpp @@ -95,7 +95,7 @@ class AdminServer { } public: - AdminServer(const Arguments& args) + explicit AdminServer(const Arguments& args) : args(args), server_pid(-1), pipe_to_server(boost::process::pipe()), pipe_to_client(boost::process::pipe()) { start(); } diff --git a/bindings/c/test/mako/async.cpp b/bindings/c/test/mako/async.cpp index 25bd8640254..21682e87b94 100644 --- a/bindings/c/test/mako/async.cpp +++ b/bindings/c/test/mako/async.cpp @@ -101,15 +101,15 @@ void ResumableStateForRunWorkload::postNextTick() { void ResumableStateForRunWorkload::runOneTick() { assert(iter != OpEnd); + auto f = Future{}; + // to minimize context switch overhead, repeat immediately completed ops + // in a loop, not an async continuation. +repeat_immediate_steps: if (iter.step == 0 /* first step */) prepareKeys(iter.op, key1, key2, args); watch_step.start(); if (iter.step == 0) watch_op = Stopwatch(watch_step.getStart()); - auto f = Future{}; - // to minimize context switch overhead, repeat immediately completed ops - // in a loop, not an async continuation. -repeat_immediate_steps: f = opTable[iter.op].stepFunction(iter.step)(tx, args, key1, key2, val); if (!f) { // immediately completed client-side ops: e.g. set, setrange, clear, clearrange, ... diff --git a/bindings/c/test/mako/ddsketch.hpp b/bindings/c/test/mako/ddsketch.hpp index 72cf4738a08..0c429c32e7d 100644 --- a/bindings/c/test/mako/ddsketch.hpp +++ b/bindings/c/test/mako/ddsketch.hpp @@ -220,7 +220,7 @@ class DDSketch : public DDSketchBase, T> { template class DDSketchSlow : public DDSketchBase, T> { public: - DDSketchSlow(double errorGuarantee = 0.1) + explicit DDSketchSlow(double errorGuarantee = 0.1) : DDSketchBase, T>(errorGuarantee), gamma((1.0 + errorGuarantee) / (1.0 - errorGuarantee)), logGamma(log(gamma)) { offset = getIndex(1.0 / DDSketchBase, T>::EPS) + 5; diff --git a/bindings/c/test/mako/future.hpp b/bindings/c/test/mako/future.hpp index 99b1956eb8a..97461b3d73d 100644 --- a/bindings/c/test/mako/future.hpp +++ b/bindings/c/test/mako/future.hpp @@ -36,7 +36,7 @@ enum class FutureRC { OK, RETRY, ABORT }; struct LogContext { static constexpr const bool do_log = true; - LogContext(std::string_view step) noexcept : step(step), transaction_timeout_expected(false) {} + explicit LogContext(std::string_view step) noexcept : step(step), transaction_timeout_expected(false) {} LogContext(std::string_view step, bool transaction_timeout_expected) noexcept : step(step), transaction_timeout_expected(transaction_timeout_expected) {} std::string_view step; diff --git a/bindings/c/test/mako/mako.cpp b/bindings/c/test/mako/mako.cpp index 01859e5300f..0d8f50b0914 100644 --- a/bindings/c/test/mako/mako.cpp +++ b/bindings/c/test/mako/mako.cpp @@ -2037,18 +2037,18 @@ void printReport(Arguments const& args, } const auto warmup_duration_sec = warmup_snapshot.has_value() ? warmup_snapshot->duration_sec : 0.0; const auto measurement_duration_sec = std::max(run_duration_sec - warmup_duration_sec, 1e-9); + const auto num_worker_threads = static_cast(args.num_processes) * args.num_threads; double cpu_time_worker_threads = - std::accumulate(thread_stats, - thread_stats + args.num_processes * args.num_threads, - 0.0, - [](double x, const ThreadStatistics& s) { return x + s.getCPUTime(); }); + std::accumulate(thread_stats, thread_stats + num_worker_threads, 0.0, [](double x, const ThreadStatistics& s) { + return x + s.getCPUTime(); + }); double total_duration_worker_threads = std::accumulate(thread_stats, - thread_stats + args.num_processes * args.num_threads, + thread_stats + num_worker_threads, 0.0, [](double x, const ThreadStatistics& s) { return x + s.getTotalDuration(); }) / - (args.num_processes * args.num_threads); // average + num_worker_threads; // average double cpu_util_worker_threads = 100. * cpu_time_worker_threads / total_duration_worker_threads; @@ -2348,7 +2348,7 @@ int statsProcessMain(Arguments const& args, throttle_factor = 1 - (sin_factor * (1.0 - (tpsmin / tpsmax))); break; case TPS_SQUARE: - if (pos < (args.tpsinterval / 2)) { + if (pos < tpsinterval / 2.0) { /* set to max */ throttle_factor = 1.0; } else { diff --git a/bindings/c/test/mako/stats.hpp b/bindings/c/test/mako/stats.hpp index 6348da9515b..42963b48f1b 100644 --- a/bindings/c/test/mako/stats.hpp +++ b/bindings/c/test/mako/stats.hpp @@ -311,7 +311,7 @@ class CPUUtilizationTimer { TimerKind kind; public: - CPUUtilizationTimer(TimerKind kind) : kind(kind) {} + explicit CPUUtilizationTimer(TimerKind kind) : kind(kind) {} void start() { timepoint_start = steady_clock::now(); cpu_time_start = (kind == THREAD) ? getProcessorTimeThread() : getProcessorTimeProcess(); diff --git a/bindings/c/test/mako/time.hpp b/bindings/c/test/mako/time.hpp index d962421ac12..f45ee86d9d4 100644 --- a/bindings/c/test/mako/time.hpp +++ b/bindings/c/test/mako/time.hpp @@ -53,8 +53,8 @@ class Stopwatch { public: Stopwatch() noexcept : p1(), p2() {} - Stopwatch(StartAtCtor) noexcept { start(); } - Stopwatch(timepoint_t start_time) noexcept : p1(start_time), p2() {} + explicit Stopwatch(StartAtCtor) noexcept { start(); } + explicit Stopwatch(timepoint_t start_time) noexcept : p1(start_time), p2() {} Stopwatch(const Stopwatch&) noexcept = default; Stopwatch& operator=(const Stopwatch&) noexcept = default; timepoint_t getStart() const noexcept { return p1; } diff --git a/bindings/c/test/mako/utils.hpp b/bindings/c/test/mako/utils.hpp index 80ee8dfc261..dbec5a1f5d0 100644 --- a/bindings/c/test/mako/utils.hpp +++ b/bindings/c/test/mako/utils.hpp @@ -163,7 +163,7 @@ class ExitGuard { std::decay_t fn; public: - ExitGuard(Func&& fn) : fn(std::forward(fn)) {} + explicit ExitGuard(Func&& fn) : fn(std::forward(fn)) {} ~ExitGuard() { fn(); } }; @@ -174,7 +174,7 @@ class FailGuard { std::decay_t fn; public: - FailGuard(Func&& fn) : fn(std::forward(fn)) {} + explicit FailGuard(Func&& fn) : fn(std::forward(fn)) {} ~FailGuard() { if (std::uncaught_exceptions()) { diff --git a/bindings/c/test/ryw_benchmark.c b/bindings/c/test/ryw_benchmark.c index 2a20bcd46a8..ba9ac4915cd 100644 --- a/bindings/c/test/ryw_benchmark.c +++ b/bindings/c/test/ryw_benchmark.c @@ -179,7 +179,8 @@ int singleClearGetRange(FDBTransaction* tr, struct ResultSet* rs) { double end = getTime(); insertData(tr); - return 100 * numKeys / 2 / (end - start); + const int keysRead = 100 * numKeys / 2; + return keysRead / (end - start); } int clearRangeGetRange(FDBTransaction* tr, struct ResultSet* rs) { @@ -221,7 +222,8 @@ int clearRangeGetRange(FDBTransaction* tr, struct ResultSet* rs) { double end = getTime(); insertData(tr); - return 100 * numKeys * 3 / 4 / (end - start); + const int keysRead = 100 * numKeys * 3 / 4; + return keysRead / (end - start); } int interleavedSetsGets(FDBTransaction* tr, struct ResultSet* rs) { diff --git a/bindings/c/test/unit/fdb_api.hpp b/bindings/c/test/unit/fdb_api.hpp index 335b62dc62f..cafccdb335f 100644 --- a/bindings/c/test/unit/fdb_api.hpp +++ b/bindings/c/test/unit/fdb_api.hpp @@ -73,7 +73,7 @@ class Future { // } protected: - Future(FDBFuture* f) : future_(f) {} + explicit Future(FDBFuture* f) : future_(f) {} FDBFuture* future_; }; @@ -86,7 +86,7 @@ class Int64Future : public Future { private: friend class Transaction; friend class Database; - Int64Future(FDBFuture* f) : Future(f) {} + explicit Int64Future(FDBFuture* f) : Future(f) {} }; class DoubleFuture : public Future { @@ -98,7 +98,7 @@ class DoubleFuture : public Future { private: friend class Transaction; friend class Database; - DoubleFuture(FDBFuture* f) : Future(f) {} + explicit DoubleFuture(FDBFuture* f) : Future(f) {} }; class KeyFuture : public Future { @@ -110,7 +110,7 @@ class KeyFuture : public Future { private: friend class Transaction; friend class Database; - KeyFuture(FDBFuture* f) : Future(f) {} + explicit KeyFuture(FDBFuture* f) : Future(f) {} }; class ValueFuture : public Future { @@ -121,7 +121,7 @@ class ValueFuture : public Future { private: friend class Transaction; - ValueFuture(FDBFuture* f) : Future(f) {} + explicit ValueFuture(FDBFuture* f) : Future(f) {} }; class StringArrayFuture : public Future { @@ -133,7 +133,7 @@ class StringArrayFuture : public Future { private: friend class Transaction; - StringArrayFuture(FDBFuture* f) : Future(f) {} + explicit StringArrayFuture(FDBFuture* f) : Future(f) {} }; class KeyValueArrayFuture : public Future { @@ -145,7 +145,7 @@ class KeyValueArrayFuture : public Future { private: friend class Transaction; - KeyValueArrayFuture(FDBFuture* f) : Future(f) {} + explicit KeyValueArrayFuture(FDBFuture* f) : Future(f) {} }; class MappedKeyValueArrayFuture : public Future { @@ -157,7 +157,7 @@ class MappedKeyValueArrayFuture : public Future { private: friend class Transaction; - MappedKeyValueArrayFuture(FDBFuture* f) : Future(f) {} + explicit MappedKeyValueArrayFuture(FDBFuture* f) : Future(f) {} }; class KeyRangeArrayFuture : public Future { @@ -169,14 +169,14 @@ class KeyRangeArrayFuture : public Future { private: friend class Transaction; - KeyRangeArrayFuture(FDBFuture* f) : Future(f) {} + explicit KeyRangeArrayFuture(FDBFuture* f) : Future(f) {} }; class EmptyFuture : public Future { private: friend class Transaction; friend class Database; - EmptyFuture(FDBFuture* f) : Future(f) {} + explicit EmptyFuture(FDBFuture* f) : Future(f) {} }; class Result { @@ -184,7 +184,7 @@ class Result { virtual ~Result() = 0; protected: - Result(FDBResult* r) : result_(r) {} + explicit Result(FDBResult* r) : result_(r) {} FDBResult* result_; }; @@ -197,7 +197,7 @@ class KeyValueArrayResult : public Result { private: friend class Transaction; - KeyValueArrayResult(FDBResult* r) : Result(r) {} + explicit KeyValueArrayResult(FDBResult* r) : Result(r) {} }; // Wrapper around FDBDatabase, providing database-level API @@ -222,7 +222,7 @@ class Database final { class Transaction final { public: // Given an FDBDatabase, initializes a new transaction. - Transaction(FDBDatabase* db); + explicit Transaction(FDBDatabase* db); ~Transaction(); // Wrapper around fdb_transaction_reset. diff --git a/bindings/c/test/unit/unit_tests.cpp b/bindings/c/test/unit/unit_tests.cpp index 31e2f25c746..ac584d120d1 100644 --- a/bindings/c/test/unit/unit_tests.cpp +++ b/bindings/c/test/unit/unit_tests.cpp @@ -1972,6 +1972,36 @@ TEST_CASE("fdb_database_get_server_protocol") { fdb_future_destroy(protocolFuture); } +TEST_CASE("CDC C binding rejects invalid registration ranges") { + const uint8_t name[] = "invalid-cdc-stream"; + const uint8_t begin[] = "a"; + const uint8_t end[] = "b"; + FDBKeyRange range{ begin, 1, end, 1 }; + auto checkInvalid = [&](uint8_t const* nameInput, int nameLength, FDBKeyRange const* ranges, int rangeCount) { + FDBFuture* future = fdb_database_register_cdc_stream(db, nameInput, nameLength, ranges, rangeCount); + REQUIRE(future != nullptr); + fdb_check(fdb_future_block_until_ready(future)); + CHECK(fdb_future_get_error(future) == 2000); // client_invalid_operation + fdb_future_destroy(future); + }; + + checkInvalid(name, sizeof(name) - 1, &range, -1); + checkInvalid(name, sizeof(name) - 1, &range, 0); + checkInvalid(name, sizeof(name) - 1, &range, 1025); + checkInvalid(name, sizeof(name) - 1, nullptr, 1); + checkInvalid(nullptr, 1, &range, 1); + checkInvalid(name, -1, &range, 1); + checkInvalid(name, 0, &range, 1); + for (FDBKeyRange invalid : { FDBKeyRange{ begin, -1, end, 1 }, + FDBKeyRange{ begin, 1, end, -1 }, + FDBKeyRange{ nullptr, 1, end, 1 }, + FDBKeyRange{ begin, 1, nullptr, 1 }, + FDBKeyRange{ begin, 1, begin, 1 }, + FDBKeyRange{ end, 1, begin, 1 } }) { + checkInvalid(name, sizeof(name) - 1, &invalid, 1); + } +} + TEST_CASE("CDC C binding end-to-end") { using FuturePtr = std::unique_ptr; using ConsumerPtr = std::unique_ptr; @@ -2073,10 +2103,13 @@ TEST_CASE("CDC C binding end-to-end") { }; const std::string streamName = key("cdc-stream"); - const std::string rangeBegin = key("cdc-data/"); + const std::string rangeBegin = key("cdc-data/a/"); const std::string rangeEnd = strinc_str(rangeBegin); + const std::string secondRangeBegin = key("cdc-data/c/"); + const std::string secondRangeEnd = strinc_str(secondRangeBegin); const std::string firstKey = rangeBegin + "first"; - const std::string secondKey = rangeBegin + "second"; + const std::string secondKey = secondRangeBegin + "second"; + const std::string gapKey = key("cdc-data/b/gap"); const std::string outsideKey = key("outside-cdc-range"); const std::string firstValue = "first-value"; const std::string secondValue = "second-value"; @@ -2088,18 +2121,30 @@ TEST_CASE("CDC C binding end-to-end") { std::string nameInput = streamName; std::string beginInput = rangeBegin; std::string endInput = rangeEnd; - auto registerFuture = - ownFuture(fdb_database_register_cdc_stream(db, - reinterpret_cast(nameInput.data()), - nameInput.size(), - reinterpret_cast(beginInput.data()), - beginInput.size(), - reinterpret_cast(endInput.data()), - endInput.size())); + std::string secondBeginInput = secondRangeBegin; + std::string secondEndInput = secondRangeEnd; + std::vector rangesInput{ { reinterpret_cast(secondBeginInput.data()), + static_cast(secondBeginInput.size()), + reinterpret_cast(secondEndInput.data()), + static_cast(secondEndInput.size()) }, + { reinterpret_cast(beginInput.data()), + static_cast(beginInput.size()), + reinterpret_cast(endInput.data()), + static_cast(endInput.size()) } }; + // Repeated intervals must not duplicate delivered mutations. + rangesInput.push_back(rangesInput.back()); + auto registerFuture = ownFuture(fdb_database_register_cdc_stream(db, + reinterpret_cast(nameInput.data()), + nameInput.size(), + rangesInput.data(), + rangesInput.size())); REQUIRE(registerFuture != nullptr); std::fill(nameInput.begin(), nameInput.end(), 'x'); std::fill(beginInput.begin(), beginInput.end(), 'x'); std::fill(endInput.begin(), endInput.end(), 'x'); + std::fill(secondBeginInput.begin(), secondBeginInput.end(), 'x'); + std::fill(secondEndInput.begin(), secondEndInput.end(), 'x'); + std::fill(rangesInput.begin(), rangesInput.end(), FDBKeyRange{ nullptr, -1, nullptr, -1 }); waitForSuccess(registerFuture.get()); uint64_t streamId = 0; @@ -2119,10 +2164,16 @@ TEST_CASE("CDC C binding end-to-end") { } foundStream = true; CHECK(streams[i].stream_id == streamId); - CHECK(std::string(reinterpret_cast(streams[i].key_range.begin_key), - streams[i].key_range.begin_key_length) == rangeBegin); - CHECK(std::string(reinterpret_cast(streams[i].key_range.end_key), - streams[i].key_range.end_key_length) == rangeEnd); + REQUIRE(streams[i].range_count == 2); + REQUIRE(streams[i].ranges != nullptr); + CHECK(std::string(reinterpret_cast(streams[i].ranges[0].begin_key), + streams[i].ranges[0].begin_key_length) == rangeBegin); + CHECK(std::string(reinterpret_cast(streams[i].ranges[0].end_key), + streams[i].ranges[0].end_key_length) == rangeEnd); + CHECK(std::string(reinterpret_cast(streams[i].ranges[1].begin_key), + streams[i].ranges[1].begin_key_length) == secondRangeBegin); + CHECK(std::string(reinterpret_cast(streams[i].ranges[1].end_key), + streams[i].ranges[1].end_key_length) == secondRangeEnd); CHECK(streams[i].min_version >= 0); } REQUIRE(foundStream); @@ -2145,8 +2196,10 @@ TEST_CASE("CDC C binding end-to-end") { CHECK(positionStreamId == streamId); CHECK(positionVersion == -1); - const int64_t setVersion = - commitSetValues({ { firstKey, firstValue }, { secondKey, secondValue }, { outsideKey, "outside-value" } }); + const int64_t setVersion = commitSetValues({ { firstKey, firstValue }, + { secondKey, secondValue }, + { gapKey, "gap-value" }, + { outsideKey, "outside-value" } }); auto setReply = consumeThroughVersion(consumer.get(), setVersion); CHECK(setReply.version == setVersion); CHECK(setReply.lastConsumedVersion >= setVersion); @@ -2181,14 +2234,20 @@ TEST_CASE("CDC C binding end-to-end") { CHECK(positionStreamId == streamId); CHECK(positionVersion == setReply.lastConsumedVersion); - const std::string clearEnd = strinc_str(firstKey); + const std::string clearEnd = strinc_str(secondKey); const int64_t clearVersion = commitClearRange(firstKey, clearEnd); auto clearReply = consumeThroughVersion(resumedConsumer.get(), clearVersion); CHECK(clearReply.version == clearVersion); - REQUIRE(clearReply.mutations.size() == 1); - CHECK(clearReply.mutations[0].type == FDB_CDC_MUTATION_TYPE_CLEAR_RANGE); - CHECK(clearReply.mutations[0].param1 == firstKey); - CHECK(clearReply.mutations[0].param2 == clearEnd); + REQUIRE(clearReply.mutations.size() == 2); + std::map expectedClears{ { firstKey, rangeEnd }, { secondRangeBegin, clearEnd } }; + for (auto const& mutation : clearReply.mutations) { + CHECK(mutation.type == FDB_CDC_MUTATION_TYPE_CLEAR_RANGE); + auto expected = expectedClears.find(mutation.param1); + REQUIRE(expected != expectedClears.end()); + CHECK(mutation.param2 == expected->second); + expectedClears.erase(expected); + } + CHECK(expectedClears.empty()); auto resumedAcknowledgeFuture = ownFuture(fdb_cdc_consumer_acknowledge(resumedConsumer.get())); REQUIRE(resumedAcknowledgeFuture != nullptr); diff --git a/bindings/java/JavaWorkload.cpp b/bindings/java/JavaWorkload.cpp index 332bd6d55cc..27b17377135 100644 --- a/bindings/java/JavaWorkload.cpp +++ b/bindings/java/JavaWorkload.cpp @@ -27,6 +27,7 @@ #include "com_apple_foundationdb_testing_WorkloadContext.h" #include +#include #include #include #include @@ -80,7 +81,6 @@ void printTrace(JNIEnv* env, jclass, jlong logger, jint severity, jstring messag } else if (severity < 40) { sev = FDBSeverity::WarnAlways; } else { - assert(false); std::abort(); } log->trace(sev, msg, detailsMap); diff --git a/bindings/python/CMakeLists.txt b/bindings/python/CMakeLists.txt index ab1708f6291..083bde73a62 100644 --- a/bindings/python/CMakeLists.txt +++ b/bindings/python/CMakeLists.txt @@ -126,6 +126,20 @@ if(NOT OPEN_FOR_IDE AND NOT USE_SANITIZER) DISABLE_LOG_DUMP ) + add_fdbclient_test( + NAME python_native_cdc_tests + COMMAND python ${CMAKE_CURRENT_SOURCE_DIR}/tests/native_cdc_tests.py + --cluster-file @CLUSTER_FILE@ --verbose + ) + set_property(TEST python_native_cdc_tests APPEND PROPERTY ENVIRONMENT + "FDB_KNOB_enable_native_cdc=true") + + add_fdbclient_test( + NAME python_native_cdc_legacy_api_tests + COMMAND python ${CMAKE_CURRENT_SOURCE_DIR}/tests/native_cdc_tests.py + --cluster-file @CLUSTER_FILE@ --api-version 740 --verbose + ) + # FIXME Windows support set(PYPKG_TEST_DIR "${CMAKE_BINARY_DIR}/pypkg-test-venv") set(PYPKG_TEST_PY3 "${PYPKG_TEST_DIR}/bin/python3") diff --git a/bindings/python/fdb/__init__.py b/bindings/python/fdb/__init__.py index b4dea358a43..dd790895aea 100644 --- a/bindings/python/fdb/__init__.py +++ b/bindings/python/fdb/__init__.py @@ -109,6 +109,14 @@ def api_version(ver): "transactional", "options", "StreamingMode", + "CdcMutationType", + "CdcCursor", + "CdcKeyRange", + "CdcStreamInfo", + "CdcMutation", + "CdcVersionedMutations", + "CdcConsumeResult", + "CdcConsumer", ) _add_symbols(fdb.impl, list) diff --git a/bindings/python/fdb/impl.py b/bindings/python/fdb/impl.py index 6b7a71f4c1a..856ae2dd874 100644 --- a/bindings/python/fdb/impl.py +++ b/bindings/python/fdb/impl.py @@ -23,9 +23,11 @@ import atexit import ctypes import ctypes.util +import enum import functools import inspect import multiprocessing +import operator import os import platform import struct @@ -33,6 +35,7 @@ import threading import traceback import weakref +from typing import NamedTuple, Tuple import fdb from fdb.tuple import int2byte @@ -42,6 +45,7 @@ _network_thread = None _network_thread_reentrant_lock = threading.RLock() +_cdc_c_api_initialized = False _open_file = open @@ -891,6 +895,82 @@ def wait(self): return list(strings[0 : count.value]) +class FutureCdcStreamInfoArray(Future): + def wait(self): + self.block_until_ready() + streams = ctypes.POINTER(CdcStreamInfoStruct)() + count = ctypes.c_int() + self.capi.fdb_future_get_cdc_stream_info_array( + self.fpointer, ctypes.byref(streams), ctypes.byref(count) + ) + return [ + CdcStreamInfo( + ctypes.string_at(stream.name.key, stream.name.key_length), + stream.stream_id, + tuple( + CdcKeyRange( + ctypes.string_at(r.begin_key, r.begin_key_length), + ctypes.string_at(r.end_key, r.end_key_length), + ) + for r in stream.ranges[: stream.range_count] + ), + stream.min_version, + ) + for stream in streams[: count.value] + ] + + +class FutureCdcConsumer(Future): + def __init__(self, fpointer): + super().__init__(fpointer) + self._consumer = None + self._lock = threading.Lock() + + def wait(self): + self.block_until_ready() + with self._lock: + if self._consumer is None: + consumer = CdcConsumer() + self.capi.fdb_future_get_cdc_consumer( + self.fpointer, ctypes.byref(consumer._pointer) + ) + # Each C getter call transfers a reference. Keep one Python + # owner even when callbacks or callers retrieve the result again. + self._consumer = consumer + return self._consumer + + +class FutureCdcConsumeResult(Future): + def wait(self): + self.block_until_ready() + groups = ctypes.POINTER(CdcVersionedMutationsStruct)() + count = ctypes.c_int() + last_consumed_version = ctypes.c_int64() + self.capi.fdb_future_get_cdc_versioned_mutations( + self.fpointer, + ctypes.byref(groups), + ctypes.byref(count), + ctypes.byref(last_consumed_version), + ) + return CdcConsumeResult( + tuple( + CdcVersionedMutations( + group.version, + tuple( + CdcMutation( + mutation.type, + ctypes.string_at(mutation.param1, mutation.param1_length), + ctypes.string_at(mutation.param2, mutation.param2_length), + ) + for mutation in group.mutations[: group.mutation_count] + ), + ) + for group in groups[: count.value] + ), + last_consumed_version.value, + ) + + class replaceable_property(object): def __get__(self, obj, cls=None): return self.method(obj) @@ -1342,10 +1422,240 @@ def create_transaction(self): def get_client_status(self): return Key(self.capi.fdb_database_get_client_status(self.dpointer)) + def register_cdc_stream(self, name, begin_key=None, end_key=None, *, ranges=None): + """Register a named CDC range union and return its stream ID future. + + Supply either begin_key and end_key, or ranges as an iterable of + (begin_key, end_key) pairs. The native client canonicalizes the union. + """ + _require_cdc_api_version() + name = keyToBytes(name) + if ranges is None: + if begin_key is None or end_key is None: + raise TypeError("Supply both begin_key and end_key, or ranges") + ranges = ((begin_key, end_key),) + elif begin_key is not None or end_key is not None: + raise TypeError("ranges cannot be combined with begin_key or end_key") + # Keep converted key bytes alive until the C API has copied every range. + ranges = [(keyToBytes(begin), keyToBytes(end)) for begin, end in ranges] + native_ranges = (KeyRangeStruct * len(ranges))( + *( + KeyRangeStruct( + ctypes.cast(begin, ctypes.POINTER(ctypes.c_byte)), + len(begin), + ctypes.cast(end, ctypes.POINTER(ctypes.c_byte)), + len(end), + ) + for begin, end in ranges + ) + ) + return FutureUInt64( + self.capi.fdb_database_register_cdc_stream( + self.dpointer, + name, + len(name), + native_ranges, + len(ranges), + ) + ) + + def remove_cdc_stream(self, name): + """Remove a stream and relinquish its unread history; return a void future.""" + _require_cdc_api_version() + name = keyToBytes(name) + return FutureVoid( + self.capi.fdb_database_remove_cdc_stream(self.dpointer, name, len(name)) + ) + + def list_cdc_streams(self): + """Return a future containing a list of CdcStreamInfo records.""" + _require_cdc_api_version() + return FutureCdcStreamInfoArray( + self.capi.fdb_database_list_cdc_streams(self.dpointer) + ) + + def create_cdc_consumer(self, name): + """Return a future containing a consumer for an existing stream name.""" + _require_cdc_api_version() + name = keyToBytes(name) + return FutureCdcConsumer( + self.capi.fdb_database_create_cdc_consumer(self.dpointer, name, len(name)) + ) + + def resume_cdc_consumer(self, cursor): + """Return a consumer future from a durably checkpointed CdcCursor. + + The native client validates the stream and position on consume or + acknowledge, not when constructing the handle. + """ + _require_cdc_api_version() + stream_id = operator.index(cursor.stream_id) + version = operator.index(cursor.last_consumed_version) + if not 0 <= stream_id < 2**64: + raise ValueError("stream_id must fit in an unsigned 64-bit integer") + if not -(2**63) <= version < 2**63: + raise ValueError( + "last_consumed_version must fit in a signed 64-bit integer" + ) + return FutureCdcConsumer( + self.capi.fdb_database_resume_cdc_consumer( + self.dpointer, stream_id, version + ) + ) + fill_operations() +class CdcMutationType(enum.IntEnum): + """Known raw CDC mutation codes; a reply may also contain unknown codes.""" + + SET_VALUE = 0 + CLEAR_RANGE = 1 + ADD = 2 + AND = 6 + OR = 7 + XOR = 8 + APPEND_IF_FITS = 9 + MAX = 12 + MIN = 13 + SET_VERSIONSTAMPED_KEY = 14 + SET_VERSIONSTAMPED_VALUE = 15 + BYTE_MIN = 16 + BYTE_MAX = 17 + MIN_V2 = 18 + AND_V2 = 19 + COMPARE_AND_CLEAR = 20 + + +class CdcCursor(NamedTuple): + """A stream's stable ID and delivered position, suitable for checkpointing.""" + + stream_id: int + last_consumed_version: int + + +class CdcKeyRange(NamedTuple): + """A half-open CDC key range with Python-owned endpoint bytes.""" + + begin_key: bytes + end_key: bytes + + +class CdcStreamInfo(NamedTuple): + """A registered stream and its durable minimum required version.""" + + name: bytes + stream_id: int + ranges: Tuple[CdcKeyRange, ...] + min_version: int + + @property + def begin_key(self): + """Return the begin key of a single-range stream.""" + if len(self.ranges) != 1: + raise ValueError("Use ranges for a multi-range CDC stream") + return self.ranges[0].begin_key + + @property + def end_key(self): + """Return the end key of a single-range stream.""" + if len(self.ranges) != 1: + raise ValueError("Use ranges for a multi-range CDC stream") + return self.ranges[0].end_key + + +class CdcMutation(NamedTuple): + """One raw mutation with Python-owned parameter bytes.""" + + type: int + param1: bytes + param2: bytes + + +class CdcVersionedMutations(NamedTuple): + """A complete group of mutations sharing one commit version.""" + + version: int + mutations: Tuple[CdcMutation, ...] + + +class CdcConsumeResult(NamedTuple): + """Copied mutation groups and the delivered watermark, even for empty replies.""" + + mutations: Tuple[CdcVersionedMutations, ...] + last_consumed_version: int + + +class CdcConsumer(_FDBBase): + """An owned native CDC consumer handle. + + Only one consume or acknowledge operation may be outstanding per handle. + Acknowledgements affect the whole stream, which must have only one active + logical consumer. Closing a handle does not acknowledge or remove its stream. + """ + + def __init__(self): + self._lock = threading.Lock() + self._pointer = ctypes.c_void_p() + + def __del__(self): + if getattr(self, "_pointer", None): + self.close() + + def _check_open(self): + if not self._pointer: + raise ValueError("CDC consumer is closed") + + def close(self): + """Release this handle without acknowledging or removing the stream.""" + with self._lock: + if self._pointer: + pointer = self._pointer + self._pointer = None + self.capi.fdb_cdc_consumer_destroy(pointer) + + def __enter__(self): + with self._lock: + self._check_open() + return self + + def __exit__(self, exc_type, exc_value, traceback): + self.close() + + def consume(self): + """Long-poll for complete commit groups; return a CdcConsumeResult future. + + Consumption advances the delivered cursor, not durable retention. Do not + consume again until the previous reply has been durably processed. + """ + with self._lock: + self._check_open() + return FutureCdcConsumeResult( + self.capi.fdb_cdc_consumer_consume(self._pointer) + ) + + def acknowledge(self): + """Durably acknowledge the current delivered position; return a void future. + + Call only after durably processing every mutation through that position. + """ + with self._lock: + self._check_open() + return FutureVoid(self.capi.fdb_cdc_consumer_acknowledge(self._pointer)) + + def get_position(self): + """Return the current CdcCursor; this does not acknowledge the position.""" + with self._lock: + self._check_open() + stream_id = ctypes.c_uint64() + version = ctypes.c_int64() + self.capi.fdb_cdc_consumer_get_position( + self._pointer, ctypes.byref(stream_id), ctypes.byref(version) + ) + return CdcCursor(stream_id.value, version.value) + + class Cluster(_FDBBase): def __init__(self, cluster_file): self.cluster_file = cluster_file @@ -1438,6 +1748,47 @@ class KeyStruct(ctypes.Structure): _pack_ = 4 +class KeyRangeStruct(ctypes.Structure): + _pack_ = 4 + _fields_ = [ + ("begin_key", ctypes.POINTER(ctypes.c_byte)), + ("begin_key_length", ctypes.c_int), + ("end_key", ctypes.POINTER(ctypes.c_byte)), + ("end_key_length", ctypes.c_int), + ] + + +class CdcStreamInfoStruct(ctypes.Structure): + _pack_ = 4 + _fields_ = [ + ("name", KeyStruct), + ("stream_id", ctypes.c_uint64), + ("ranges", ctypes.POINTER(KeyRangeStruct)), + ("range_count", ctypes.c_int), + ("min_version", ctypes.c_int64), + ] + + +class CdcMutationStruct(ctypes.Structure): + _pack_ = 4 + _fields_ = [ + ("type", ctypes.c_uint8), + ("param1", ctypes.POINTER(ctypes.c_byte)), + ("param1_length", ctypes.c_int), + ("param2", ctypes.POINTER(ctypes.c_byte)), + ("param2_length", ctypes.c_int), + ] + + +class CdcVersionedMutationsStruct(ctypes.Structure): + _pack_ = 4 + _fields_ = [ + ("version", ctypes.c_int64), + ("mutations", ctypes.POINTER(CdcMutationStruct)), + ("mutation_count", ctypes.c_int), + ] + + class KeyValue(object): def __init__(self, key, value): self.key = key @@ -1558,6 +1909,16 @@ def optionalParamToBytes(v): return (v, len(v)) +def _require_cdc_api_version(): + global _cdc_c_api_initialized + if fdb.get_api_version() < 800: + raise RuntimeError("Native CDC requires API version 800 or later") + with _network_thread_reentrant_lock: + if not _cdc_c_api_initialized: + _init_cdc_c_api() + _cdc_c_api_initialized = True + + _FDBBase.capi = _capi _CBFUNC = ctypes.CFUNCTYPE(None, ctypes.c_void_p) @@ -1910,6 +2271,87 @@ def init_c_api(): _capi.fdb_transaction_reset.restype = None +def _init_cdc_c_api(): + # Older libraries may support API 800 without these experimental symbols. + # Resolve the whole surface before using it, and leave non-CDC users alone. + signatures = ( + ( + "fdb_future_get_cdc_stream_info_array", + [ + ctypes.c_void_p, + ctypes.POINTER(ctypes.POINTER(CdcStreamInfoStruct)), + ctypes.POINTER(ctypes.c_int), + ], + ctypes.c_int, + ), + ( + "fdb_future_get_cdc_consumer", + [ctypes.c_void_p, ctypes.POINTER(ctypes.c_void_p)], + ctypes.c_int, + ), + ( + "fdb_future_get_cdc_versioned_mutations", + [ + ctypes.c_void_p, + ctypes.POINTER(ctypes.POINTER(CdcVersionedMutationsStruct)), + ctypes.POINTER(ctypes.c_int), + ctypes.POINTER(ctypes.c_int64), + ], + ctypes.c_int, + ), + ( + "fdb_database_register_cdc_stream", + [ + ctypes.c_void_p, + ctypes.c_void_p, + ctypes.c_int, + ctypes.POINTER(KeyRangeStruct), + ctypes.c_int, + ], + ctypes.c_void_p, + ), + ( + "fdb_database_remove_cdc_stream", + [ctypes.c_void_p, ctypes.c_void_p, ctypes.c_int], + ctypes.c_void_p, + ), + ("fdb_database_list_cdc_streams", [ctypes.c_void_p], ctypes.c_void_p), + ( + "fdb_database_create_cdc_consumer", + [ctypes.c_void_p, ctypes.c_void_p, ctypes.c_int], + ctypes.c_void_p, + ), + ( + "fdb_database_resume_cdc_consumer", + [ctypes.c_void_p, ctypes.c_uint64, ctypes.c_int64], + ctypes.c_void_p, + ), + ("fdb_cdc_consumer_destroy", [ctypes.c_void_p], None), + ("fdb_cdc_consumer_consume", [ctypes.c_void_p], ctypes.c_void_p), + ("fdb_cdc_consumer_acknowledge", [ctypes.c_void_p], ctypes.c_void_p), + ( + "fdb_cdc_consumer_get_position", + [ + ctypes.c_void_p, + ctypes.POINTER(ctypes.c_uint64), + ctypes.POINTER(ctypes.c_int64), + ], + ctypes.c_int, + ), + ) + try: + functions = [getattr(_capi, name) for name, _, _ in signatures] + except AttributeError as error: + raise RuntimeError( + "The loaded FoundationDB C library does not support native CDC" + ) from error + for function, (_, argtypes, restype) in zip(functions, signatures): + function.argtypes = argtypes + function.restype = restype + if restype is ctypes.c_int: + function.errcheck = check_error_code + + if hasattr(ctypes.pythonapi, "Py_IncRef"): def _pin_callback(cb): diff --git a/bindings/python/tests/native_cdc_tests.py b/bindings/python/tests/native_cdc_tests.py new file mode 100644 index 00000000000..5614bbcb0f0 --- /dev/null +++ b/bindings/python/tests/native_cdc_tests.py @@ -0,0 +1,496 @@ +#!/usr/bin/env python3 +# +# native_cdc_tests.py +# +# This source file is part of the FoundationDB open source project +# +# Copyright 2026 Apple Inc. and the FoundationDB project authors +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +import argparse +import ctypes +import gc +import threading +import time +import unittest +import uuid +from unittest import mock + +import fdb + + +def wait(future, timeout=30): + ready = threading.Event() + future.on_ready(lambda _: ready.set()) + if not ready.wait(timeout): + future.cancel() + raise AssertionError( + "CDC operation did not complete within {} seconds".format(timeout) + ) + return future.wait() + + +class CdcDecodingTests(unittest.TestCase): + def test_packed_c_layout(self): + impl = fdb.impl + pointer_size = ctypes.sizeof(ctypes.c_void_p) + layouts = ( + (impl.KeyRangeStruct, 2 * pointer_size + 8), + (impl.CdcStreamInfoStruct, 2 * pointer_size + 24), + (impl.CdcMutationStruct, 2 * pointer_size + 12), + (impl.CdcVersionedMutationsStruct, pointer_size + 12), + ) + for structure, size in layouts: + with self.subTest(structure=structure.__name__): + self.assertEqual(structure._pack_, 4) + self.assertEqual(ctypes.sizeof(structure), size) + self.assertEqual(impl.CdcMutationStruct.param1.offset, 4) + self.assertEqual(impl.CdcStreamInfoStruct.stream_id.offset, pointer_size + 4) + + def test_unknown_mutation_type_and_copied_bytes(self): + impl = fdb.impl + key = ctypes.create_string_buffer(b"key\x00\xff") + value = ctypes.create_string_buffer(b"\x00value\xff") + mutations = (impl.CdcMutationStruct * 2)( + impl.CdcMutationStruct( + 255, + ctypes.cast(key, ctypes.POINTER(ctypes.c_byte)), + len(key) - 1, + ctypes.cast(value, ctypes.POINTER(ctypes.c_byte)), + len(value) - 1, + ), + impl.CdcMutationStruct(fdb.CdcMutationType.SET_VALUE, None, 0, None, 0), + ) + groups = (impl.CdcVersionedMutationsStruct * 2)( + impl.CdcVersionedMutationsStruct(100, mutations, 2), + impl.CdcVersionedMutationsStruct(101, None, 0), + ) + + def get_result(pointer, out_groups, out_count, out_version): + ctypes.cast( + out_groups, + ctypes.POINTER(ctypes.POINTER(impl.CdcVersionedMutationsStruct)), + )[0] = groups + ctypes.cast(out_count, ctypes.POINTER(ctypes.c_int))[0] = len(groups) + ctypes.cast(out_version, ctypes.POINTER(ctypes.c_int64))[0] = 150 + + future = impl.FutureCdcConsumeResult(1) + future.capi = mock.Mock() + future.capi.fdb_future_is_ready.return_value = 1 + future.capi.fdb_future_get_cdc_versioned_mutations.side_effect = get_result + result = future.wait() + ctypes.memset(key, 0, len(key)) + ctypes.memset(value, 0, len(value)) + mutations[0].type = 0 + groups[0].version = 0 + del future + gc.collect() + + self.assertEqual( + result, + fdb.CdcConsumeResult( + ( + fdb.CdcVersionedMutations( + 100, + ( + fdb.CdcMutation(255, b"key\x00\xff", b"\x00value\xff"), + fdb.CdcMutation(fdb.CdcMutationType.SET_VALUE, b"", b""), + ), + ), + fdb.CdcVersionedMutations(101, ()), + ), + 150, + ), + ) + self.assertIs(type(result.mutations[0].mutations[0].type), int) + with self.assertRaises(AttributeError): + result.last_consumed_version = 0 + + def test_empty_reply_preserves_progress(self): + def get_result(pointer, out_groups, out_count, out_version): + ctypes.cast(out_count, ctypes.POINTER(ctypes.c_int))[0] = 0 + ctypes.cast(out_version, ctypes.POINTER(ctypes.c_int64))[0] = 2**63 - 1 + + future = fdb.impl.FutureCdcConsumeResult(1) + future.capi = mock.Mock() + future.capi.fdb_future_is_ready.return_value = 1 + future.capi.fdb_future_get_cdc_versioned_mutations.side_effect = get_result + self.assertEqual(future.wait(), fdb.CdcConsumeResult((), 2**63 - 1)) + + +class NativeCdcTests(unittest.TestCase): + def setUp(self): + self.prefix = b"python-cdc/" + uuid.uuid4().bytes + b"\x00/" + self.name = self.prefix + b"stream\x00\xff" + self.begin = self.prefix + b"range/" + self.end = self.prefix + b"range0" + self.addCleanup(self.db.clear_range, self.prefix, self.prefix + b"\xff") + + def register(self): + stream_id = wait(self.db.register_cdc_stream(self.name, self.begin, self.end)) + self.addCleanup(lambda: wait(self.db.remove_cdc_stream(self.name))) + return stream_id + + def stream_info(self): + future = self.db.list_cdc_streams() + streams = wait(future) + self.assertEqual(future.wait(), streams) + future._release_memory() + del future + gc.collect() + return next(stream for stream in streams if stream.name == self.name) + + def commit(self, write): + tr = self.db.create_transaction() + tr.options.set_timeout(30000) + while True: + try: + write(tr) + wait(tr.commit()) + return tr.get_committed_version() + except fdb.FDBError as error: + wait(tr.on_error(error)) + + def consume_through(self, consumer, version): + groups = {} + deadline = time.monotonic() + 60 + while time.monotonic() < deadline: + future = consumer.consume() + result = wait(future) + self.assertEqual(future.wait(), result) + future._release_memory() + del future + gc.collect() + self.assertEqual( + consumer.get_position().last_consumed_version, + result.last_consumed_version, + ) + self.assertIsInstance(result.mutations, tuple) + for group in result.mutations: + self.assertIsInstance(group.mutations, tuple) + if group.version in groups: + self.assertEqual(groups[group.version], group.mutations) + groups[group.version] = group.mutations + if result.last_consumed_version >= version: + return groups + self.fail("CDC did not deliver the committed version within 60 seconds") + + def assert_closed(self, consumer): + consumer.close() + consumer.close() + for operation in ( + consumer.consume, + consumer.acknowledge, + consumer.get_position, + ): + with self.subTest(operation=operation.__name__): + with self.assertRaises(ValueError): + operation() + + def test_stream_lifecycle_and_result_ownership(self): + stream_id = self.register() + self.assertGreater(stream_id, 0) + self.assertEqual( + wait(self.db.register_cdc_stream(self.name, self.begin, self.end)), + stream_id, + ) + info = self.stream_info() + self.assertEqual( + (info.name, info.stream_id, info.begin_key, info.end_key), + (self.name, stream_id, self.begin, self.end), + ) + self.assertEqual(info.ranges, (fdb.CdcKeyRange(self.begin, self.end),)) + self.assertGreaterEqual(info.min_version, 0) + with self.assertRaises(AttributeError): + info.name = b"different" + + create_future = self.db.create_cdc_consumer(self.name) + consumer = wait(create_future) + self.addCleanup(consumer.close) + self.assertIs(create_future.wait(), consumer) + self.assertIs(create_future.result(), consumer) + self.assertIsNone(create_future.exception()) + create_future._release_memory() + del create_future + gc.collect() + self.assertEqual(consumer.get_position(), fdb.CdcCursor(stream_id, -1)) + + first_key = self.begin + b"first\x00\xff" + second_key = self.begin + b"second" + empty_key = self.begin + b"empty" + first_value = b"\x00first\xffvalue\x00" + second_value = b"second-value" + + def write_sets(tr): + tr[first_key] = first_value + tr[second_key] = second_value + tr[self.prefix + b"outside"] = b"not in the stream" + + set_version = self.commit(write_sets) + empty_version = self.commit(lambda tr: tr.set(empty_key, b"")) + groups = self.consume_through(consumer, empty_version) + self.assertCountEqual( + groups[set_version], + ( + fdb.CdcMutation(fdb.CdcMutationType.SET_VALUE, first_key, first_value), + fdb.CdcMutation( + fdb.CdcMutationType.SET_VALUE, second_key, second_value + ), + ), + ) + self.assertEqual( + groups[empty_version], + (fdb.CdcMutation(fdb.CdcMutationType.SET_VALUE, empty_key, b""),), + ) + cursor = consumer.get_position() + self.assert_closed(consumer) + self.assertEqual(self.stream_info().min_version, info.min_version) + + resume_future = self.db.resume_cdc_consumer(cursor) + resumed = wait(resume_future) + self.addCleanup(resumed.close) + self.assertIs(resume_future.wait(), resumed) + del resume_future + gc.collect() + with resumed: + self.assertEqual(resumed.get_position(), cursor) + # A resumed handle has no local delivery proof. Its acknowledgement + # must be at or behind a fresh database read version. + deadline = time.monotonic() + 30 + while True: + remaining = deadline - time.monotonic() + self.assertGreater(remaining, 0, "Read version did not reach cursor") + tr = self.db.create_transaction() + read_version = wait( + tr.get_read_version(), + timeout=max(0, deadline - time.monotonic()), + ) + if read_version >= cursor.last_consumed_version: + break + time.sleep(min(0.01, max(0, deadline - time.monotonic()))) + # Reconcile the durable checkpoint, then reissue the acknowledgement. + for _ in range(2): + self.assertIsNone(wait(resumed.acknowledge())) + self.assertEqual( + self.stream_info().min_version, + cursor.last_consumed_version + 1, + ) + clear_end = first_key + b"\x00" + counter_key = self.begin + b"counter" + operand = b"\x01\x00\x00\x00" + + def write_raw_mutations(tr): + tr.clear_range(first_key, clear_end) + tr.add(counter_key, operand) + + raw_version = self.commit(write_raw_mutations) + groups = self.consume_through(resumed, raw_version) + self.assertCountEqual( + groups[raw_version], + ( + fdb.CdcMutation( + fdb.CdcMutationType.CLEAR_RANGE, first_key, clear_end + ), + fdb.CdcMutation(fdb.CdcMutationType.ADD, counter_key, operand), + ), + ) + self.assertIsNone(wait(resumed.acknowledge())) + self.assert_closed(resumed) + + self.assertIsNone(wait(self.db.remove_cdc_stream(self.name))) + self.assertIsNone(wait(self.db.remove_cdc_stream(self.name))) + self.assertNotIn( + self.name, [stream.name for stream in wait(self.db.list_cdc_streams())] + ) + + def test_multi_range_union_filtering_and_resume(self): + first = fdb.CdcKeyRange(self.prefix + b"a\x00", self.prefix + b"c\xff") + second = fdb.CdcKeyRange(self.prefix + b"x\x00", self.prefix + b"z\xff") + split = self.prefix + b"b" + ranges = ( + second, + (split, first.end_key), + (first.begin_key, split), + first, + ) + stream_id = wait( + self.db.register_cdc_stream(self.name, ranges=(r for r in ranges)) + ) + self.addCleanup(lambda: wait(self.db.remove_cdc_stream(self.name))) + self.assertEqual( + wait(self.db.register_cdc_stream(self.name, ranges=(first, second))), + stream_id, + ) + info = self.stream_info() + self.assertEqual(info.ranges, (first, second)) + for field in ("begin_key", "end_key"): + with self.assertRaisesRegex(ValueError, "Use ranges"): + getattr(info, field) + with self.assertRaises(AttributeError): + info.ranges[0].begin_key = b"different" + + with wait(self.db.create_cdc_consumer(self.name)) as consumer: + + def write_sets(tr): + tr[first.begin_key] = b"first\x00\xff" + tr[second.begin_key] = b"second\x00\xff" + tr[first.end_key] = b"excluded end" + tr[self.prefix + b"gap"] = b"excluded gap" + tr[second.end_key] = b"excluded end" + + version = self.commit(write_sets) + groups = self.consume_through(consumer, version) + self.assertCountEqual( + groups[version], + ( + fdb.CdcMutation( + fdb.CdcMutationType.SET_VALUE, first.begin_key, b"first\x00\xff" + ), + fdb.CdcMutation( + fdb.CdcMutationType.SET_VALUE, + second.begin_key, + b"second\x00\xff", + ), + ), + ) + wait(consumer.acknowledge()) + cursor = consumer.get_position() + self.assertEqual( + self.stream_info().min_version, cursor.last_consumed_version + 1 + ) + + with wait(self.db.resume_cdc_consumer(cursor)) as resumed: + self.assertEqual(resumed.get_position(), cursor) + version = self.commit( + lambda tr: tr.clear_range(self.prefix, self.prefix + b"\xff") + ) + groups = self.consume_through(resumed, version) + self.assertCountEqual( + groups[version], + tuple( + fdb.CdcMutation(fdb.CdcMutationType.CLEAR_RANGE, *r) + for r in (first, second) + ), + ) + wait(resumed.acknowledge()) + self.assertEqual(self.stream_info().ranges, (first, second)) + + def test_native_errors_propagate(self): + with self.assertRaises(fdb.FDBError): + wait(self.db.create_cdc_consumer(self.name)) + self.register() + with self.assertRaises(fdb.FDBError): + wait(self.db.register_cdc_stream(self.name, self.begin, self.end + b"\x00")) + self.assertEqual(self.stream_info().end_key, self.end) + for ranges in ((), ((self.end, self.begin),), ((self.begin, self.begin),)): + with self.subTest(ranges=ranges): + with self.assertRaises(fdb.FDBError): + wait(self.db.register_cdc_stream(self.name, ranges=ranges)) + for kwargs in ( + {}, + {"begin_key": self.begin}, + {"end_key": self.end}, + {"begin_key": self.begin, "ranges": ((self.begin, self.end),)}, + {"end_key": self.end, "ranges": ((self.begin, self.end),)}, + ): + with self.subTest(kwargs=kwargs): + with self.assertRaises(TypeError): + self.db.register_cdc_stream(self.name, **kwargs) + + def test_missing_cdc_symbols_preserves_normal_database_use(self): + impl = fdb.impl + capi = impl._capi + + class WithoutCdcSymbols: + def __getattr__(self, name): + if "_cdc_" in name: + raise AttributeError(name) + return getattr(capi, name) + + with mock.patch.object(impl, "_capi", WithoutCdcSymbols()): + with mock.patch.object(impl, "_cdc_c_api_initialized", False): + impl.init_c_api() + with self.assertRaisesRegex( + RuntimeError, "does not support native CDC" + ): + self.db.list_cdc_streams() + self.assertFalse(impl._cdc_c_api_initialized) + key = self.prefix + b"compatibility" + self.db[key] = b"ordinary value" + self.assertEqual(self.db[key], b"ordinary value") + + def test_cursor_integer_boundaries(self): + for cursor in ( + fdb.CdcCursor(0, -(2**63)), + fdb.CdcCursor(2**64 - 1, 2**63 - 1), + ): + with self.subTest(cursor=cursor): + with wait(self.db.resume_cdc_consumer(cursor)) as consumer: + self.assertEqual(consumer.get_position(), cursor) + for cursor in ( + fdb.CdcCursor(-1, -1), + fdb.CdcCursor(2**64, -1), + fdb.CdcCursor(1, -(2**63) - 1), + fdb.CdcCursor(1, 2**63), + ): + with self.subTest(cursor=cursor): + with self.assertRaises(ValueError): + self.db.resume_cdc_consumer(cursor) + for cursor in (fdb.CdcCursor(1.5, -1), fdb.CdcCursor(1, "0")): + with self.subTest(cursor=cursor): + with self.assertRaises(TypeError): + self.db.resume_cdc_consumer(cursor) + + +class LegacyApiTests(unittest.TestCase): + def test_normal_database_use_and_cdc_version_gate(self): + key = b"python-cdc-legacy/" + uuid.uuid4().bytes + self.addCleanup(self.db.clear, key) + self.db[key] = b"normal database operations still work" + self.assertEqual(self.db[key], b"normal database operations still work") + operations = ( + lambda: self.db.register_cdc_stream(b"legacy", b"a", b"z"), + lambda: self.db.register_cdc_stream(b"legacy", ranges=((b"a", b"z"),)), + lambda: self.db.remove_cdc_stream(b"legacy"), + lambda: self.db.list_cdc_streams(), + lambda: self.db.create_cdc_consumer(b"legacy"), + lambda: self.db.resume_cdc_consumer(fdb.CdcCursor(1, -1)), + ) + for operation in operations: + with self.assertRaisesRegex(RuntimeError, "requires API version 800"): + operation() + + +if __name__ == "__main__": + parser = argparse.ArgumentParser(description="Native CDC Python binding tests") + parser.add_argument("--cluster-file", "-C", required=True) + parser.add_argument("--api-version", type=int, default=fdb.LATEST_API_VERSION) + parser.add_argument("--verbose", "-V", action="store_true") + args = parser.parse_args() + fdb.api_version(args.api_version) + db = fdb.open(args.cluster_file) + db.options.set_transaction_timeout(30000) + NativeCdcTests.db = db + LegacyApiTests.db = db + classes = ( + (CdcDecodingTests, NativeCdcTests) + if args.api_version >= 800 + else (LegacyApiTests,) + ) + suite = unittest.TestSuite( + unittest.defaultTestLoader.loadTestsFromTestCase(cls) for cls in classes + ) + result = unittest.TextTestRunner(verbosity=2 if args.verbose else 1).run(suite) + raise SystemExit(0 if result.wasSuccessful() else 1) diff --git a/cmake/ConfigureCompiler.cmake b/cmake/ConfigureCompiler.cmake index 7e0b26b0404..700372b5db6 100644 --- a/cmake/ConfigureCompiler.cmake +++ b/cmake/ConfigureCompiler.cmake @@ -10,32 +10,37 @@ env_set(USE_GCOV OFF BOOL "Compile with gcov instrumentation") env_set(USE_MSAN OFF BOOL "Compile with memory sanitizer. To avoid false positives you need to dynamically link to a msan-instrumented libc++ and libc++abi, which you must compile separately. See https://github.com/google/sanitizers/wiki/MemorySanitizerLibcxxHowTo#instrumented-libc.") env_set(USE_TSAN OFF BOOL "Compile with thread sanitizer. It is recommended to dynamically link to a tsan-instrumented libc++ and libc++abi, which you can compile separately.") env_set(USE_UBSAN OFF BOOL "Compile with undefined behavior sanitizer") -env_set(FDB_RELEASE_CANDIDATE OFF BOOL "This is a building of a release candidate") -env_set(FDB_RELEASE OFF BOOL "This is a building of a final release") -env_set(USE_CCACHE OFF BOOL "Use ccache for compilation if available") +env_set(USE_CCACHE OFF BOOL "Use ccache for compilation") +env_set(USE_SCCACHE ON BOOL "Use sccache if found") env_set(USE_CLANG_TIDY OFF BOOL "Run clang-tidy during C/C++ compilation") env_set(CLANG_TIDY "" STRING "Path to clang-tidy executable (empty to auto-detect)") env_set(CLANG_TIDY_EXTRA_ARGS "" STRING "Additional clang-tidy arguments (space-separated)") -env_set(RELATIVE_DEBUG_PATHS OFF BOOL "Use relative file paths in debug info") env_set(USE_WERROR OFF BOOL "Compile with -Werror. Recommended for local development and CI.") + default_linker(_use_ld) -env_set(USE_LD "${_use_ld}" STRING - "The linker to use for building: can be LD (system default and same as DEFAULT), BFD, GOLD, or LLD - will be LLD for Clang if available, DEFAULT otherwise") +env_set(USE_LD "${_use_ld}" STRING "The linker to use for building: can be LD (system default and same as DEFAULT), BFD, GOLD, or LLD - will be LLD for Clang if available, DEFAULT otherwise") use_libcxx(_use_libcxx) env_set(USE_LIBCXX "${_use_libcxx}" BOOL "Use libc++") static_link_libcxx(_static_link_libcxx) env_set(STATIC_LINK_LIBCXX "${_static_link_libcxx}" BOOL "Statically link libstdcpp/libc++") +env_set(MAX_LINK_JOBS "4" STRING "Maximum number of link jobs to run in parallel (to avoid OOM)") + env_set(TRACE_PC_GUARD_INSTRUMENTATION_LIB "" STRING "Path to a library containing an implementation for __sanitizer_cov_trace_pc_guard. See https://clang.llvm.org/docs/SanitizerCoverage.html for more info.") env_set(PROFILE_INSTR_GENERATE OFF BOOL "If set, build FDB as an instrumentation build to generate profiles") env_set(PROFILE_INSTR_USE "" STRING "If set, build FDB with profile") + +env_set(RELATIVE_DEBUG_PATHS OFF BOOL "Use relative file paths in debug info") env_set(FULL_DEBUG_SYMBOLS OFF BOOL "Generate full debug symbols") -env_set(ENABLE_LONG_RUNNING_TESTS OFF BOOL "Add a long running tests package") +env_set(COMPRESS_DEBUG_SYMBOLS ON BOOL "Compress debug symbols") set(is_swift_compile "$") set(is_cxx_compile "$,$>") set(is_swift_link "$") set(is_cxx_link "$,$>") +set_property(GLOBAL PROPERTY JOB_POOLS link_job_pool=${MAX_LINK_JOBS}) +set(CMAKE_JOB_POOL_LINK link_job_pool) + set(USE_SANITIZER OFF) if(USE_ASAN OR USE_VALGRIND OR USE_MSAN OR USE_TSAN OR USE_UBSAN) set(USE_SANITIZER ON) @@ -94,6 +99,11 @@ if (USE_CCACHE) set(CMAKE_CXX_COMPILER_LAUNCHER "${CCACHE_PROGRAM}") endif() +if(USE_SCCACHE) + find_package(sccache) +endif() + + include(CheckFunctionExists) set(CMAKE_REQUIRED_INCLUDES stdlib.h malloc.h) if(NOT WIN32) @@ -207,36 +217,37 @@ else() add_compile_options("$<${is_cxx_compile}:-fno-omit-frame-pointer>") + # The default DWARF 5 format does not play nicely with GNU Binutils 2.39 and earlier, resulting + # in tools like addr2line omitting line numbers. + # - Our rockylinux9 gcc compile uses rh/gcc--toolset-13 -> binutils 2.40 + # - Our rockylinux9 clang compile uses rhel9 default -> binutils 2.35.2 + # - MacOS/Darwin ld and lldb do not fully support dwarf-5 either (and also use clang) if(CLANG) - # The default DWARF 5 format does not play nicely with GNU Binutils 2.39 and earlier, resulting - # in tools like addr2line omitting line numbers. We can consider removing this once we are able - # to use a version that has a fix. add_compile_options("$<${is_cxx_compile}:-gdwarf-4>") - endif() - - if(FDB_RELEASE OR FULL_DEBUG_SYMBOLS OR CMAKE_BUILD_TYPE STREQUAL "Debug") - # Configure with FULL_DEBUG_SYMBOLS=ON to generate all symbols for debugging with gdb - # Also generating full debug symbols in release builds. CPack will strip them out - # and create a debuginfo rpm - add_compile_options("$<${is_cxx_compile}:-ggdb>") else() - # Generating minimal debug symbols by default. They are sufficient for testing purposes - add_compile_options("$<${is_cxx_compile}:-ggdb1>") - endif() - - if(CLANG) - # The default DWARF 5 format does not play nicely with GNU Binutils 2.39 and earlier, resulting - # in tools like addr2line omitting line numbers. We can consider removing this once we are able - # to use a version that has a fix. - add_compile_options("$<${is_cxx_compile}:-gdwarf-4>") + add_compile_options("$<${is_cxx_compile}:-gdwarf-5>") + endif() + + # Also generate debug symbols in release builds, + # CPack will strip them out and create a debuginfo rpm. + # Just -g1 by default because g2+ is huge, and cpack rpm debugedit is very slow. + if(FULL_DEBUG_SYMBOLS) + # As much as possible, including macros etc. + add_compile_options("$<${is_cxx_compile}:-g3>") + elseif(CMAKE_BUILD_TYPE STREQUAL "Debug" OR CMAKE_BUILD_TYPE STREQUAL "RelWithDebInfo") + # Reasonable for debugging, including function locals and c++ namespaces. + add_compile_options("$<${is_cxx_compile}:-g2>") + else() + # Minimal debug symbols: enough for backtraces with line numbers, but no locals. + add_compile_options("$<${is_cxx_compile}:-g1>") endif() - if(NOT FDB_RELEASE) - # Enable compression of the debug sections. This reduces the size of the binaries several times. - # We do not enable it release builds, because CPack fails to generate debuginfo packages when - # compression is enabled + # Enable compression of the debuginfo sections, reducing size to ~ 1/3 or less. + # CPack RPM gen used to have problems with compressed debuginfo due to "debugedit" + # but recent versions including the one in rhel9 support it. + if(COMPRESS_DEBUG_SYMBOLS) add_compile_options("$<${is_cxx_compile}:-gz>") - add_link_options("$<${is_cxx_compile}:-gz>") + add_link_options("$<${is_cxx_link}:-gz>") endif() if(TRACE_PC_GUARD_INSTRUMENTATION_LIB) diff --git a/cmake/FDBInstall.cmake b/cmake/FDBInstall.cmake index 5395a67a8a7..322b71ede97 100644 --- a/cmake/FDBInstall.cmake +++ b/cmake/FDBInstall.cmake @@ -6,85 +6,6 @@ function(fdb_install_dirs) set(FDB_INSTALL_DIRS ${ARGV} PARENT_SCOPE) endfunction() -function(install_symlink_impl) - if (NOT WIN32) - return() - endif() - set(options "") - set(one_value_options TO DESTINATION) - set(multi_value_options COMPONENTS) - cmake_parse_arguments(SYM "${options}" "${one_value_options}" "${multi_value_options}" "${ARGN}") - - file(MAKE_DIRECTORY ${CMAKE_CURRENT_BINARY_DIR}/symlinks) - get_filename_component(fname ${SYM_DESTINATION} NAME) - get_filename_component(dest_dir ${SYM_DESTINATION} DIRECTORY) - set(sl ${CMAKE_CURRENT_BINARY_DIR}/symlinks/${fname}) - execute_process(COMMAND ${CMAKE_COMMAND} -E create_symlink ${SYM_TO} ${sl}) - foreach(component IN LISTS SYM_COMPONENTS) - install(FILES ${sl} DESTINATION ${dest_dir} COMPONENT ${component}) - endforeach() -endfunction() - -function(install_symlink) - if(NOT WIN32 AND NOT OPEN_FOR_IDE) - return() - endif() - set(options "") - set(one_value_options COMPONENT LINK_DIR FILE_DIR LINK_NAME FILE_NAME) - set(multi_value_options "") - cmake_parse_arguments(IN "${options}" "${one_value_options}" "${multi_value_options}" "${ARGN}") - - set(rel_path "") - string(REGEX MATCHALL "\\/" slashes "${IN_LINK_NAME}") - foreach(ignored IN LISTS slashes) - set(rel_path "../${rel_path}") - endforeach() - if("${IN_FILE_DIR}" MATCHES "bin") - if("${IN_LINK_DIR}" MATCHES "lib") - install_symlink_impl( - TO "../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "lib/${IN_LINK_NAME}" - COMPONENTS "${IN_COMPONENT}-tgz") - install_symlink_impl( - TO "../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "usr/lib64/${IN_LINK_NAME}" - COMPONENTS - "${IN_COMPONENT}-el9" - "${IN_COMPONENT}-deb") - install_symlink_impl( - TO "../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "usr/lib64/${IN_LINK_NAME}" - COMPONENTS "${IN_COMPONENT}-deb") - elseif("${IN_LINK_DIR}" MATCHES "bin") - install_symlink_impl( - TO "../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "bin/${IN_LINK_NAME}" - COMPONENTS "${IN_COMPONENT}-tgz") - install_symlink_impl( - TO "../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "usr/bin/${IN_LINK_NAME}" - COMPONENTS - "${IN_COMPONENT}-el9" - "${IN_COMPONENT}-deb") - elseif("${IN_LINK_DIR}" MATCHES "fdbmonitor") - install_symlink_impl( - TO "../../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "lib/foundationdb/${IN_LINK_NAME}" - COMPONENTS "${IN_COMPONENT}-tgz") - install_symlink_impl( - TO "../../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "usr/lib/foundationdb/${IN_LINK_NAME}" - COMPONENTS - "${IN_COMPONENT}-el9" - "${IN_COMPONENT}-deb") - else() - message(FATAL_ERROR "Unknown LINK_DIR ${IN_LINK_DIR}") - endif() - else() - message(FATAL_ERROR "Unknown FILE_DIR ${IN_FILE_DIR}") - endif() -endfunction() - function(symlink_files) if (NOT WIN32) set(options "") diff --git a/cmake/InstallLayout.cmake b/cmake/InstallLayout.cmake index 02bb8971f4f..ed4e6e8535b 100644 --- a/cmake/InstallLayout.cmake +++ b/cmake/InstallLayout.cmake @@ -1,93 +1,5 @@ include(FDBInstall) -function(install_symlink_impl) - if (NOT WIN32) - set(options "") - set(one_value_options TO DESTINATION) - set(multi_value_options COMPONENTS) - cmake_parse_arguments(SYM "${options}" "${one_value_options}" "${multi_value_options}" "${ARGN}") - - file(MAKE_DIRECTORY ${CMAKE_CURRENT_BINARY_DIR}/symlinks) - get_filename_component(fname ${SYM_DESTINATION} NAME) - get_filename_component(dest_dir ${SYM_DESTINATION} DIRECTORY) - set(sl ${CMAKE_CURRENT_BINARY_DIR}/symlinks/${fname}) - execute_process(COMMAND ${CMAKE_COMMAND} -E create_symlink ${SYM_TO} ${sl}) - foreach(component IN LISTS SYM_COMPONENTS) - install(FILES ${sl} DESTINATION ${dest_dir} COMPONENT ${component}) - endforeach() - endif() -endfunction() - -function(install_symlink) - if(NOT WIN32 AND NOT OPEN_FOR_IDE) - set(options "") - set(one_value_options COMPONENT LINK_DIR FILE_DIR LINK_NAME FILE_NAME) - set(multi_value_options "") - cmake_parse_arguments(IN "${options}" "${one_value_options}" "${multi_value_options}" "${ARGN}") - - set(rel_path "") - string(REGEX MATCHALL "\\/" slashes "${IN_LINK_NAME}") - foreach(ignored IN LISTS slashes) - set(rel_path "../${rel_path}") - endforeach() - if("${IN_FILE_DIR}" MATCHES "bin") - if("${IN_LINK_DIR}" MATCHES "lib") - install_symlink_impl( - TO "../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "lib/${IN_LINK_NAME}" - COMPONENTS "${IN_COMPONENT}-tgz") - install_symlink_impl( - TO "../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "usr/lib64/${IN_LINK_NAME}" - COMPONENTS "${IN_COMPONENT}-el9" - "${IN_COMPONENT}-deb") - install_symlink_impl( - TO "../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "usr/lib64/${IN_LINK_NAME}" - COMPONENTS "${IN_COMPONENT}-deb") - elseif("${IN_LINK_DIR}" MATCHES "bin") - install_symlink_impl( - TO "../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "bin/${IN_LINK_NAME}" - COMPONENTS "${IN_COMPONENT}-tgz") - install_symlink_impl( - TO "../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "usr/bin/${IN_LINK_NAME}" - COMPONENTS "${IN_COMPONENT}-el9" - "${IN_COMPONENT}-deb") - elseif("${IN_LINK_DIR}" MATCHES "fdbmonitor") - install_symlink_impl( - TO "../../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "lib/foundationdb/${IN_LINK_NAME}" - COMPONENTS "${IN_COMPONENT}-tgz") - install_symlink_impl( - TO "../../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "usr/lib/foundationdb/${IN_LINK_NAME}" - COMPONENTS "${IN_COMPONENT}-el9" - "${IN_COMPONENT}-deb") - else() - message(FATAL_ERROR "Unknown LINK_DIR ${IN_LINK_DIR}") - endif() - else() - message(FATAL_ERROR "Unknown FILE_DIR ${IN_FILE_DIR}") - endif() - endif() -endfunction() - -function(symlink_files) - if (NOT WIN32) - set(options "") - set(one_value_options LOCATION SOURCE) - set(multi_value_options TARGETS) - cmake_parse_arguments(SYM "${options}" "${one_value_options}" "${multi_value_options}" "${ARGN}") - - file(MAKE_DIRECTORY ${CMAKE_BINARY_DIR}/${SYM_LOCATION}) - foreach(component IN LISTS SYM_TARGETS) - execute_process(COMMAND ${CMAKE_COMMAND} -E create_symlink ${SYM_SOURCE} ${CMAKE_BINARY_DIR}/${SYM_LOCATION}/${component} WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/${SYM_LOCATION}) - endforeach() - endif() -endfunction() - fdb_install_packages(TGZ DEB EL9 VERSIONED) fdb_install_dirs(BIN SBIN LIB INCLUDE ETC LOG DATA) message(STATUS "FDB_INSTALL_DIRS -> ${FDB_INSTALL_DIRS}") @@ -132,7 +44,7 @@ set(CPACK_PROJECT_CONFIG_FILE "${CMAKE_BINARY_DIR}/packaging/CPackConfig.cmake") # User config ################################################################################ -set(GENERATE_DEBUG_PACKAGES "${FDB_RELEASE}" CACHE BOOL "Build debug rpm/deb packages (default: only ON for FDB_RELEASE)") +set(GENERATE_DEBUG_PACKAGES ON CACHE BOOL "Build debug rpm/deb packages") ################################################################################ # Alternatives config @@ -230,6 +142,7 @@ string(REPLACE "-" "_" FDB_PACKAGE_VERSION ${FDB_VERSION}) set(CPACK_RPM_PACKAGE_GROUP ${CURRENT_GIT_VERSION}) set(CPACK_RPM_PACKAGE_LICENSE "Apache 2.0") set(CPACK_RPM_PACKAGE_NAME "foundationdb") + set(CPACK_RPM_CLIENTS-EL9_PACKAGE_NAME "${CPACK_RPM_PACKAGE_NAME}-clients") set(CPACK_RPM_CLIENTS-EL9_FILE_NAME "${CPACK_RPM_CLIENTS-EL9_PACKAGE_NAME}-${FDB_PACKAGE_VERSION}${package_version_postfix}.el9.${CMAKE_SYSTEM_PROCESSOR}.rpm") set(CPACK_RPM_CLIENTS-EL9_DEBUGINFO_FILE_NAME "${CPACK_RPM_CLIENTS-EL9_PACKAGE_NAME}-${FDB_PACKAGE_VERSION}${package_version_postfix}.el9-debuginfo.${CMAKE_SYSTEM_PROCESSOR}.rpm") @@ -261,6 +174,14 @@ set(CPACK_RPM_SERVER-VERSIONED_PACKAGE_REQUIRES "${CPACK_COMPONENT_CL set(CPACK_RPM_SERVER-VERSIONED_POST_INSTALL_SCRIPT_FILE ${CMAKE_BINARY_DIR}/packaging/multiversion/server/postinst-rpm) set(CPACK_RPM_SERVER-VERSIONED_PRE_UNINSTALL_SCRIPT_FILE ${CMAKE_BINARY_DIR}/packaging/multiversion/server/prerm) +# Avoid VERSIONED client package conflicting with main client package of same exact version, +# due to the build-id links in /usr/lib/.build-id/ +set(CPACK_RPM_SPEC_MORE_DEFINE " +%if \\\"%{name}\\\" == \\\"${CPACK_RPM_CLIENTS-VERSIONED_PACKAGE_NAME}\\\" +%define _build_id_links none +%endif +") + file(MAKE_DIRECTORY "${CMAKE_BINARY_DIR}/packaging/emptydir") fdb_install(DIRECTORY "${CMAKE_BINARY_DIR}/packaging/emptydir/" DESTINATION data COMPONENT server) fdb_install(DIRECTORY "${CMAKE_BINARY_DIR}/packaging/emptydir/" DESTINATION log COMPONENT server) @@ -273,8 +194,6 @@ set(CPACK_RPM_EXCLUDE_FROM_AUTO_FILELIST_ADDITION "/usr/lib64/cmake" "/etc/foundationdb" "/usr/lib64/pkgconfig" - "/usr/lib64/python2.7" - "/usr/lib64/python2.7/site-packages" "/var" "/var/log" "/var/lib" @@ -286,10 +205,9 @@ set(CPACK_RPM_EXCLUDE_FROM_AUTO_FILELIST_ADDITION "/usr/lib/foundationdb" "/usr/lib/cmake" "/usr/lib/foundationdb-${FDB_VERSION}${FDB_BUILDTIME_STRING}/etc/foundationdb" - ) +) set(CPACK_RPM_BUILD_SOURCE_DIRS_PREFIX "/usr/src") set(CPACK_RPM_DEBUGINFO_PACKAGE ${GENERATE_DEBUG_PACKAGES}) -#set(CPACK_RPM_BUILD_SOURCE_FDB_INSTALL_DIRS_PREFIX /usr/src) set(CPACK_RPM_COMPONENT_INSTALL ON) ################################################################################ diff --git a/contrib/Joshua/scripts/binding_test_start.sh b/contrib/Joshua/scripts/binding_test_start.sh index be4f84a3d8c..2ecb837fb6c 100755 --- a/contrib/Joshua/scripts/binding_test_start.sh +++ b/contrib/Joshua/scripts/binding_test_start.sh @@ -3,5 +3,12 @@ set -e set -o pipefail +# The Joshua agent image sets these to point at an external client directory that +# may hold libfdb_c versions too old for the binaries under test, which makes the +# multi-version client fail to load (api_function_missing, 2204). The tests here +# use the client shipped in the package, so drop them (see also bindingTest.sh). +unset FDB_NETWORK_OPTION_EXTERNAL_CLIENT_DIRECTORY +unset FDB_NETWORK_OPTION_EXTERNAL_CLIENT_LIBRARY + # It is necessary to tee to output.log in case timeout happens python3 ./binding_test.py --stop-at-failure 10 --fdbserver-path $(pwd)/fdbserver --fdbcli-path $(pwd)/fdbcli --libfdb-path $(pwd) --num-ops 1000 --num-hca-ops 100 --concurrency 5 --test-timeout 60 --random 2>&1 | tee output.log diff --git a/contrib/local_cluster/lib/fdb_process.py b/contrib/local_cluster/lib/fdb_process.py index 9d91d31f9ab..33267698814 100644 --- a/contrib/local_cluster/lib/fdb_process.py +++ b/contrib/local_cluster/lib/fdb_process.py @@ -15,7 +15,7 @@ FDBSERVER_TIMEOUT: float = 180.0 -FDBCLI_TIMEOUT: float = 180.0 +FDBCLI_TIMEOUT: float = 30.0 FDBCLI_RETRY_TIME: float = 1.0 @@ -37,7 +37,7 @@ def __init__(self, strerror: str, filename: str, *args, **kwargs): @property def filename(self) -> str: """Name of the file""" - return self._strerror + return self._filename @property def strerror(self) -> str: @@ -207,10 +207,30 @@ async def get_server_status(cluster_file: str) -> Union[Dict, None]: cluster_file=cluster_file, commands="status json" ).run() try: - output = await asyncio.wait_for(fdbcli_process.stdout.read(-1), FDBCLI_TIMEOUT) + stdout, stderr = await asyncio.wait_for( + fdbcli_process.communicate(), FDBCLI_TIMEOUT + ) + except asyncio.TimeoutError: + logger.warning("Timed out waiting for fdbcli [status json]") + fdbcli_process.kill() await fdbcli_process.wait() - return json.loads(output.decode()) - except TimeoutError: + return None + + if fdbcli_process.returncode != 0: + logger.warning( + f"fdbcli [status json] exited with {fdbcli_process.returncode}, " + f"stderr: {stderr.decode(errors='replace').strip()[:1024]!r}" + ) + return None + + try: + return json.loads(stdout.decode()) + except json.JSONDecodeError: + logger.warning( + f"fdbcli [status json] emitted non-JSON output: " + f"{stdout.decode(errors='replace')[:1024]!r}, " + f"stderr: {stderr.decode(errors='replace').strip()[:1024]!r}" + ) return None diff --git a/contrib/mtlsbenchmark/client.sh b/contrib/mtlsbenchmark/client.sh index dafa2369a44..d6935f2f08a 100644 --- a/contrib/mtlsbenchmark/client.sh +++ b/contrib/mtlsbenchmark/client.sh @@ -9,9 +9,8 @@ # knob_disable_mainthread_tls_handshake is enabled to use background threads for TLS handshakes only # knob_tls_handshake_flowlock_priority is set to 8900 for enabling TLS flowlock priority as high as the handshake priority. Default is 7000. -taskset -c 0-0 /root/build_output/bin/fdbserver \ - -r unittests \ - -f :/network/p2ptest \ +taskset -c 0-0 "${FDBRPC_NETWORK_TEST:-/root/build_output/bin/fdbrpc_network_test}" \ + --mode p2p \ --test_remoteAddresses=127.0.0.1:4500:tls \ --test_targetDuration=10 \ --test_connectionsOut=10 \ diff --git a/contrib/mtlsbenchmark/readme.md b/contrib/mtlsbenchmark/readme.md index 34c2b58e1cd..418e082931c 100644 --- a/contrib/mtlsbenchmark/readme.md +++ b/contrib/mtlsbenchmark/readme.md @@ -5,7 +5,8 @@ A testing framework for benchmarking TLS performance in peer-to-peer network sce ## Prerequisites - OpenSSL or compatible tool for certificate generation -- Environment for FoundationDB unit test execution +- Build `fdbrpc_network_test` (`cmake --build build --target fdbrpc_network_test`). +- Set `FDBRPC_NETWORK_TEST` to the executable path if it differs from `/root/build_output/bin/fdbrpc_network_test`. ## Quick Start @@ -22,14 +23,17 @@ Generate the required certificate files: ### Step 2: Configure Test in Scripts -The test scripts support two unit tests that can be configured: +The standalone network diagnostic supports two P2P modes: -| Test Mode | Purpose | Configuration (set by -f) | +| Test Mode | Purpose | Configuration (set by --mode) | |-----------|---------|---------------| -| **Long Running** | Testing with connections and messages | `:/network/p2ptest` | -| **One Shot** | One-time connection only and no message | `:/network/p2poneshottest` | +| **Long Running** | Testing with connections and messages | `p2p` | +| **One Shot** | One-time connection only and no message | `p2p-oneshot` | -Set the desired test mode in your script before running. +Set the desired test mode in your script before running. The `--test_*`, +`--knob_*`, and `--tls_*` options retain their meanings. These modes were +previously run through `fdbserver -r unittests`; they now use the standalone +[`fdbrpc_network_test`](../../fdbrpc/tests/networktest.md) executable. ## Folder Structure diff --git a/contrib/mtlsbenchmark/server.sh b/contrib/mtlsbenchmark/server.sh index 037a9395252..731f1b023cf 100644 --- a/contrib/mtlsbenchmark/server.sh +++ b/contrib/mtlsbenchmark/server.sh @@ -11,9 +11,8 @@ # knob_tls_handshake_flowlock_priority is set to 8900 for enabling TLS flowlock priority as high as the handshake priority. Default is 7000. # knob_tls_handshake_timeout_seconds is set to 3.0 seconds to timeout a handshake if not completed in 3 seconds. The default is 2.0 seconds. -taskset -c 1-1 /root/build_output/bin/fdbserver \ - -r unittests \ - -f :/network/p2ptest \ +taskset -c 1-1 "${FDBRPC_NETWORK_TEST:-/root/build_output/bin/fdbrpc_network_test}" \ + --mode p2p \ --test_listenerAddresses=0.0.0.0:4500:tls \ --test_targetDuration=0 \ --knob_tls_handshake_limit=1000 \ diff --git a/design/cdc.md b/design/cdc.md index d5bfe37da2e..0bddd1aa5fd 100644 --- a/design/cdc.md +++ b/design/cdc.md @@ -3,29 +3,29 @@ ## Objective Native Change Data Capture (CDC) provides a FoundationDB-native mechanism for -reading committed mutations for a registered key range. A client registers a -named stream, creates a consumer for that name, consumes batches of mutations, -and acknowledges processed versions. The implementation persists enough state -to retain unread TLog data and to resume stream service after CDC proxy failure -or transaction-system recovery. +reading committed mutations for a registered set of key ranges. A client +registers a named stream, creates a consumer for that name, consumes batches of +mutations, and acknowledges processed versions. The implementation persists +enough state to retain unread TLog data and to resume stream service after CDC +proxy failure or transaction-system recovery. ## Background -This design describes the native C++ interface, its C binding, and its server -implementation. The feature is disabled by default behind `ENABLE_NATIVE_CDC`; +This design describes the native C++ interface, its C and Python bindings, and +its server implementation. The feature is disabled by default behind `ENABLE_NATIVE_CDC`; the native CDC workloads explicitly enable it, and simulation may randomly -enable it. The client interface is exposed through the native C++ API and C -binding; it does not expose an external protocol compatibility guarantee. +enable it. The client interface is exposed through the native C++ API and C and +Python bindings; it does not expose an external protocol compatibility guarantee. The implementation uses the following terms: -* A **stream** is a durable named registration for a fixed user key range. +* A **stream** is a durable named registration for a fixed set of user key ranges. * A **cursor** identifies one stream and the version through which a consumer has read. * A **CDC tag** is a TLog tag with locality `tagLocalityCDC`. Commit proxies append these tags to mutations covered by registered streams. * A **CDC proxy** reads tagged TLog mutation streams, filters mutations to a - registered range, serves consumers, and coordinates acknowledgement-driven + registered range set, serves consumers, and coordinates acknowledgement-driven log popping. CDC is not implemented as a storage server change feed. It captures mutations @@ -36,12 +36,11 @@ and release its own log history without changing user data storage. Native CDC is intended to provide: -* Durable, named registrations for single key ranges in normal user key - space. The initial API intentionally registers exactly one half-open - `[begin, end)` range per stream; callers that need multiple disjoint ranges - register multiple streams. +* Durable, named registrations for non-empty sets of half-open `[begin, end)` + ranges in normal user key space. A stream captures the union of its ranges + and excludes the gaps between them. * A consumer API in which a client only needs a stream name after - registration, rather than repeating its registered range on every read. + registration, rather than repeating its registered ranges on every read. * Ordered mutation batches identified by FoundationDB commit versions. * Durable acknowledgements that determine how much CDC-tagged TLog history may be popped. @@ -81,11 +80,11 @@ The current implementation does not attempt to provide: arbitrary application state. Such an API would be useful for queue-like transactional asynchronous processing pipelines, but the initial interface leaves that composition to the consumer. -* Dynamic stream range changes. A name is registered for one range; changing a - range requires removing and registering a stream. +* Dynamic stream range changes. A name is registered for an immutable range + set; changing membership requires removing and registering a stream. * Throughput-aware assignment of streams across CDC proxies. * Throughput-aware movement of streams between CDC tags. -* Language-specific bindings beyond the C API. +* Language-specific bindings beyond the C and Python APIs. ## Client interface @@ -95,12 +94,13 @@ value types and the thread-safe surface shared with language bindings are in roles are in the private `fdbclient/NativeCdcInternal.h`; cursor and wire request types are in `fdbclient/CDCProxyInterface.h`. The public C binding is declared in `bindings/c/foundationdb/fdb_c.h` and documented in -`documentation/sphinx/source/api-c.rst`. +`documentation/sphinx/source/api-c.rst`. The Python binding is documented in +`documentation/sphinx/source/api-python.rst`. `CDCStreamId` is a `uint64_t` typedef. CDC tag IDs are 16-bit, so one configured tag pool can contain at most 65,536 distinct tags. ```cpp -Future registerNativeCdcStreamClient(Database cx, Key name, KeyRange keys); +Future registerNativeCdcStreamClient(Database cx, Key name, std::vector ranges); Future removeNativeCdcStreamClient(Database cx, Key name); Future> listNativeCdcStreamsClient(Database cx); @@ -108,9 +108,11 @@ Future> createNativeCdcConsumer(Database cx, Key na Reference resumeNativeCdcConsumer(Database cx, CDCCursor position); ``` -`registerNativeCdcStreamClient()` accepts exactly one `KeyRange`. The range is -interpreted with FoundationDB's usual half-open `[begin, end)` semantics. -Multi-range registration is not part of the initial API. +`registerNativeCdcStreamClient()` accepts a non-empty vector of `KeyRange` +values, each interpreted with FoundationDB's usual half-open `[begin, end)` +semantics. Registration sorts the ranges and merges overlaps, duplicates, and +adjacent intervals into a canonical union. All ranges share one stream identity, +CDC tag assignment, proxy owner, cursor, and durable acknowledgement watermark. Registration and removal are low-rate control-plane operations intended for stable stream lifecycles, not per-request stream churn. The initial implementation does not define a supported registrations-per-second target: @@ -123,7 +125,7 @@ A stream registration contains: struct NativeCdcStreamInfo { Key name; CDCStreamId streamId; - KeyRange keys; + std::vector ranges; Version minVersion; }; ``` @@ -175,7 +177,8 @@ struct CDCConsumeReply { A typical consumer loop is: ```cpp -co_await registerNativeCdcStreamClient(db, "orders"_sr, KeyRangeRef("order/"_sr, "order0"_sr)); +co_await registerNativeCdcStreamClient( + db, "orders"_sr, { KeyRangeRef("order/"_sr, "order0"_sr), KeyRangeRef("payment/"_sr, "payment0"_sr) }); Reference consumer = co_await createNativeCdcConsumer(db, "orders"_sr); while (true) { @@ -220,17 +223,20 @@ version. ### Registration and removal semantics -`registerNativeCdcStreamClient()` accepts a non-empty stream name and a -non-empty range entirely within normal user keys. Registration of an existing -name with the same range is idempotent. Registering an existing name with a -different range is rejected. +`registerNativeCdcStreamClient()` accepts a non-empty stream name and between +1 and 1,024 non-empty ranges entirely within normal user keys. The encoded +canonical range set must fit within FoundationDB's value-size limit. +Registration of an existing name with the same canonical union is idempotent, +regardless of input order, duplication, overlap, or adjacent subdivisions. +Registering an existing name with a different union is rejected. Listing a +stream returns its canonical ranges in key order. Registration establishes an initial minimum version using the registration transaction's commit version. Mutations committed after the registration has become visible are routed to the stream's CDC tag. The initial minimum version also supplies the first retention watermark for its TLog history. When CDC admission is disabled, this gate applies only to creation of a new -name. Repeating an existing same-name/same-range registration remains +name. Repeating an existing same-name/same-range-set registration remains idempotent, including repair of a missing durable owner, so an administrator can still drain state created while the feature was enabled. @@ -244,9 +250,18 @@ will never be assigned again. ### Consumption and expiration -Consumption is ordered by commit version. Mutations from a clear range are -intersected with the stream's registered range before being returned; a -single-key mutation is returned only if its key is within that range. +Consumption is ordered by commit version. A single-key mutation is returned +only if its key is in one of the registered ranges. A clear range is intersected +with every selected interval it overlaps and emits one clear for each non-empty +intersection, in key order. These fragments remain in the original mutation +position relative to other mutations from the same commit version. For example, +a stream selecting `[a,c)` and `[x,z)` returns `[b,c)` and `[x,y)` for a clear +of `[b,y)`, and never clears the unselected gap `[c,x)`. + +A single cursor and acknowledgement cover all selected ranges. The consumer +must process every range through the acknowledged version; a slow range holds +back retention for the whole stream. Consumers that need independent progress +or lifecycle management should use separate streams. For an active stream, unacknowledged CDC mutations are retained by its durable minimum version: TLogs must not pop tagged data that the stream may still @@ -312,7 +327,7 @@ consumers. CDC proxies do not participate in committing user transactions. They consume the extra tagged log streams, buffer readable results, filter shared tagged -data back to each stream's registered range, and pop data after durable +data back to each stream's registered range set, and pop data after durable acknowledgement permits it. The cluster controller recruits CDC proxies, publishes their interfaces, and @@ -335,21 +350,28 @@ in transaction state: | --- | --- | --- | | `\xff/cdc/name/` | `CDCStreamId` | Resolves a user-visible name to its durable stream identity. | | `\xff/cdc/maxStreamId` | `CDCStreamId` | Allocates monotonic stream identifiers. | -| `\xff/cdc/keys/` | `KeyRange` | Stores the immutable registered range for an active stream. | -| `\xff/cdc/tagHistory///` | empty | Records the CDC tag assignment history used for routing and historical reads. | +| `\xff/cdc/keys/` | `std::vector` | Stores the canonical immutable registered range set for an active stream. | +| `\xff/cdc/tagHistory///` | empty or commit versionstamp | Records initial assignments and exact committed live-retag boundaries. | | `\xff/cdc/proxies//` | empty | Stores the CDC proxy assigned to an active stream. | | `\xff/cdc/proxyAssignmentChange` | version/change signal | Wakes ownership monitoring when durable assignments change. | | `\xff/cdc/retiredTagPop/` | empty | Retains recovery-visible pending final-pop work after removal. | Tag history is versioned so the data model can support a stream moving between tags without forgetting which old log streams may still contain unread -mutations. The initial implementation writes the initial assignment and reads -the history; dynamic throughput-driven reassignment is future work. The initial -history entry uses the registration transaction's read version as a +mutations. Production writers currently create only the initial assignment; +throughput-driven reassignment is future work. The initial history entry uses +the registration transaction's read version as a conservative inclusive lower bound. The versionstamped `minVersion` uses the commit version, and the proxy starts at the maximum of those two values, so the earlier history boundary cannot expose pre-registration mutations. +A live retag preserves that key layout, but writes a ten-byte commit +versionstamp in the value. Its key uses the transaction read version only to +order successive assignments; readers use the committed version from the value +as the exact cutover. Empty values retain their original interpretation. A +retag transaction reads and revalidates its previous history, so its read version +is later than that history key and the key order remains monotonic. + ### Storage-backed system data These keys are in the storage-server-backed `\xff\x02` system key range rather @@ -376,15 +398,16 @@ actual final pop to perform. Registration runs as a durable metadata transaction: -1. It validates the stream name and registered normal key range. +1. It validates the stream name, range count, normal key ranges, and encoded + metadata size, and canonicalizes the range union. 2. It checks whether the name is already registered and applies the idempotent - same-name/same-range rule, even when admission is disabled. + same-name/same-range-set rule, even when admission is disabled. 3. For a new name, it validates the feature knob. 4. It allocates a new monotonically increasing `CDCStreamId`. 5. It selects a CDC tag using current active stream counts. The allocator uses the least populated tag among `NATIVE_CDC_TAG_COUNT` tags (256 by default), choosing the lowest tag ID on a tie. -6. It records the stream name, range, initial tag history entry, and +6. It records the stream name, canonical ranges, initial tag history entry, and versionstamped initial minimum version. 7. It records an available CDC proxy owner and signals assignment monitoring. @@ -412,6 +435,63 @@ validation also rejects stale representatives left by older metadata writers. The allocator's stream-count scan remains necessary, and ownership discovery still scans global metadata when the representative is absent or invalid. +### Balancing ownership between live CDC proxies + +Live CDC proxies can own unequal numbers of active streams. The balancer reduces +this stream-count skew while preserving ownership of each shared current-tag +group. It does not balance producer bytes, filtering cost, or consumer lag, and +cannot divide one hot stream among proxies. Groups with mixed owners are left +unchanged; repairing their ownership is outside this policy. Throughput-aware +proxy placement needs measured load and a separate policy; the opt-in +tag-retagging controller can inform a later version. + +`CDC_PROXY_REBALANCE_ENABLED` is disabled by default. When enabled, the cluster +controller makes at most one live ownership move per +`CDC_PROXY_REBALANCE_INTERVAL` (60 seconds by default) while fully recovered. +It groups active streams by their current CDC tag and moves a complete group +only when that strictly reduces the stream-count difference between two +published proxies. The controller +skips a pass when the metadata exceeds 512 active streams or assignments, or +2,048 tag-history rows, or one MiB in any of those three ranges. Each pass has +a five-second transaction timeout and at most three attempts. It skips +individual groups larger than 64 streams or +with an in-progress versionstamped tag transition. These are conservative +balancer limits, not CDC registration or cluster capacity limits. + +The move validates durable streams, tags, and owners in one transaction, changes +every member's assignment, and signals the existing ownership monitor. Its +publication wakes the old proxy to drop buffered state and the new proxy to +reload from durable acknowledgement watermarks. Clients may replay delivered +but unacknowledged mutations, as with proxy replacement. Tag routing, stream +identities, acknowledgement, and safe-pop metadata do not change. Disabling the +balancer or CDC admission stops future moves without requiring a stream drain. + +### Retag compatibility and cleanup + +A committed target history row at version `C` divides delivery into the old tag +below `C` and the target tag starting at `C`. Consumers refresh history even +when the owner has not changed. Delivery is bounded by the metadata snapshot's +read version, so an old-tag read cannot skip across a cutover that committed +after that snapshot. Existing unacknowledged delivery positions remain valid +on the same owner. Recovery retains both required tagged log intervals. + +Each stream has at most one pending move. Both history rows remain until its +durable minimum required version reaches `C`. The data distributor then +atomically replaces them with one canonical empty-valued target row at `C` +and records retired-pop work for the old tag. Shared-tag acknowledgement, +recovery, and final-pop checks still govern physical cleanup. + +The cleanup worker pages through durable streams independently of admission +or any future sampling/move policy. `NATIVE_CDC_RETAG_CLEANUP_INTERVAL` defaults +to 30 seconds. Assignment changes trigger another traversal without restarting +an in-progress traversal, and pending histories are revisited even when an +acknowledgement notification is lost. Disabling CDC admission still allows +existing streams and transitions to drain or be removed. + +This compatibility foundation does not schedule new retags. Its transactional +writer helper is exercised by simulation fixtures to create future-format +histories, including writes between the metadata read and commit versions. + ### Metadata lifecycle example Assume a client registers stream name `orders` for range @@ -421,7 +501,7 @@ assigns proxy `P1`. Registration writes: * Transaction state `\xff/cdc/name/orders -> 7`. -* Transaction state `\xff/cdc/keys/7 -> ["order/", "order0")`. +* Transaction state `\xff/cdc/keys/7 -> { ["order/", "order0") }`. * Transaction state `\xff/cdc/tagHistory/7/995/tagLocalityCDC:3 -> empty`. * Transaction state `\xff/cdc/proxies/7/P1 -> empty` and the assignment-change signal. @@ -429,7 +509,7 @@ Registration writes: If the consumer later acknowledges mutations through version `1200`, `\xff\x02/cdc/minVersion/7` advances to `1201`. If the stream is then removed -at version `1500`, removal deletes the active name, range, proxy, tag-history, +at version `1500`, removal deletes the active name, ranges, proxy, tag-history, and `minVersion` rows, and writes retired final-pop work for `tagLocalityCDC:3`: a transaction-state `\xff/cdc/retiredTagPop/` marker and a storage-backed `\xff\x02/cdc/retiredTagPopVersion/` watermark for @@ -454,7 +534,7 @@ proxy processing. The cost of a broad clear range is proportional to the number of CDC stream ranges and tags it intersects, not to the number of keys in the cleared range. The logged CDC payload remains a clear-range mutation on each relevant tag, and -the CDC proxy later clips that clear to the consumer's registered range. A +the CDC proxy later clips that clear to the consumer's registered ranges. A clear that spans many CDC ranges can therefore add many CDC tag destinations and later produce many per-stream clipped clears. Commit-proxy and CDC-proxy metrics should make this visible by reporting CDC routing matches, CDC tag @@ -463,8 +543,9 @@ fanout, filtered bytes, and consumer lag. A shared CDC tag is a multiplexed log stream. A mutation routed because of stream A may be read by the proxy serving stream B if both share the tag. Consequently, the CDC proxy filters every read mutation against B's registered -range before returning it to B's consumer. Filtering also clips clear ranges -to the stream range. +range set before returning it to B's consumer. Filtering splits clear ranges +at unselected gaps. A clear intersecting multiple ranges of one stream receives +that stream's CDC tag only once. Shared-tag false positives are expected, especially when `NATIVE_CDC_TAG_COUNT` is small or active streams are unevenly distributed. The @@ -496,14 +577,14 @@ mapping for the normally configured IDs. A CDC proxy owns a set of active stream IDs. For each owned stream it loads: -* The registered key range. +* The canonical registered key ranges. * The durable minimum required version. * Its current CDC tag and versioned tag history. The proxy reads data from TLogs through `LogSystemConsumer::peekSingle()`. When a stream has historical assignments, the proxy uses the history to select the tag appropriate for the version interval it is reading. It filters -mutations to the registered range and stores versioned mutation batches in a +mutations to the registered range set and stores versioned mutation batches in a per-stream in-memory buffer. All raw peek windows and stream buffers owned by one CDC proxy share a @@ -511,12 +592,16 @@ All raw peek windows and stream buffers owned by one CDC proxy share a retain one separately capped reply arena from every candidate TLog it consults. CDC history cursors therefore disable cross-generation constructor prefetch, report the maximum number of reply arenas one active generation can retain, -and reserve that count times `MAXIMUM_PEEK_BYTES` before issuing a peek. The -proxy marks these delivery cursors with the same per-reply limit; recovery +and cap each reply at the smaller of `MAXIMUM_PEEK_BYTES` and +`CDC_PROXY_BUFFER_BYTES / (retainedReplyCount + 1)`. The proxy reserves the aggregate raw reply +budget plus one reply-sized materialization window before issuing a peek. +It marks these delivery cursors with the same per-reply limit; recovery cursors remain uncapped so that transaction-system replay is not constrained -by a delivery memory knob. The pass also reserves a bounded materialization -window. It retains the aggregate -raw reservation while filtering and copying, then releases it and transfers +by a delivery memory knob. If materialization needs a larger window, the reader +releases its cursor and reservation before retrying with the full required +reservation. Competing readers cannot hold partial reservations while waiting +for each other to release capacity. The pass retains the aggregate raw reservation +while filtering and copying, then releases it and transfers only accepted filtered bytes to the stream buffers. Acknowledgement or stream removal releases those retained permits. The usable retained-batch capacity is the configured CDC budget minus this topology-dependent raw reservation; a @@ -530,16 +615,17 @@ TLog retention are the source of resumability, while the proxy buffer is a delivery optimization. One tagged TLog message can match many overlapping streams. The proxy estimates -that expansion per stream and commit version, materializes only a subset that -fits the current bounded pass, and reopens the tag cursor for the remaining -streams. It never requests raw-plus-retained permits beyond +that expansion, including clear fragments across disjoint ranges, per stream +and commit version, materializes only a subset that fits the current bounded +pass, and reopens the tag cursor for the remaining streams. It never requests +raw-plus-retained permits beyond `CDC_PROXY_BUFFER_BYTES`. If the filtered mutations for one stream at one commit version exceed the capacity remaining after the raw peek reservation, that consume fails with `server_overloaded`; operators must configure the budget to hold both the largest raw peek and the largest supported filtered transaction for one stream. -The TLog applies `MAXIMUM_PEEK_BYTES` at complete commit-version boundaries. +The TLog applies the requested reply limit at complete commit-version boundaries. When several individually valid versions would exceed one raw reply, it returns the prefix and leaves the next version for a later peek. It reports an oversized CDC peek only when one complete version cannot fit by itself; a @@ -556,7 +642,9 @@ default) and contains only complete commit-version groups. If more buffered data is available, the reply stops before the next version and advances `lastConsumedVersion` only through the delivered prefix, including any empty version gap before that next mutation. If one complete filtered version cannot -fit in the reply budget, the consume fails with `server_overloaded`. +fit in the reply budget, the consume fails with `server_overloaded`. The bound +applies to the combined mutations across every selected range; a version is +never split by range to fit a reply. The owning proxy accepts a cursor only when the position has already been delivered by that owner or is covered by the stream's durable acknowledgement watermark. A fabricated or otherwise unproven cursor is rejected instead of @@ -568,7 +656,11 @@ consumer must resume from its last acknowledged checkpoint. An existing native consumer detects the replacement and automatically rewinds an unacknowledged later position to that durable checkpoint. Only one consume RPC may be active for a stream because all consumers would share the same durable acknowledgement -frontier; overlapping logical consumers are rejected. +frontier; overlapping logical consumers are rejected. Native consumers attach a +stable identity to consume RPCs. A transport retry with that identity cancels +the preceding server request before starting another, so a lost connection +does not turn one consumer into two. The identity is an optional trailing RPC +field; requests from older clients retain the strict overlap rejection. When no later version is available, `consume()` is intentionally a client-side long poll. Each server request has a bounded @@ -610,7 +702,7 @@ metadata scan. ### Removing a stream -Removing a stream eliminates its active name, range, tag history, minimum +Removing a stream eliminates its active name, ranges, tag history, minimum version, and ownership rows. Removal must not unconditionally pop each tag in the removed history: a different live stream may share a tag and still need older data. @@ -702,6 +794,14 @@ cluster role. ## Rollout and migration considerations +Native CDC is unreleased. The multi-range metadata and client interfaces replace +the earlier single-range representation without a compatibility decoder or API +overload. Test deployments using the earlier representation must remove their +streams and finish retired cleanup before upgrading. Upgrade every CDC-capable +server binary and client before registering streams in the new format; mixed +old and new CDC implementations are unsupported, and existing cursors do not +bridge that change. + ### Feature gating `ENABLE_NATIVE_CDC` defaults to false. In simulation it may be randomly enabled @@ -739,6 +839,17 @@ rollback must keep CDC-capable binaries available until those records have been consumed or removed and retired cleanup has completed. Disabling the knob stops new allocation but is not a rollback mechanism for already durable CDC state. +Retag activation requires every process that may serve or recover CDC to +understand commit-stamped history values. The original `withNativeCdc` +capability alone does not establish this. This foundation can read, recover, +and finalize those values without enabling a production writer, so a later +patch can add move policy while preserving rollback to the foundation. +Rolling back to a binary without this foundation requires disabling new moves, +acknowledging or removing streams with pending transitions, and verifying that +all retained history rows have canonical empty values. Keep compatible +replacement binaries available until that state is verified. Disabling a +writer does not itself make older readers safe. + ## Correctness properties The implementation is structured around the following properties: @@ -747,7 +858,7 @@ The implementation is structured around the following properties: reuse of a removed stream name cannot cause an existing consumer to read a new stream. * **Range correctness:** CDC proxies return only mutations within a stream's - registered range, even when its tag is shared with other streams. + registered range union, even when its tag is shared with other streams. * **Acknowledgement monotonicity:** durable minimum required versions advance only forward. * **Shared-tag retention:** tagged data is popped no farther than the minimum @@ -787,10 +898,11 @@ policy simple. size and lifetime limits until these paths are sharded or incrementally maintained. * There is no background process that changes a live stream's CDC tag in - response to load. A future implementation can use versioned tag history to - make such changes without losing the ability to read earlier tagged data. + response to load. Commit-stamped history readers, recovery, and cleanup are + present so a future policy can make such changes without losing earlier + tagged data. * The CDC client surface does not yet provide language-specific bindings beyond - the C API or a higher-level consumer checkpoint abstraction. Administrative + the C and Python APIs or a higher-level consumer checkpoint abstraction. Administrative status and identity-guarded removal are available through `fdbcli`. These improvements must preserve the acknowledgement and retired-pop @@ -834,7 +946,13 @@ The basic native CDC workload covers: * Registering, listing, consuming, acknowledging, and removing streams. * Name-based consumer creation, including end-to-end clear-range clipping. -* Rejection of incompatible same-name registrations. +* Canonical multi-range registration, equivalent same-name registrations, and + rejection of incompatible same-name registrations. +* Gap exclusion for shared-tag mutations, ordered clear splitting across + disjoint intervals, and replay of a complete unacknowledged multi-range + version after proxy replacement using one cursor and acknowledgement. +* Multi-range unread history retained across transaction-system recovery and + new commits routed through the recovered range metadata. * Targeted CDC proxy termination, durable reassignment, and recovery of stream service, including independent publication when two proxies fail together. * Errors for stale consume and acknowledgement requests after removal. @@ -853,7 +971,20 @@ Unit coverage checks the CDC recovery-recruitment truth table with the feature enabled and disabled, both with and without durable CDC state. The process-static knob transition is covered by a paired restart simulation that creates durable CDC work while enabled, restarts with registration disabled, and drains the -existing stream and its retired state. +existing stream and its retired state. A separate paired restart preserves a +pending retag, consumes exact-version mutations from both sides of the cutover, +and verifies acknowledgement-driven finalization with admission disabled. +The same pair can run the writer phase with a newer binary and the drain phase +with the compatibility foundation to test persisted-state rollback. + +`NativeCdcRetagCompatibility` commits synthetic retags around intervening user +writes, verifies replay after proxy replacement and transaction-system recovery, +and checks shared-tag retention and returning to an old tag. It covers the +reader/cleanup contract without a load-driven move policy. +`NativeCdcRetaggingMemoryBound` verifies a pending retag with a 4.5 KiB proxy +budget, verifies both streams deliver before acknowledgement, then acknowledges +and checks history finalization and retired cleanup. Neither fixture qualifies +simultaneous mixed-version processes or throughput under balancing. The shared-tag workload forces streams to share routing tags and verifies both range filtering and acknowledgement coordination. In particular, removing one @@ -863,7 +994,8 @@ mutations needed by the remaining consumer. The simulation configurations enable CDC explicitly when testing these behaviors, while the default-disabled knob and randomized simulation admission exercise the requirement that clusters without active or pending CDC work do -not carry CDC service overhead. +not recruit CDC proxies or retain CDC TLog tags. The data distributor retains +a low-rate metadata-generation check for pending-history finalization. ## Observability and supportability considerations diff --git a/documentation/sphinx/source/api-c.rst b/documentation/sphinx/source/api-c.rst index c568dea5ffe..db35a55756b 100644 --- a/documentation/sphinx/source/api-c.rst +++ b/documentation/sphinx/source/api-c.rst @@ -564,8 +564,8 @@ An |database-blurb1| Modifications to a database are performed via transactions. CDC --- -CDC exposes durable, named streams of committed mutations for one half-open -user-key range. New stream registration requires CDC admission +CDC exposes durable, named streams of committed mutations for an immutable union of +half-open user-key ranges. New stream registration requires CDC admission to be enabled on the cluster. Listing, removal, consumer creation, resume, consume, and acknowledgement remain available for already durable streams while new admission is disabled so that callers can drain or remove them. @@ -586,7 +586,10 @@ select API version 800 or later before calling these functions. .. type:: FDBCdcStreamInfo A listed CDC stream, including its name, stable stream ID, registered - key range, and durable minimum required version. + key ranges, and durable minimum required version. ``ranges`` points to + ``range_count`` sorted, non-empty, disjoint ranges; overlapping and adjacent + registered ranges are merged. The array and its key bytes have the same + lifetime as the stream-info result. .. type:: FDBCdcMutation @@ -605,11 +608,21 @@ select API version 800 or later before calling these functions. remains valid after the originating future is destroyed. Destroy it exactly once with :func:`fdb_cdc_consumer_destroy()`. -.. function:: FDBFuture* fdb_database_register_cdc_stream(FDBDatabase* database, uint8_t const* name, int name_length, uint8_t const* begin_key, int begin_key_length, uint8_t const* end_key, int end_key_length) - - Registers ``name`` for the non-empty half-open range ``[begin_key, - end_key)`` in normal user key space. Repeating the same name and range is - idempotent; reusing a name with a different range fails. The future returns +.. function:: FDBFuture* fdb_database_register_cdc_stream(FDBDatabase* database, uint8_t const* name, int name_length, FDBKeyRange const* ranges, int range_count) + + Registers ``name`` for the union of ``range_count`` non-empty half-open + ranges in normal user key space. The input must contain between 1 and 1024 + ranges. Ranges may arrive in any order; overlapping, duplicate, and adjacent + ranges are merged. The encoded canonical range set must fit within the + database value-size limit. Mutations in gaps between the resulting ranges are + excluded, and clear-range mutations are clipped to each intersecting range. + + A stream's range set is immutable. Repeating the same name and equivalent + range union is idempotent; reusing a name with a different union fails. + All ranges share one consumer cursor and acknowledgement position. + The range array, key bytes, and name bytes are copied before this function + returns. Invalid counts, negative lengths, null required pointers, or invalid + ranges return a future with ``client_invalid_operation``. The future returns the ``uint64_t`` stream ID, extracted with :func:`fdb_future_get_uint64()`. .. function:: FDBFuture* fdb_database_remove_cdc_stream(FDBDatabase* database, uint8_t const* name, int name_length) diff --git a/documentation/sphinx/source/api-python.rst b/documentation/sphinx/source/api-python.rst index 45802eb3825..3c8f8a6bcc7 100644 --- a/documentation/sphinx/source/api-python.rst +++ b/documentation/sphinx/source/api-python.rst @@ -455,7 +455,249 @@ Database options .. method:: Database.options.set_snapshot_ryw_disable() |option-db-snapshot-ryw-disable-blurb| - + +.. _api-python-cdc: + +Native change data capture (CDC) +================================ + +CDC provides durable, named streams of committed mutations for a non-empty +union of half-open user-key ranges. Select API version 800 or later with +:func:`api_version` before using this interface. New stream registration also +requires the cluster's ``ENABLE_NATIVE_CDC`` admission knob. Existing streams +can still be listed or removed while admission is disabled; consumer creation, +resume, consumption, and acknowledgement also remain available. Repeating an +existing same-name, same-range-union registration remains idempotent. + +Native CDC is experimental. These bindings require the multi-range CDC C +library and server implementation; older single-range CDC binaries use an +incompatible interface even if they support API version 800. Drain and remove +existing streams before upgrading all CDC-capable clients and servers together. + +Unlike the synchronous database key-value methods, the CDC database methods +and the consumer's ``consume()`` and ``acknowledge()`` methods return +:ref:`futures `. Use their usual ``wait()``, ``on_ready()``, +or ``result()`` interfaces to observe completion and errors. These operations +are not methods on :class:`Transaction`, and cannot be made atomic with +application writes by using :func:`transactional`. + +.. warning:: + + A stream has one shared acknowledgement frontier, not one per handle. + Use only one active logical consumer per stream. Each handle permits only + one outstanding ``consume()`` or ``acknowledge()`` operation at a time. + Acknowledge only after every mutation through the delivered position and + the corresponding application checkpoint are durable. Neither consuming + nor closing a handle acknowledges automatically. An abandoned stream can + retain unread log history indefinitely until acknowledged or removed. + +Stream management +----------------- + +.. method:: Database.register_cdc_stream(name, begin_key=None, end_key=None, *, ranges=None) + + Registers the byte-string ``name`` for a range union in normal user key + space. Supply either ``begin_key`` and ``end_key`` for one half-open range, + or the keyword argument ``ranges`` with an iterable of + ``(begin_key, end_key)`` pairs (including :class:`CdcKeyRange` records). + Combining the two forms raises ``TypeError``. The name, range collection, + and each range must be non-empty. Supply at most 1024 ranges before + canonicalization; the encoded union must fit within the native metadata + value-size limit. + + Registration sorts the ranges and merges overlaps, duplicates, and adjacent + intervals into a canonical union. Repeating the same name and canonical + union is idempotent, regardless of input order or partitioning; reusing a + name with a different union fails. All ranges share one stream ID, cursor, + and acknowledgement frontier. Mutations in gaps are excluded, and a clear + spanning multiple ranges produces one clipped clear per intersected range. + Returns a future whose value is the unsigned 64-bit stream ID as a Python + ``int``. Register long-lived streams rather than a stream per request. + + For example:: + + stream_id = db.register_cdc_stream( + b"commerce", ranges=[(b"order/", b"order0"), (b"payment/", b"payment0")] + ).wait() + +.. method:: Database.remove_cdc_stream(name) + + Removes the byte-string stream name and relinquishes its unread history. + Removing a missing name succeeds. Returns a future whose value is + ``None``. Removal is terminal for existing consumers; registering the + same name again does not redirect their cursors to the new stream. + +.. method:: Database.list_cdc_streams() + + Returns a future whose value is a list of :class:`CdcStreamInfo` records. + +.. method:: Database.create_cdc_consumer(name) + + Returns a future whose value is a :class:`CdcConsumer` for the existing + byte-string stream name at its initial position. + +.. method:: Database.resume_cdc_consumer(cursor) + + Returns a future whose value is a :class:`CdcConsumer` constructed from + a checkpointed :class:`CdcCursor`. The cursor contains no process-local + state. Stream existence and cursor validity are checked when consuming + or acknowledging, not by this method. Resume only from a durably processed + checkpoint. Before the first consume, wait until a fresh database read + version reaches ``cursor.last_consumed_version``, then reissue + :meth:`CdcConsumer.acknowledge` and wait for it to complete to reconcile a + possibly interrupted acknowledgement. A resumed handle lacks the original + handle's delivery proof, so an unacknowledged cursor ahead of that read + version is rejected with ``client_invalid_operation`` even if it was + previously delivered. Bound the read-version wait rather than retrying all + invalid-operation errors. Unacknowledged mutations may be redelivered after + CDC proxy replacement, so processing must tolerate replay. + +Result types +------------ + +The following records are immutable tuples with named fields. Returned names, +keys, and mutation parameters are Python-owned ``bytes``, copied from the +native result. They remain valid after the future or consumer is released. + +.. class:: CdcCursor(stream_id, last_consumed_version) + + A stable unsigned 64-bit stream ID and the signed 64-bit version through + which mutations have been delivered. Both fields are Python ``int`` + values. A delivered cursor is not proof of processing or acknowledgement. + +.. class:: CdcKeyRange(begin_key, end_key) + + The inclusive begin and exclusive end byte strings of one registered range. + +.. class:: CdcStreamInfo(name, stream_id, ranges, min_version) + + A stream's byte-string name, integer ID, canonical range union as a tuple + of :class:`CdcKeyRange` records, and durable minimum required version. + ``min_version`` is a retention frontier, not a snapshot version for the + registered ranges. The ``begin_key`` and ``end_key`` properties are + available when the canonical union contains exactly one range; accessing + them on a multi-range stream raises ``ValueError``. Use ``ranges`` to + inspect any stream without treating gaps as registered keys. + +.. class:: CdcMutation(type, param1, param2) + + One raw mutation. ``type`` is an integer, including for unrecognized + mutation types; ``param1`` and ``param2`` are byte strings. For + ``SET_VALUE`` they are the key and value; for ``CLEAR_RANGE`` they are + the begin and end keys, clipped to a registered range; for atomic + mutations they are the key and operand. CDC returns raw mutation + operations, not a materialized post-mutation value for every key. + +.. class:: CdcMutationType + + An ``enum.IntEnum`` of known raw mutation type values, matching the C API + constants without the ``FDB_CDC_MUTATION_TYPE_`` prefix: ``SET_VALUE``, + ``CLEAR_RANGE``, ``ADD``, ``AND``, ``OR``, ``XOR``, ``APPEND_IF_FITS``, + ``MAX``, ``MIN``, ``SET_VERSIONSTAMPED_KEY``, ``SET_VERSIONSTAMPED_VALUE``, + ``BYTE_MIN``, ``BYTE_MAX``, ``MIN_V2``, ``AND_V2``, and ``COMPARE_AND_CLEAR``. + These constants are not exhaustive. Compare ``mutation.type`` to known + constants, but handle unknown integers without assuming that constructing + ``CdcMutationType(mutation.type)`` will succeed. + +.. class:: CdcVersionedMutations(version, mutations) + + One complete commit-version group, with an integer ``version`` and a + tuple of :class:`CdcMutation` records. Preserve this grouping when + processing a reply. + +.. class:: CdcConsumeResult(mutations, last_consumed_version) + + ``mutations`` is a tuple of :class:`CdcVersionedMutations` groups; + ``last_consumed_version`` is the integer delivered cursor after the reply. + The cursor may advance across commit-version gaps with no returned + mutations. Even an empty reply can advance the cursor; do not infer it + from the last mutation group or skip checkpointing and acknowledgement + solely because ``mutations`` is empty. + +Consumer lifecycle +------------------ + +.. class:: CdcConsumer + + An owned native consumer handle, obtained from a database create or resume + future. It remains valid independently of that future. Use it as a + context manager or call :meth:`CdcConsumer.close` when finished. + +.. method:: CdcConsumer.consume() + + Long-polls for the next delivered position and complete commit-version + groups. Returns a future whose value is :class:`CdcConsumeResult`. + Successful consumption advances the in-memory position, but does not + release durable retention. Finish processing a reply before consuming + again if that reply may need to be retried. Following proxy replacement, + the client may rewind to its last successful durable acknowledgement and + redeliver unacknowledged mutations. + +.. method:: CdcConsumer.acknowledge() + + Durably acknowledges the handle's current delivered position, allowing + history through that position to be released. Returns a future whose + value is ``None``. Wait for it before starting another consume or + acknowledgement. This is not atomic with writes to a downstream system + or an application checkpoint. + +.. method:: CdcConsumer.get_position() + + Returns the current :class:`CdcCursor` synchronously. + +.. method:: CdcConsumer.close() + + Releases the local handle. Repeated calls are harmless. It does not + acknowledge the delivered position or remove the stream. Exiting the + consumer's context manager calls this method, including on exceptions. + +For example, the following loop supports both an initial start and a restart. +The application-supplied ``load_checkpoint`` returns a durable +:class:`CdcCursor`, or ``None`` on the first start. ``apply_and_checkpoint`` +must durably apply complete version groups and record the cursor consistently, +and must tolerate replay without repeating non-idempotent side effects. It +must also handle an empty reply. If it raises, the loop does not acknowledge +the reply:: + + import time + + import fdb + + fdb.api_version(800) + db = fdb.open() + db.register_cdc_stream(b"orders", b"order/", b"order0").wait() + + saved_cursor = load_checkpoint() + if saved_cursor is None: + consumer_future = db.create_cdc_consumer(b"orders") + else: + consumer_future = db.resume_cdc_consumer(saved_cursor) + + with consumer_future.wait() as consumer: + if saved_cursor is not None: + deadline = time.monotonic() + 30 + while True: + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError("CDC checkpoint is ahead of the read version") + tr = db.create_transaction() + tr.options.set_read_lock_aware() + tr.options.set_timeout(max(1, int(remaining * 1000))) + if tr.get_read_version().wait() >= saved_cursor.last_consumed_version: + break + time.sleep(min(0.01, max(0, deadline - time.monotonic()))) + consumer.acknowledge().wait() + while True: + reply = consumer.consume().wait() + apply_and_checkpoint(reply.mutations, consumer.get_position()) + consumer.acknowledge().wait() + +The acknowledgement after the read-version wait closes the crash gap between +persisting the checkpoint and completing its acknowledgement; reissuing an +already durable acknowledgement is safe. This requires a cursor whose +mutations are known to have been durably processed. Never invent a cursor or +advance a checkpoint or acknowledgement beyond that position. + Transactional decoration ======================== diff --git a/documentation/sphinx/source/clang-tidy.rst b/documentation/sphinx/source/clang-tidy.rst index c710e440e49..92ab54a26fa 100644 --- a/documentation/sphinx/source/clang-tidy.rst +++ b/documentation/sphinx/source/clang-tidy.rst @@ -10,11 +10,11 @@ This guide explains how to run ``clang-tidy`` locally so you can fix issues befo What clang-tidy checks ====================== -FoundationDB configures 50 named checks in the ``.clang-tidy`` file at the repository root. The +FoundationDB configures 55 named checks in the ``.clang-tidy`` file at the repository root. The active set depends on the clang-tidy version and can be inspected with ``clang-tidy --list-checks``. The intent is to enable more as we go forward. Here are some example rules: -* **31 Bugprone rules** -- catch potential runtime errors, including mismatched argument comments, obvious infinite loops, chained comparisons, swapped arguments, missed base-class copy construction, repeated macro argument evaluation, near-miss virtual overrides, dangling returned references, and incorrect erase/remove calls +* **36 Bugprone rules** -- catch potential runtime errors, including unsafe self-assignment, forwarding constructors that hide copy or move constructors, narrow accumulation initializers, mismatched argument comments, obvious infinite loops, chained comparisons, swapped arguments, integer division in floating-point calculations, missed base-class copy construction, repeated macro argument evaluation, near-miss virtual overrides, dangling returned references, incorrect erase/remove calls, and incorrect POSIX error checks * **1 C++ Core Guidelines rule** -- catch unsafe captures in coroutine lambdas (``cppcoreguidelines-avoid-capturing-lambda-coroutines``) * **2 Misc rules** -- catch redundant expressions and RAII objects held across coroutine suspension points * **4 Modernize rules** -- encourage modern C++ practices (e.g., ``modernize-use-auto``, ``modernize-use-override``) @@ -26,6 +26,11 @@ The intent is to enable more as we go forward. Here are some example rules: ``std::scoped_lock``. These guards must leave scope before a coroutine suspension point; asynchronous locks designed to span suspension are not included. +``bugprone-unhandled-self-assignment`` retains its default restriction to types +with suspicious fields. Reference-counted assignments that acquire the incoming +reference before releasing the old one use documented, check-specific +``NOLINTNEXTLINE`` annotations where the checker cannot recognize their safety. + Basic examples of ``clang-tidy`` style and performance improvement changes: .. code-block:: cpp diff --git a/documentation/sphinx/source/getting-started-linux.rst b/documentation/sphinx/source/getting-started-linux.rst index 25ee0628e74..01add5f1300 100644 --- a/documentation/sphinx/source/getting-started-linux.rst +++ b/documentation/sphinx/source/getting-started-linux.rst @@ -48,6 +48,31 @@ To install on **RHEL/CentOS 7** use the rpm command: |simple-installation-mode-warnings| |networking-clarification| + +Installing multiple client versions +==================================== + +The regular ``foundationdb-clients`` RPM installs files in shared system paths +and is intended for a normal install or upgrade. Do not install two regular +client RPMs with ``rpm -ivh``; they own the same files and RPM will report file +conflicts. + +For multi-version client support, use the separately named CPack versioned +client RPMs. Their names have the form +``foundationdb-clients-1.versioned..rpm`` and their +files are installed below ``/usr/lib/foundationdb-/``. Non-release +builds may include build-time and prerelease text in the package name and +directory. Install each versioned package with ``rpm -ivh`` so the packages +can remain installed simultaneously. If a release does not publish its +versioned RPM, build the +``clients-versioned`` component from source with CPack (run ``cpack -G RPM`` +from the configured build directory). + +The active client is selected through the ``fdbclients`` alternatives group. +Use ``update-alternatives --display fdbclients`` to inspect it and +``update-alternatives --config fdbclients`` to select a version. The selected +alternative applies consistently to the client tools, library, headers, +pkg-config metadata, and CMake package metadata. Testing your FoundationDB installation ====================================== diff --git a/fdbbackup/FileDecoder.cpp b/fdbbackup/FileDecoder.cpp index 8e3a3cab57e..0fa7f94cf20 100644 --- a/fdbbackup/FileDecoder.cpp +++ b/fdbbackup/FileDecoder.cpp @@ -545,10 +545,11 @@ class DecodeProgress { // version batch data that are in the next file. Optional getNextBatch() { for (auto& [version, m] : mutationBlocksByVersion) { - if (m.isComplete()) { + Optional completeMutations = m.getCompleteMutations(); + if (completeMutations.present()) { VersionedMutations vms; vms.version = version; - vms.serializedMutations = m.serializedMutations; + vms.serializedMutations = completeMutations.get().toString(); vms.mutations = fileBackup::decodeMutationLogValue(vms.serializedMutations); TraceEvent("Decode").detail("Version", vms.version).detail("N", vms.mutations.size()); mutationBlocksByVersion.erase(version); diff --git a/fdbcli/BulkDumpCommand.cpp b/fdbcli/BulkDumpCommand.cpp index 89c362b18fa..724683b5218 100644 --- a/fdbcli/BulkDumpCommand.cpp +++ b/fdbcli/BulkDumpCommand.cpp @@ -41,37 +41,6 @@ static const std::string BULK_DUMP_HELP_MESSAGE = std::string(BULK_DUMP_MODE_USAGE) + std::string(BULK_DUMP_DUMP_USAGE) + std::string(BULK_DUMP_STATUS_USAGE) + std::string(BULK_DUMP_CANCEL_USAGE); -Future getOngoingBulkDumpJob(Database cx) { - Transaction tr(cx); - while (true) { - Error err; - try { - Optional job = co_await getSubmittedBulkDumpJob(&tr); - if (job.present()) { - fmt::println("Running bulk dumping job: {}", job.get().getJobId().toString()); - co_return true; - } else { - fmt::println("No bulk dumping job is running"); - co_return false; - } - } catch (Error& e) { - err = e; - } - co_await tr.onError(err); - } -} - -Future getBulkDumpCompleteRanges(Database cx, KeyRange rangeToRead) { - try { - size_t finishCount = co_await getBulkDumpCompleteTaskCount(cx, rangeToRead); - fmt::println("Finished {} tasks", finishCount); - } catch (Error& e) { - if (e.code() == error_code_timed_out) { - fmt::println("timed out"); - } - } -} - Future bulkDumpCommandActor(Database cx, std::vector tokens) { BulkDumpState bulkDumpJob; if (tokencmp(tokens[1], "mode")) { diff --git a/fdbcli/BulkLoadCommand.cpp b/fdbcli/BulkLoadCommand.cpp index 55660c7643c..70a54efd2ea 100644 --- a/fdbcli/BulkLoadCommand.cpp +++ b/fdbcli/BulkLoadCommand.cpp @@ -86,70 +86,6 @@ Future printPastBulkLoadJob(Database cx) { } } -void printBulkLoadJobTotalTaskCount(Optional count) { - if (count.present()) { - fmt::println("Total {} tasks", count.get()); - } else { - fmt::println("Total task count is unknown"); - } - return; -} - -Future printBulkLoadJobProgress(Database cx, BulkLoadJobState job) { - Transaction tr(cx); - Key readBegin = job.getJobRange().begin; - Key readEnd = job.getJobRange().end; - UID jobId = job.getJobId(); - RangeResult rangeResult; - size_t completeTaskCount = 0; - size_t submitTaskCount = 0; - size_t errorTaskCount = 0; - Optional totalTaskCount = job.getTaskCount(); - while (readBegin < readEnd) { - Error err; - bool hasErr = false; - try { - rangeResult.clear(); - tr.setOption(FDBTransactionOptions::READ_SYSTEM_KEYS); - tr.setOption(FDBTransactionOptions::LOCK_AWARE); - rangeResult = co_await krmGetRanges(&tr, bulkLoadTaskPrefix, KeyRangeRef(readBegin, readEnd)); - for (int i = 0; i < rangeResult.size() - 1; ++i) { - if (rangeResult[i].value.empty()) { - continue; - } - BulkLoadTaskState bulkLoadTask = decodeBulkLoadTaskState(rangeResult[i].value); - if (bulkLoadTask.getJobId() != jobId) { - fmt::println("Submitted {} tasks", submitTaskCount); - fmt::println("Finished {} tasks", completeTaskCount); - fmt::println("Error {} tasks", errorTaskCount); - printBulkLoadJobTotalTaskCount(totalTaskCount); - if (bulkLoadTask.phase == BulkLoadPhase::Submitted && bulkLoadTask.getJobId().isValid()) { - fmt::println("Job {} has been cancelled or has completed", jobId.toString()); - } - co_return; - } - if (bulkLoadTask.phase == BulkLoadPhase::Complete) { - completeTaskCount = completeTaskCount + bulkLoadTask.getManifests().size(); - } else if (bulkLoadTask.phase == BulkLoadPhase::Error) { - errorTaskCount = errorTaskCount + bulkLoadTask.getManifests().size(); - } - submitTaskCount = submitTaskCount + bulkLoadTask.getManifests().size(); - } - readBegin = rangeResult.back().key; - } catch (Error& e) { - err = e; - hasErr = true; - } - if (hasErr) { - co_await tr.onError(err); - } - } - fmt::println("Submitted {} tasks", submitTaskCount); - fmt::println("Finished {} tasks", completeTaskCount); - fmt::println("Error {} tasks", errorTaskCount); - printBulkLoadJobTotalTaskCount(totalTaskCount); -} - Future bulkLoadCommandActor(Database cx, std::vector tokens) { if (tokencmp(tokens[1], "mode")) { if (tokens.size() == 2) { diff --git a/fdbcli/CdcCommand.cpp b/fdbcli/CdcCommand.cpp index ebf19b86e93..07b47f7336e 100644 --- a/fdbcli/CdcCommand.cpp +++ b/fdbcli/CdcCommand.cpp @@ -62,8 +62,14 @@ json_spirit::mObject statusJson(const NativeCdcStatus& status) { json_spirit::mObject entry; entry["stream_id"] = std::to_string(stream.info.streamId); entry["name"] = stream.info.name.printable(); - entry["range_begin"] = stream.info.keys.begin.printable(); - entry["range_end"] = stream.info.keys.end.printable(); + json_spirit::mArray ranges; + for (const auto& range : stream.info.ranges) { + json_spirit::mObject keys; + keys["range_begin"] = range.begin.printable(); + keys["range_end"] = range.end.printable(); + ranges.push_back(keys); + } + entry["ranges"] = ranges; entry["min_version"] = versionJson(stream.info.minVersion); entry["acknowledgement_lag_versions"] = versionJson(acknowledgementLag(status, stream)); entry["owner_proxy_id"] = @@ -160,10 +166,12 @@ void printCdcStatus(const NativeCdcStatus& status) { fmt::println(" Retention metadata is drained. Physical TLog disk reclamation is not certified."); } for (const auto& stream : status.streams) { - fmt::println(" Stream {}: name=\"{}\", range={}", - stream.info.streamId, - stream.info.name.printable(), - stream.info.keys.toString()); + fmt::println(" Stream {}: name=\"{}\"", stream.info.streamId, stream.info.name.printable()); + fmt::print(" ranges:"); + for (const auto& range : stream.info.ranges) { + fmt::print(" {}", range.toString()); + } + fmt::println(""); fmt::println(" minimum version={}, acknowledgement lag={} versions, owner={} ({})", versionText(stream.info.minVersion), versionText(acknowledgementLag(status, stream)), diff --git a/fdbclient/ActorLineageProfiler.cpp b/fdbclient/ActorLineageProfiler.cpp index 22c13ec1bf2..88e19f14f61 100644 --- a/fdbclient/ActorLineageProfiler.cpp +++ b/fdbclient/ActorLineageProfiler.cpp @@ -74,7 +74,7 @@ class Packer : public msgpack::packer { void visit(const std::any& val, Packer& packer) { auto iter = visitorMap.find(val.type()); if (iter == visitorMap.end()) { - TraceEvent(SevError, "PackerTypeNotFound").detail("Type", val.type().name()); + TraceEvent(SevError, "PackerTypeNotFound").detail("ValueType", val.type().name()); } else { iter->second(val, packer); } diff --git a/fdbclient/AsyncFileBlobStore.cpp b/fdbclient/AsyncFileBlobStore.cpp index d2705b7e2b3..e1a1084d1b2 100644 --- a/fdbclient/AsyncFileBlobStore.cpp +++ b/fdbclient/AsyncFileBlobStore.cpp @@ -74,5 +74,6 @@ TEST_CASE("/backup/throttling") { double dur = timer() - ts; int speed = int(total / dur); printf("Speed limit was %d, measured speed was %d\n", limit, speed); - ASSERT(abs(speed - limit) / limit < .01); + // Host scheduling can delay completions; only exceeding the rate limit is an error. + ASSERT(speed <= limit * 1.01); } diff --git a/fdbclient/BackupContainer.cpp b/fdbclient/BackupContainer.cpp index 5f58605e418..43866175239 100644 --- a/fdbclient/BackupContainer.cpp +++ b/fdbclient/BackupContainer.cpp @@ -33,6 +33,7 @@ #include "fdbclient/RunRYWTransaction.h" #include #include +#include namespace IBackupFile_impl { @@ -266,7 +267,14 @@ Reference IBackupContainer::openContainer(const std::string& u const Optional& proxy, const Optional& encryptionKeyFileName, int encryptionBlockSize) { - static std::map> m_cache; + using CacheKey = std::tuple, Optional, int>; + static std::map> m_cache; + + Optional blobstoreProxy; + if (isBlobstoreUrl(url)) { + // The backup-agent fallback is part of the effective connection configuration. + blobstoreProxy = proxy.present() ? proxy : fileBackupAgentProxy; + } // In simulation, disable caching for blobstore:// URLs to prevent cross-process connection issues. // @@ -286,7 +294,8 @@ Reference IBackupContainer::openContainer(const std::string& u // Use a reference to the cache entry (for automatic cache population) unless we're skipping cache Reference r_local; - Reference& r = skipCache ? r_local : m_cache[url]; + Reference& r = + skipCache ? r_local : m_cache[{ url, blobstoreProxy, encryptionKeyFileName, encryptionBlockSize }]; if (r) { return r; } @@ -297,15 +306,6 @@ Reference IBackupContainer::openContainer(const std::string& u r = makeReference(url, encryptionKeyFileName, encryptionBlockSize); } else if (u.startsWith("blobstore://"_sr)) { std::string resource; - Optional blobstoreProxy; - - // If no proxy is passed down to the openContainer method, try to fallback to the - // fileBackupAgentProxy which is a global variable and will be set for the backup_agent. - if (proxy.present()) { - blobstoreProxy = proxy.get(); - } else if (fileBackupAgentProxy.present()) { - blobstoreProxy = fileBackupAgentProxy.get(); - } // The URL parameters contain blobstore endpoint tunables as well as possible backup-specific options. IBlobStoreEndpoint::ParametersT backupParams; diff --git a/fdbclient/BackupContainerBlobStore.cpp b/fdbclient/BackupContainerBlobStore.cpp index e76fb7881e8..9dc750dbf3d 100644 --- a/fdbclient/BackupContainerBlobStore.cpp +++ b/fdbclient/BackupContainerBlobStore.cpp @@ -24,6 +24,7 @@ #include "fdbrpc/AsyncFileEncrypted.h" #include "fdbrpc/AsyncFileReadAhead.h" #include "fdbrpc/HTTP.h" +#include "flow/Platform.h" #include "flow/UnitTest.h" class BackupContainerBlobStoreImpl { @@ -227,7 +228,10 @@ Future> BackupContainerBlobStore::readFile(const std::stri m_bstore->knobs.read_cache_blocks_per_file); } if (usesEncryption() && !StringRef(path).startsWith("properties/"_sr)) { - f = makeReference(f, AsyncFileEncrypted::Mode::READ_ONLY, encryptionBlockSize); + // Agents can open an existing container without calling create(), so key loading may still be in flight. + return map(encryptionSetupComplete(), [f, blockSize = encryptionBlockSize](Void) -> Reference { + return makeReference(f, AsyncFileEncrypted::Mode::READ_ONLY, blockSize); + }); } return f; } @@ -239,11 +243,16 @@ Future> BackupContainerBlobStore::listURLs(Reference> BackupContainerBlobStore::writeFile(const std::string& path) { - Reference f = makeReference(m_bstore, m_bucket, dataPath(path)); + Reference rawFile = makeReference(m_bstore, m_bucket, dataPath(path)); + Future> f = rawFile; if (usesEncryption() && !StringRef(path).startsWith("properties/"_sr)) { - f = makeReference(f, AsyncFileEncrypted::Mode::APPEND_ONLY, encryptionBlockSize); + f = map(encryptionSetupComplete(), [rawFile, blockSize = encryptionBlockSize](Void) -> Reference { + return makeReference(rawFile, AsyncFileEncrypted::Mode::APPEND_ONLY, blockSize); + }); } - return Future>(makeReference(path, f)); + return map(f, [path](Reference file) -> Reference { + return makeReference(path, file); + }); } Future BackupContainerBlobStore::writeEntireFile(const std::string& path, const std::string& fileContents) { @@ -281,6 +290,27 @@ std::string BackupContainerBlobStore::getPrefix() const { return m_prefix; } +TEST_CASE("/backup/containers/blobstore/encryptionSetup") { + std::string keyFile = joinPath(params.getDataDir(), "encryption-key"); + co_await BackupContainerFileSystem::createTestEncryptionKeyFile(keyFile); + std::string resource; + IBlobStoreEndpoint::ParametersT backupParams; + auto endpoint = IBlobStoreEndpoint::fromString( + "blobstore://localhost:9999/encryption-setup?bucket=test", {}, &resource, nullptr, &backupParams); + auto container = makeReference(endpoint, resource, backupParams, keyFile, 4096, true); + + // An agent reopening a container does not call create(). No blob requests are needed to open these handles. + auto read = container->readFile("range"); + auto write = container->writeFile("range"); + if (!container->encryptionSetupComplete().isReady()) { + ASSERT(!read.isReady()); + ASSERT(!write.isReady()); + } + co_await (success(read) && success(write)); + ASSERT(container->encryptionSetupComplete().isReady()); + ASSERT(!container->encryptionSetupComplete().isError()); +} + TEST_CASE("/backup/containers/blobstore/prefix") { // Normalization: leading/trailing slashes are stripped, empty selects the default layout. ASSERT(BackupContainerBlobStore::normalizePrefix("").empty()); diff --git a/fdbclient/BackupContainerLocalDirectory.cpp b/fdbclient/BackupContainerLocalDirectory.cpp index 4d3cc97f2ad..78f91eb9acf 100644 --- a/fdbclient/BackupContainerLocalDirectory.cpp +++ b/fdbclient/BackupContainerLocalDirectory.cpp @@ -260,9 +260,9 @@ Future> BackupContainerLocalDirectory::readFile(const std: // Skip encryption for properties/ folder if (usesEncryption() && !StringRef(path).startsWith("properties/"_sr)) { int encBlockSize = encryptionBlockSize; - f = map(f, [encBlockSize](Reference r) { + f = map(success(f) && encryptionSetupComplete(), [file = f, encBlockSize](Void) { return Reference( - makeReference(r, AsyncFileEncrypted::Mode::READ_ONLY, encBlockSize)); + makeReference(file.get(), AsyncFileEncrypted::Mode::READ_ONLY, encBlockSize)); }); } @@ -303,9 +303,9 @@ Future> BackupContainerLocalDirectory::writeFile(const st // Skip encryption for properties/ folder if (usesEncryption() && !StringRef(path).startsWith("properties/"_sr)) { int encBlockSize = encryptionBlockSize; - f = map(f, [encBlockSize](Reference r) { + f = map(success(f) && encryptionSetupComplete(), [file = f, encBlockSize](Void) { return Reference( - makeReference(r, AsyncFileEncrypted::Mode::APPEND_ONLY, encBlockSize)); + makeReference(file.get(), AsyncFileEncrypted::Mode::APPEND_ONLY, encBlockSize)); }); } return map(f, [=](Reference file) -> Reference { diff --git a/fdbclient/ClientKnobs.cpp b/fdbclient/ClientKnobs.cpp index c2ccfb7e807..214a88247b1 100644 --- a/fdbclient/ClientKnobs.cpp +++ b/fdbclient/ClientKnobs.cpp @@ -159,6 +159,8 @@ void ClientKnobs::initialize(Randomize randomize, IsSimulated isSimulated) { init( MAX_CLIENT_STATUS_AGE, 1.0 ); init( MAX_COMMIT_PROXY_CONNECTIONS, 5 ); if( randomize && buggify() ) MAX_COMMIT_PROXY_CONNECTIONS = 1; init( MAX_GRV_PROXY_CONNECTIONS, 3 ); if( randomize && buggify() ) MAX_GRV_PROXY_CONNECTIONS = 1; + init( SHRINK_PROXY_LIST_CLEAR_CACHE_BELOW_THRESHOLD, true ); if( randomize && isSimulated ) SHRINK_PROXY_LIST_CLEAR_CACHE_BELOW_THRESHOLD = deterministicRandom()->coinflip(); + init( DBCONTEXT_EAGER_PROXY_UPDATE, true ); if( randomize && isSimulated ) DBCONTEXT_EAGER_PROXY_UPDATE = deterministicRandom()->coinflip(); init( STATUS_IDLE_TIMEOUT, 120.0 ); init( STATUS_TIMEOUT, 30.0 ); init( GRPC_CTL_SERVICE_DEFAULT_TIMEOUT, 5.0 ); @@ -206,6 +208,10 @@ void ClientKnobs::initialize(Randomize randomize, IsSimulated isSimulated) { init( LOCATION_CACHE_EVICTION_SIZE_SIM, 10 ); if( randomize && buggify() ) LOCATION_CACHE_EVICTION_SIZE_SIM = 3; init( LOCATION_CACHE_ENDPOINT_FAILURE_GRACE_PERIOD, 60 ); init( LOCATION_CACHE_FAILED_ENDPOINT_RETRY_INTERVAL, 60 ); + init( LOCATION_CACHE_PEER_EVICTOR_ENABLED, true ); if( randomize && isSimulated ) LOCATION_CACHE_PEER_EVICTOR_ENABLED = deterministicRandom()->coinflip(); + init( LOCATION_CACHE_PEER_EVICTOR_DELAY, 60.0 ); + init( LOCATION_CACHE_PEER_EVICTOR_FAILED_THRESHOLD, 0 ); + init( LOCATION_CACHE_PEER_EVICTOR_SCAN_CHUNK, 1000000 ); if( randomize && buggify() ) LOCATION_CACHE_PEER_EVICTOR_SCAN_CHUNK = deterministicRandom()->randomInt(1, 11); init( GET_RANGE_SHARD_LIMIT, 2 ); init( WARM_RANGE_SHARD_LIMIT, 100 ); @@ -264,7 +270,7 @@ void ClientKnobs::initialize(Randomize randomize, IsSimulated isSimulated) { init( BACKUP_SIMULATED_LIMIT_BYTES, 1e6 ); if( randomize && buggify() ) BACKUP_SIMULATED_LIMIT_BYTES = 1000; init( BACKUP_GET_RANGE_LIMIT_BYTES, 1e6 ); init( BACKUP_LOCK_BYTES, 1e8 ); - init( BACKUP_RANGE_TIMEOUT, TASKBUCKET_TIMEOUT_VERSIONS/CORE_VERSIONSPERSECOND/2.0 ); + init( BACKUP_RANGE_TIMEOUT, static_cast(TASKBUCKET_TIMEOUT_VERSIONS)/CORE_VERSIONSPERSECOND/2.0 ); init( BACKUP_RANGE_MINWAIT, std::max(1.0, BACKUP_RANGE_TIMEOUT/2.0)); init( BULKDUMP_JOB_TIMEOUT, 3600 * 24 ); // 24 hours - large DBs may take days init( BULKLOAD_JOB_TIMEOUT, 3600 * 24 ); // 24 hours - large DBs may take days diff --git a/fdbclient/DatabaseContext.cpp b/fdbclient/DatabaseContext.cpp index 2d872de1a6d..134341154cd 100644 --- a/fdbclient/DatabaseContext.cpp +++ b/fdbclient/DatabaseContext.cpp @@ -26,6 +26,8 @@ #include #include #include +#include +#include #include #include "fdbclient/Knobs.h" @@ -34,6 +36,7 @@ #include "fdbclient/FDBOptions.g.h" #include "fdbclient/FDBTypes.h" #include "fdbrpc/MultiInterface.h" +#include "fdbrpc/FlowTransport.h" #include "fdbclient/ClusterInterface.h" #include "fdbclient/CoordinationInterface.h" @@ -919,6 +922,13 @@ static Future monitorClientDBInfoChange(DatabaseContext* cx, // Clear the version vector to ensure the latest commit versions are received. cx->ssVersionVectorCache.clear(); proxiesChangeTrigger->trigger(); + // Eagerly rebuild the published proxy ModelInterface the instant the + // proxy list changes, so a killed proxy's RequestStream is dropped from + // cx->commitProxies/grvProxies on clientInfo rotation rather than waiting + // for the next transaction's lazy getCommitProxies()/getGrvProxies(). + if (CLIENT_KNOBS->DBCONTEXT_EAGER_PROXY_UPDATE) { + cx->updateProxies(); + } } } else if (res.index() == 1) { UNSTOPPABLE_ASSERT(false); @@ -1125,6 +1135,200 @@ void DatabaseContext::initializeSpecialCounters() { specialCounter(cc, "WatchMapSize", [this] { return watchMap.size(); }); } +// Evicts cached ranges mapping to any server address in the input addresses set +// Yields every LOCATION_CACHE_PEER_EVICTOR_SCAN_CHUNK ranges +static Future invalidateCacheByAddresses(DatabaseContext* self, std::unordered_set addresses) { + // Initial checks + if (addresses.empty()) { + co_return; + } + int rangeChunkThreshold = CLIENT_KNOBS->LOCATION_CACHE_PEER_EVICTOR_SCAN_CHUNK; + ASSERT(rangeChunkThreshold >= 1); + + // State across phase 1 and phase 2 below + std::vector rangesToInvalidate; + double startT = now(); + + // Phase 1: scan the cache in chunks, and compute invalid ranges + TraceEvent("LocationCacheInvalidatedByAddresses_Phase1_Begin") + .detail("DbId", self->dbId) + .detail("AddressCount", addresses.size()); + Key cursor = allKeys.begin; + int phase1RangesScanned = 0; + int phase1Yields = 0; + + for (;;) { + TraceEvent("LocationCacheInvalidatedByAddresses_Phase1_ChunkIter") + .suppressFor(5.0) + .detail("DbId", self->dbId) + .detail("AddressCount", addresses.size()) + .detail("InvalidatedRanges", rangesToInvalidate.size()) + .detail("RangeChunkThreshold", rangeChunkThreshold) + .detail("Phase1RangesScanned", phase1RangesScanned) + .detail("Phase1Yields", phase1Yields); + + // Process as many ranges as possible within chunk threshold + auto iter = self->locationCache.rangeContaining(cursor); + auto endIter = self->locationCache.ranges().end(); + bool rangesRemaining = false; + int currRangesScanned = 0; + for (; iter != endIter; ++iter) { + if (currRangesScanned >= rangeChunkThreshold) { + rangesRemaining = true; + break; + } + ++currRangesScanned; + cursor = iter->end(); + if (!iter->value()) { + continue; + } + auto& loc = iter->value(); + for (int i = 0; i < loc->size(); ++i) { + if (addresses.contains(loc->getInterface(i).address())) { + rangesToInvalidate.push_back(KeyRange(KeyRangeRef(iter->begin(), iter->end()))); + break; + } + } + } + phase1RangesScanned += currRangesScanned; + + if (!rangesRemaining) { + TraceEvent("LocationCacheInvalidatedByAddresses_Phase1_End") + .detail("DbId", self->dbId) + .detail("AddressCount", addresses.size()) + .detail("InvalidatedRanges", rangesToInvalidate.size()) + .detail("RangeChunkThreshold", rangeChunkThreshold) + .detail("Phase1RangesScanned", phase1RangesScanned) + .detail("Phase1Yields", phase1Yields) + .detail("Phase1Duration", now() - startT); + break; + } + + ++phase1Yields; + TraceEvent("LocationCacheInvalidatedByAddresses_Phase1_Yield") + .suppressFor(5.0) + .detail("DbId", self->dbId) + .detail("AddressCount", addresses.size()) + .detail("InvalidatedRanges", rangesToInvalidate.size()) + .detail("RangeChunkThreshold", rangeChunkThreshold) + .detail("Phase1RangesScanned", phase1RangesScanned) + .detail("Phase1Yields", phase1Yields); + co_await yield(); + } + + // Phase 2: invalidate the cache based on invalid ranges computed in Phase 1 + TraceEvent("LocationCacheInvalidatedByAddresses_Phase2_Begin") + .detail("DbId", self->dbId) + .detail("AddressCount", addresses.size()) + .detail("InvalidatedRanges", rangesToInvalidate.size()); + int phase2Idx = 0; + int phase2Yields = 0; + double phase2StartT = now(); + for (; phase2Idx < rangesToInvalidate.size(); phase2Idx++) { + self->locationCache.insert(rangesToInvalidate[phase2Idx], Reference()); + if ((phase2Idx + 1) % rangeChunkThreshold == 0) { + ++phase2Yields; + TraceEvent("LocationCacheInvalidatedByAddresses_Phase2_Yield") + .suppressFor(5.0) + .detail("DbId", self->dbId) + .detail("AddressCount", addresses.size()) + .detail("InvalidatedRanges", rangesToInvalidate.size()) + .detail("RangeChunkThreshold", rangeChunkThreshold) + .detail("Phase2RangesScanned", phase2Idx + 1) + .detail("Phase2Yields", phase2Yields); + co_await yield(); + } + } + TraceEvent("LocationCacheInvalidatedByAddresses_Phase2_End") + .detail("DbId", self->dbId) + .detail("AddressCount", addresses.size()) + .detail("InvalidatedRanges", rangesToInvalidate.size()) + .detail("RangeChunkThreshold", rangeChunkThreshold) + .detail("Phase2RangesScanned", phase2Idx) + .detail("Phase2Yields", phase2Yields) + .detail("Phase2Duration", now() - phase2StartT) + .detail("OverallDuration", now() - startT); + + co_return; +} + +// Periodically samples FlowTransport's persistent per-address connect-failed +// counter and evicts any address whose count advanced since the previous tick +// (a "flap"). This is a direct ConnectionTimeout (CTO) signal: every connect failure increments +// the counter, and any positive delta within an evictor interval indicates an +// address that is still being targeted by RPCs but cannot establish a +// connection. +static Future locationCachePeerEvictorActor(DatabaseContext* cx) { + double evictorDelay = CLIENT_KNOBS->LOCATION_CACHE_PEER_EVICTOR_DELAY; + int evictorFailedThreshold = CLIENT_KNOBS->LOCATION_CACHE_PEER_EVICTOR_FAILED_THRESHOLD; + ASSERT(evictorDelay > 0); + ASSERT(evictorFailedThreshold >= 0); + // Per-address snapshot of FlowTransport's persistent connect-failed counter + // taken on the previous tick. The delta to the current count is the flap + // signal: a positive delta means the address is still being targeted by RPCs + // but cannot connect. + std::unordered_map lastConnectFailedSnapshot; + for (;;) { + try { + co_await delay(evictorDelay); + + std::unordered_set deadAddressSet; + const auto& persistent = FlowTransport::transport().getPersistentConnectFailedCounts(); + for (const auto& [addr, cur] : persistent) { + if (!addr.isValid()) { + continue; + } + int64_t prev = 0; + auto snapIt = lastConnectFailedSnapshot.find(addr); + if (snapIt != lastConnectFailedSnapshot.end()) { + prev = snapIt->second; + } + // If the persistent counter went backwards, the entry was TTL-pruned and re-added + // since the last sweep (its count reset to a small value). Count from zero in that + // case so a genuine post-reset connect failure isn't missed for a sweep. + int64_t delta = (cur.count >= prev) ? (cur.count - prev) : cur.count; + lastConnectFailedSnapshot[addr] = cur.count; + if (delta > evictorFailedThreshold) { + TraceEvent("LocationCachePeerEvictor_FoundDeadAddr") + .suppressFor(5.0) + .detail("DbId", cx->dbId) + .detail("Addr", addr) + .detail("ConnectFailedDelta", delta) + .detail("ConnectFailedTotal", cur.count); + deadAddressSet.insert(addr); + } + } + // Drop snapshot entries for addrs FlowTransport no longer reports a counter + // for, so this map stays bounded alongside the persistent one. + for (auto it = lastConnectFailedSnapshot.begin(); it != lastConnectFailedSnapshot.end();) { + if (persistent.find(it->first) == persistent.end()) { + TraceEvent("LocationCachePeerEvictor_ClearAddrInSnapshot") + .suppressFor(5.0) + .detail("DbId", cx->dbId) + .detail("Addr", it->first); + it = lastConnectFailedSnapshot.erase(it); + } else { + ++it; + } + } + if (!deadAddressSet.empty()) { + TraceEvent("LocationCachePeerEvictor_DeadAddrSummary") + .detail("DbId", cx->dbId) + .detail("DeadAddrSetSize", deadAddressSet.size()); + } + co_await invalidateCacheByAddresses(cx, deadAddressSet); + } catch (Error& e) { + // actor_cancelled must propagate so ~DatabaseContext can tear down the + // evictor; any other error should not kill the loop (that would stop the + // eviction sweep). + if (e.code() == error_code_actor_cancelled) { + throw; + } + TraceEvent(SevWarn, "LocationCachePeerEvictor_Error").error(e).detail("DbId", cx->dbId); + } + } +} + DatabaseContext::DatabaseContext(Reference>> connectionRecord, Reference> clientInfo, Reference> const> coordinator, @@ -1201,6 +1405,9 @@ DatabaseContext::DatabaseContext(ReferenceLOCATION_CACHE_PEER_EVICTOR_ENABLED) { + locationCachePeerEvictor = locationCachePeerEvictorActor(this); + } smoothMidShardSize.reset(CLIENT_KNOBS->INIT_MID_SHARD_BYTES); globalConfig = std::make_unique(this); @@ -1499,6 +1706,7 @@ DatabaseContext::~DatabaseContext() { clientStatusUpdater.actor.cancel(); throttleExpirer.cancel(); statusLeaderMon.cancel(); + locationCachePeerEvictor.cancel(); if (grvUpdateHandler.isValid()) { grvUpdateHandler.cancel(); diff --git a/fdbclient/FileBackupAgent.cpp b/fdbclient/FileBackupAgent.cpp index d07cc508d1b..de44442f40d 100644 --- a/fdbclient/FileBackupAgent.cpp +++ b/fdbclient/FileBackupAgent.cpp @@ -5032,6 +5032,10 @@ void AccumulatedMutations::addChunk(int chunkNumber, const KeyValueRef& kv) { } bool AccumulatedMutations::isComplete() const { + return getCompleteMutations().present(); +} + +Optional AccumulatedMutations::getCompleteMutations() const { if (lastChunkNumber >= 0) { StringRefReader reader(serializedMutations, restore_corrupted_data()); @@ -5041,10 +5045,12 @@ bool AccumulatedMutations::isComplete() const { } uint32_t vLen = reader.consume(); - return vLen == reader.remainder().size(); + if (vLen == reader.remainder().size()) { + return StringRef(serializedMutations); + } } - return false; + return {}; } // Returns true if a complete chunk contains any MutationRefs which intersect with any @@ -5112,7 +5118,8 @@ std::vector filterLogMutationKVPairs(VectorRef data, c // If the mutations are incomplete or match one of the ranges, include in results. if (!m.isComplete() || m.matchesAnyRange(filters)) { - output.insert(output.end(), m.kvs.begin(), m.kvs.end()); + const auto& chunks = m.getChunks(); + output.insert(output.end(), chunks.begin(), chunks.end()); } } diff --git a/fdbclient/GlobalConfig.cpp b/fdbclient/GlobalConfig.cpp index c4df4018c09..590f178f452 100644 --- a/fdbclient/GlobalConfig.cpp +++ b/fdbclient/GlobalConfig.cpp @@ -71,16 +71,16 @@ Key GlobalConfig::prefixedKey(KeyRef key) { return key.withPrefix(SpecialKeySpace::getModuleRange(SpecialKeySpace::MODULE::GLOBALCONFIG).begin); } -Reference GlobalConfig::get(KeyRef name) { +Reference GlobalConfig::get(KeyRef name) { auto it = data.find(name); if (it == data.end()) { - return Reference(); + return Reference(); } return it->second; } -std::map> GlobalConfig::get(KeyRangeRef range) { - std::map> results; +std::map> GlobalConfig::get(KeyRangeRef range) { + std::map> results; for (const auto& [key, value] : data) { if (range.contains(key)) { results[key] = value; @@ -128,7 +128,7 @@ void GlobalConfig::insert(KeyRef key, ValueRef value) { data[stableKey] = makeReference(std::move(arena), std::move(any)); if (callbacks.find(stableKey) != callbacks.end()) { - callbacks[stableKey](data[stableKey]->value); + callbacks[stableKey](data[stableKey]->getValue()); } } catch (Error& e) { TraceEvent(SevWarn, "GlobalConfigTupleParseError").detail("What", e.what()); diff --git a/fdbclient/KeyRangeMap.cpp b/fdbclient/KeyRangeMap.cpp index f521f46a0d7..fd59bdba9bd 100644 --- a/fdbclient/KeyRangeMap.cpp +++ b/fdbclient/KeyRangeMap.cpp @@ -25,6 +25,10 @@ #include "fdbclient/ReadYourWrites.h" #include "flow/UnitTest.h" +#include +#include +#include + void KeyRangeActorMap::getRangesAffectedByInsertion(const KeyRangeRef& keys, std::vector& affectedRanges) { auto s = map.rangeContaining(keys.begin); if (s.begin() != keys.begin && s.value().isValid() && !s.value().isReady()) @@ -318,6 +322,56 @@ Future krmSetRangeCoalescing(Reference const& t return holdWhile(tr, krmSetRangeCoalescing_(tr.getPtr(), mapPrefix, range, maxRange, value)); } +TEST_CASE("/keyrangemap/coalesced/singleKey") { + Arena arena; + const Key mapEnd = "z\x00"_sr; + CoalescedKeyRangeMap> owning(0, mapEnd); + CoalescedKeyRefRangeMap> arenaBacked(0, mapEnd); + + auto check = [&](const auto& map, std::initializer_list> expected) { + auto actual = map.ranges().begin(); + int expectedMetric = 0; + for (auto boundary = expected.begin(); boundary != expected.end(); ++boundary) { + ASSERT(actual != map.ranges().end()); + ASSERT(actual.begin() == boundary->first); + ASSERT(actual.value() == boundary->second); + auto next = boundary + 1; + ASSERT(actual.end() == (next == expected.end() ? mapEnd : next->first)); + using StoredKey = std::decay_t; + expectedMetric += boundary->first.size() + sizeof(MapPair); + ++actual; + } + ASSERT(actual == map.ranges().end()); + ASSERT(map.sumRange(allKeys.begin, KeyRef(mapEnd)) == expectedMetric); + }; + auto insertAndCheck = [&](KeyRef key, int value, std::initializer_list> expected) { + owning.insert(key, value); + arenaBacked.insert(key, value, arena); + check(owning, expected); + check(arenaBacked, expected); + }; + + // Split a range, repeat an insertion, and extend through the immediate successor. + insertAndCheck("b"_sr, 1, { { ""_sr, 0 }, { "b"_sr, 1 }, { "b\x00"_sr, 0 } }); + insertAndCheck("b"_sr, 1, { { ""_sr, 0 }, { "b"_sr, 1 }, { "b\x00"_sr, 0 } }); + insertAndCheck("b\x00"_sr, 1, { { ""_sr, 0 }, { "b"_sr, 1 }, { "b\x00\x00"_sr, 0 } }); + + // Overwrite a boundary, then coalesce with the left, right, and both neighbors. + insertAndCheck("b"_sr, 2, { { ""_sr, 0 }, { "b"_sr, 2 }, { "b\x00"_sr, 1 }, { "b\x00\x00"_sr, 0 } }); + insertAndCheck("b"_sr, 0, { { ""_sr, 0 }, { "b\x00"_sr, 1 }, { "b\x00\x00"_sr, 0 } }); + insertAndCheck("b"_sr, 1, { { ""_sr, 0 }, { "b"_sr, 1 }, { "b\x00\x00"_sr, 0 } }); + insertAndCheck("b\x00"_sr, 0, { { ""_sr, 0 }, { "b"_sr, 1 }, { "b\x00"_sr, 0 } }); + insertAndCheck("b"_sr, 0, { { ""_sr, 0 } }); + insertAndCheck("c"_sr, 0, { { ""_sr, 0 } }); + + // Keep the beginning and end sentinels, including when keyAfter(key) equals mapEnd. + insertAndCheck(""_sr, 1, { { ""_sr, 1 }, { "\x00"_sr, 0 } }); + insertAndCheck(""_sr, 0, { { ""_sr, 0 } }); + insertAndCheck("z"_sr, 1, { { ""_sr, 0 }, { "z"_sr, 1 } }); + insertAndCheck("z"_sr, 0, { { ""_sr, 0 } }); + return Void(); +} + TEST_CASE("/keyrangemap/decoderange/aligned") { Arena arena; Key prefix = "/prefix/"_sr; diff --git a/fdbclient/ManagementAPI.cpp b/fdbclient/ManagementAPI.cpp index 6247bd1c697..c83b44233ed 100644 --- a/fdbclient/ManagementAPI.cpp +++ b/fdbclient/ManagementAPI.cpp @@ -2320,7 +2320,8 @@ Future timeKeeperSetDisable(Database cx) { } } -Future lockDatabase(Transaction* tr, UID id) { +template +static Future lockDatabaseImpl(TransactionHandle tr, UID id) { tr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); tr->setOption(FDBTransactionOptions::LOCK_AWARE); Optional val = co_await tr->get(databaseLockedKey); @@ -2340,24 +2341,12 @@ Future lockDatabase(Transaction* tr, UID id) { tr->addWriteConflictRange(normalKeys); } -Future lockDatabase(Reference tr, UID id) { - tr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); - tr->setOption(FDBTransactionOptions::LOCK_AWARE); - Optional val = co_await tr->get(databaseLockedKey); - - if (val.present()) { - if (BinaryReader::fromStringRef(val.get().substr(10), Unversioned()) == id) { - co_return; - } else { - //TraceEvent("DBA_LockLocked").detail("Expecting", id).detail("Lock", BinaryReader::fromStringRef(val.get().substr(10), Unversioned())); - throw database_locked(); - } - } +Future lockDatabase(Transaction* tr, UID id) { + return lockDatabaseImpl(tr, id); +} - tr->atomicOp(databaseLockedKey, - BinaryWriter::toValue(id, Unversioned()).withPrefix("0123456789"_sr).withSuffix("\x00\x00\x00\x00"_sr), - MutationRef::SetVersionstampedValue); - tr->addWriteConflictRange(normalKeys); +Future lockDatabase(Reference tr, UID id) { + return lockDatabaseImpl(std::move(tr), id); } Future lockDatabase(Database cx, UID id) { @@ -2380,7 +2369,8 @@ Future lockDatabase(Database cx, UID id) { } } -Future unlockDatabase(Transaction* tr, UID id) { +template +static Future unlockDatabaseImpl(TransactionHandle tr, UID id) { tr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); tr->setOption(FDBTransactionOptions::LOCK_AWARE); Optional val = co_await tr->get(databaseLockedKey); @@ -2396,20 +2386,12 @@ Future unlockDatabase(Transaction* tr, UID id) { tr->clear(singleKeyRange(databaseLockedKey)); } -Future unlockDatabase(Reference tr, UID id) { - tr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); - tr->setOption(FDBTransactionOptions::LOCK_AWARE); - Optional val = co_await tr->get(databaseLockedKey); - - if (!val.present()) - co_return; - - if (val.present() && BinaryReader::fromStringRef(val.get().substr(10), Unversioned()) != id) { - //TraceEvent("DBA_UnlockLocked").detail("Expecting", id).detail("Lock", BinaryReader::fromStringRef(val.get().substr(10), Unversioned())); - throw database_locked(); - } +Future unlockDatabase(Transaction* tr, UID id) { + return unlockDatabaseImpl(tr, id); +} - tr->clear(singleKeyRange(databaseLockedKey)); +Future unlockDatabase(Reference tr, UID id) { + return unlockDatabaseImpl(std::move(tr), id); } Future unlockDatabase(Database cx, UID id) { @@ -2429,7 +2411,8 @@ Future unlockDatabase(Database cx, UID id) { } } -Future checkDatabaseLock(Transaction* tr, UID id) { +template +static Future checkDatabaseLockImpl(TransactionHandle tr, UID id) { tr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); tr->setOption(FDBTransactionOptions::LOCK_AWARE); Optional val = co_await tr->get(databaseLockedKey); @@ -2440,15 +2423,12 @@ Future checkDatabaseLock(Transaction* tr, UID id) { } } -Future checkDatabaseLock(Reference tr, UID id) { - tr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); - tr->setOption(FDBTransactionOptions::LOCK_AWARE); - Optional val = co_await tr->get(databaseLockedKey); +Future checkDatabaseLock(Transaction* tr, UID id) { + return checkDatabaseLockImpl(tr, id); +} - if (val.present() && BinaryReader::fromStringRef(val.get().substr(10), Unversioned()) != id) { - //TraceEvent("DBA_CheckLocked").detail("Expecting", id).detail("Lock", BinaryReader::fromStringRef(val.get().substr(10), Unversioned())).backtrace(); - throw database_locked(); - } +Future checkDatabaseLock(Reference tr, UID id) { + return checkDatabaseLockImpl(std::move(tr), id); } Future advanceVersion(Database cx, Version v) { diff --git a/fdbclient/MonitorLeader.cpp b/fdbclient/MonitorLeader.cpp index b7d435d9ad1..b70f6913a60 100644 --- a/fdbclient/MonitorLeader.cpp +++ b/fdbclient/MonitorLeader.cpp @@ -889,6 +889,9 @@ void shrinkProxyList(ClientDBInfo& ni, std::vector& lastGrvProxyUIDs, std::vector& lastGrvProxies) { if (ni.commitProxies.size() > CLIENT_KNOBS->MAX_COMMIT_PROXY_CONNECTIONS) { + // Cache the sampled subset to prevent proxy churn: while above MAX, we keep + // talking to the same subset across ClientDBInfo updates instead of reshuffling + // on every update. std::vector commitProxyUIDs; commitProxyUIDs.reserve(ni.commitProxies.size()); for (auto& commitProxy : ni.commitProxies) { @@ -905,8 +908,40 @@ void shrinkProxyList(ClientDBInfo& ni, } ni.firstCommitProxy = ni.commitProxies[0]; ni.commitProxies = lastCommitProxies; + } else if (CLIENT_KNOBS->SHRINK_PROXY_LIST_CLEAR_CACHE_BELOW_THRESHOLD) { + // Why clear here: say MAX=5 and we recruited 6 CPs, so we sampled+cached 5 of + // them above. Now a CP is killed, a recovery occurs, but the new recruited count + // is 5 (6 -> 5), which falls at/below MAX so we land in this branch and never + // re-enter the shrink path. The cached list still holds the pre-recovery sample + // -- including the killed CP's interface -- so it keeps that CP's RequestStreams + // (and their peer references), which causes stale peer issues + // (see StalePeerTest.toml with killRole=commit_proxy). + // + // Clearing is safe: + // - Below MAX the cache is not used at all (we talk to every recruited proxy), + // so clearing has zero effect on churn or perf here. + // - The one case where the cache could have helped is 6 -> 5 -> 6 returning + // to the *same* set: baseline would reuse, we re-sample. But crossing MAX + // again means a recovery, which recruits CPs with new UIDs, so the set is + // not the same and baseline would re-sample too. + // Either way it is a one-time reshuffle, not steady-state, and + // the knob means we can experiment and have it be off in case of performance concerns. + if (!lastCommitProxyUIDs.empty()) { + CODE_PROBE(true, + "commit proxy recruited count dropped to at/below MAX_COMMIT_PROXY_CONNECTIONS, clearing cache"); + TraceEvent("ShrinkProxyListCacheCleared") + .detail("PeerRole", "commit_proxy") + .detail("PrevCachedCount", lastCommitProxyUIDs.size()) + .detail("RecruitedCount", ni.commitProxies.size()) + .detail("MaxConnections", CLIENT_KNOBS->MAX_COMMIT_PROXY_CONNECTIONS); + lastCommitProxyUIDs.clear(); + lastCommitProxies.clear(); + } } if (ni.grvProxies.size() > CLIENT_KNOBS->MAX_GRV_PROXY_CONNECTIONS) { + // Cache the sampled subset to prevent proxy churn: while above MAX, we keep + // talking to the same subset across ClientDBInfo updates instead of reshuffling + // on every update. std::vector grvProxyUIDs; grvProxyUIDs.reserve(ni.grvProxies.size()); for (auto& grvProxy : ni.grvProxies) { @@ -922,6 +957,21 @@ void shrinkProxyList(ClientDBInfo& ni, } } ni.grvProxies = lastGrvProxies; + } else if (CLIENT_KNOBS->SHRINK_PROXY_LIST_CLEAR_CACHE_BELOW_THRESHOLD) { + // Same as the commit-proxy branch above (see that comment for the why and the + // no-perf-cost reasoning): a GP is killed, a recovery occurs, but the new + // recruited count drops to at/below MAX, so we clear the cache to avoid + // pinning the pre-recovery GP interfaces. + if (!lastGrvProxyUIDs.empty()) { + CODE_PROBE(true, "grv proxy recruited count dropped to at/below MAX_GRV_PROXY_CONNECTIONS, clearing cache"); + TraceEvent("ShrinkProxyListCacheCleared") + .detail("PeerRole", "grv_proxy") + .detail("PrevCachedCount", lastGrvProxyUIDs.size()) + .detail("RecruitedCount", ni.grvProxies.size()) + .detail("MaxConnections", CLIENT_KNOBS->MAX_GRV_PROXY_CONNECTIONS); + lastGrvProxyUIDs.clear(); + lastGrvProxies.clear(); + } } } diff --git a/fdbclient/MultiVersionTransaction.cpp b/fdbclient/MultiVersionTransaction.cpp index d974afbb941..a6ab21161e6 100644 --- a/fdbclient/MultiVersionTransaction.cpp +++ b/fdbclient/MultiVersionTransaction.cpp @@ -399,9 +399,13 @@ NativeCdcStreamInfo copyNativeCdcStreamInfo(FdbCApi::FDBNativeCdcStreamInfo cons NativeCdcStreamInfo result; result.name = Key(StringRef(source.name.key, source.name.keyLength)); result.streamId = source.streamId; - result.keys = KeyRange( - KeyRangeRef(KeyRef(static_cast(source.keyRange.beginKey), source.keyRange.beginKeyLength), - KeyRef(static_cast(source.keyRange.endKey), source.keyRange.endKeyLength))); + result.ranges.reserve(source.rangeCount); + for (int i = 0; i < source.rangeCount; ++i) { + auto const& range = source.ranges[i]; + result.ranges.emplace_back( + KeyRangeRef(KeyRef(static_cast(range.beginKey), range.beginKeyLength), + KeyRef(static_cast(range.endKey), range.endKeyLength))); + } result.minVersion = source.minVersion; return result; } @@ -555,13 +559,21 @@ ThreadFuture DLDatabase::createSnapshot(const StringRef& uid, const String return toThreadFuture(api, f, [](FdbCApi::FDBFuture* f, FdbCApi* api) { return Void(); }); } -ThreadFuture DLDatabase::registerNativeCdcStream(const KeyRef& name, const KeyRangeRef& keys) { +ThreadFuture DLDatabase::registerNativeCdcStream(const KeyRef& name, const std::vector& ranges) { if (!api->databaseRegisterNativeCdcStream) { return unsupported_operation(); } + if (ranges.empty() || ranges.size() > NATIVE_CDC_MAX_RANGES) { + return client_invalid_operation(); + } - FdbCApi::FDBFuture* f = api->databaseRegisterNativeCdcStream( - db, name.begin(), name.size(), keys.begin.begin(), keys.begin.size(), keys.end.begin(), keys.end.size()); + std::vector cRanges; + cRanges.reserve(ranges.size()); + for (auto const& range : ranges) { + cRanges.push_back({ range.begin.begin(), range.begin.size(), range.end.begin(), range.end.size() }); + } + FdbCApi::FDBFuture* f = + api->databaseRegisterNativeCdcStream(db, name.begin(), name.size(), cRanges.data(), cRanges.size()); return toThreadFuture(api, f, [](FdbCApi::FDBFuture* f, FdbCApi* api) { uint64_t streamId; FdbCApi::fdb_error_t error = api->futureGetUInt64(f, &streamId); @@ -1656,8 +1668,9 @@ ThreadFuture MultiVersionDatabase::createSnapshot(const StringRef& uid, co return executeOperation(&IDatabase::createSnapshot, uid, snapshot_command); } -ThreadFuture MultiVersionDatabase::registerNativeCdcStream(const KeyRef& name, const KeyRangeRef& keys) { - return executeOperation(&IDatabase::registerNativeCdcStream, name, keys); +ThreadFuture MultiVersionDatabase::registerNativeCdcStream(const KeyRef& name, + const std::vector& ranges) { + return executeOperation(&IDatabase::registerNativeCdcStream, name, ranges); } ThreadFuture MultiVersionDatabase::removeNativeCdcStream(const KeyRef& name) { diff --git a/fdbclient/NativeAPI.cpp b/fdbclient/NativeAPI.cpp index 15ac1d1f171..8dd78c6f951 100644 --- a/fdbclient/NativeAPI.cpp +++ b/fdbclient/NativeAPI.cpp @@ -3170,7 +3170,7 @@ Optional> maybeDuplicateTSSStr ReplyPromiseStream tssReplyStream = tssRequestStream.getReplyStream(req); PromiseStream ssDuplicateReplyStream; TSSDuplicateStreamData streamData(ssDuplicateReplyStream); - model->addActor.send(tssStreamComparison(req, streamData, tssReplyStream, tssData.get())); + model->addBackgroundActor(tssStreamComparison(req, streamData, tssReplyStream, tssData.get())); return Optional>(streamData); } } diff --git a/fdbclient/NativeCdc.cpp b/fdbclient/NativeCdc.cpp index c1403266ac6..916e0fdf3cd 100644 --- a/fdbclient/NativeCdc.cpp +++ b/fdbclient/NativeCdc.cpp @@ -36,16 +36,6 @@ #include "flow/Trace.h" #include "flow/UnitTest.h" -namespace { - -using CDCTagId = uint16_t; - -constexpr uint32_t maxNativeCdcTagCount = static_cast(std::numeric_limits::max()) + 1; - -bool validNativeCdcTagCount(int tagCount) { - return tagCount > 0 && static_cast(tagCount) <= maxNativeCdcTagCount; -} - void validateNativeCdcEnabled(bool enabled) { if (!enabled) { CODE_PROBE(true, "Native CDC registration rejected while feature disabled"); @@ -53,192 +43,45 @@ void validateNativeCdcEnabled(bool enabled) { } } -class NativeCdcIdentifierAllocator { - bool sawStream = false; - CDCStreamId maxStreamId = 0; - std::unordered_map tagStreamCounts; - -public: - void observeStreamId(CDCStreamId streamId) { - sawStream = true; - maxStreamId = std::max(maxStreamId, streamId); - } - - void observeTag(Tag tag) { - ASSERT_WE_THINK(tag.locality == tagLocalityCDC); - ++tagStreamCounts[tag.id]; - } - - bool hasStreams(Tag tag) const { return tagStreamCounts.contains(tag.id); } - - std::pair allocate(int tagCount) const { - if (sawStream && maxStreamId == std::numeric_limits::max()) { - throw operation_failed(); - } - - const CDCStreamId streamId = sawStream ? maxStreamId + 1 : 1; - if (!validNativeCdcTagCount(tagCount)) { - throw invalid_option_value(); - } - uint32_t leastStreams = std::numeric_limits::max(); - CDCTagId selectedTagId = 0; - // TODO: Use data-distributor-observed per-tag write throughput to rebalance CDC tags, including - // migrating active streams with versioned tag-history assignments. - for (uint32_t tagId = 0; tagId < static_cast(tagCount); ++tagId) { - auto count = tagStreamCounts.find(static_cast(tagId)); - const uint32_t streamCount = count == tagStreamCounts.end() ? 0 : count->second; - if (streamCount < leastStreams) { - leastStreams = streamCount; - selectedTagId = static_cast(tagId); - } - } - return { streamId, Tag(tagLocalityCDC, selectedTagId) }; - } -}; - -void validateNativeCdcStream(KeyRef const& name, KeyRangeRef const& keys) { - if (name.empty() || keys.empty() || !normalKeys.contains(keys)) { +void normalizeNativeCdcStreamRanges(KeyRef const& name, std::vector& ranges) { + if (name.empty() || ranges.empty() || ranges.size() > NATIVE_CDC_MAX_RANGES) { throw client_invalid_operation(); } -} - -Future> getNativeCdcProxyAssignment(Transaction* tr, CDCStreamId streamId) { - RangeResult assignments = co_await tr->getRange(cdcProxyRangeFor(streamId), 2); - ASSERT_LE(assignments.size(), 1); - if (assignments.empty()) { - co_return Optional(); - } - const auto [assignedStreamId, proxyId] = decodeCDCProxyKey(assignments[0].key); - ASSERT_WE_THINK(assignedStreamId == streamId); - co_return proxyId; -} - -Future getNativeCdcCurrentTag(Transaction* tr, CDCStreamId streamId) { - // Tag-history keys sort by their big-endian assignment version, so the final - // key in this stream's prefix range contains its current tag. - RangeResult history = co_await tr->getRange(cdcTagHistoryRangeFor(streamId), 1, Snapshot::False, Reverse::True); - if (history.empty()) { - throw client_invalid_operation(); - } - co_return decodeCDCTagHistoryKey(history.front().key).tag; -} - -Future> getNativeCdcProxyAssignmentForTag(Transaction* tr, Tag targetTag) { - const Key ownerKey = cdcTagOwnerKeyFor(targetTag); - Optional indexedStream = co_await tr->get(ownerKey); - if (indexedStream.present()) { - const CDCStreamId streamId = decodeCDCTagOwnerValue(indexedStream.get()); - Future> activeStream = tr->get(cdcStreamKeyFor(streamId)); - RangeResult history = co_await tr->getRange(cdcTagHistoryRangeFor(streamId), 1, Snapshot::False, Reverse::True); - // Keep the await separate so GCC 13 does not evaluate history.front() before the short-circuit guards. - const Optional activeStreamValue = co_await activeStream; - // The index is derived: removal or retagging can invalidate its representative, and the per-stream - // assignment remains authoritative across proxy replacement, including by older metadata writers. - if (activeStreamValue.present() && !history.empty() && - decodeCDCTagHistoryKey(history.front().key).tag == targetTag) { - Optional proxyId = co_await getNativeCdcProxyAssignment(tr, streamId); - if (proxyId.present()) { - CODE_PROBE(true, "Native CDC resolves a shared tag owner from its persisted index"); - co_return proxyId; - } - } - CODE_PROBE(true, "Native CDC rebuilds a stale tag owner index"); - tr->clear(ownerKey); - } - - std::set activeStreamIds; - Key begin = cdcStreamKeys.begin; - while (begin < cdcStreamKeys.end) { - RangeResult streams = co_await tr->getRange(KeyRangeRef(begin, cdcStreamKeys.end), CLIENT_KNOBS->TOO_MANY); - for (const auto& stream : streams) { - activeStreamIds.insert(decodeCDCStreamKey(stream.key)); - } - if (!streams.more) { - break; + for (const auto& range : ranges) { + if (range.begin >= range.end || !normalKeys.contains(range)) { + throw client_invalid_operation(); } - begin = keyAfter(streams.back().key); } - - std::unordered_map currentTags; - begin = cdcTagHistoryKeys.begin; - while (begin < cdcTagHistoryKeys.end) { - RangeResult histories = - co_await tr->getRange(KeyRangeRef(begin, cdcTagHistoryKeys.end), CLIENT_KNOBS->TOO_MANY); - for (const auto& history : histories) { - const CDCTagHistoryEntry decoded = decodeCDCTagHistoryKey(history.key); - if (activeStreamIds.contains(decoded.streamId)) { - currentTags[decoded.streamId] = decoded.tag; + std::sort( + ranges.begin(), ranges.end(), [](const KeyRange& lhs, const KeyRange& rhs) { return lhs.begin < rhs.begin; }); + size_t count = 0; + for (const auto& range : ranges) { + if (count > 0 && range.begin <= ranges[count - 1].end) { + if (range.end > ranges[count - 1].end) { + ranges[count - 1] = KeyRange(KeyRangeRef(ranges[count - 1].begin, range.end)); } + } else { + ranges[count++] = range; } - if (!histories.more) { - break; - } - begin = keyAfter(histories.back().key); } + ranges.resize(count); - for (const auto& [streamId, tag] : currentTags) { - if (tag == targetTag) { - Optional proxyId = co_await getNativeCdcProxyAssignment(tr, streamId); - if (proxyId.present()) { - tr->set(ownerKey, cdcTagOwnerValue(streamId)); - CODE_PROBE(true, "Native CDC reconstructs a missing tag owner index from active streams"); - co_return proxyId; - } - } + int64_t keyBytes = 0; + for (const auto& range : ranges) { + // Single-key ranges serialize only the end key, so count the encoded payload before allocating metadata. + keyBytes += static_cast(range.end.size()) + (range.singleKeyRange() ? 0 : range.begin.size()); + } + if (keyBytes > CLIENT_KNOBS->VALUE_SIZE_LIMIT || + cdcStreamKeysValue(ranges).size() > CLIENT_KNOBS->VALUE_SIZE_LIMIT) { + throw client_invalid_operation(); } - co_return Optional(); } -void signalNativeCdcProxyAssignmentChange(Transaction* tr) { - // Assignment updates are low-rate control-plane operations. A single - // coalescing signal lets the cluster controller rescan all durable owners. - tr->set(cdcProxyAssignmentChangeKey, - BinaryWriter::toValue(deterministicRandom()->randomUniqueID(), - IncludeVersion(ProtocolVersion::withNativeCdc()))); +bool nativeCdcNameMatchesStream(Optional const& currentId, CDCStreamId streamId) { + return currentId.present() && decodeCDCStreamNameValue(currentId.get()) == streamId; } -Future observeNativeCdcMetadata(Transaction* tr, NativeCdcIdentifierAllocator* allocator) { - Optional maxStreamId = co_await tr->get(cdcMaxStreamIdKey); - if (maxStreamId.present()) { - allocator->observeStreamId(decodeCDCMaxStreamIdValue(maxStreamId.get())); - } - - std::set activeStreamIds; - Key begin = cdcStreamKeys.begin; - while (begin < cdcStreamKeys.end) { - RangeResult streams = co_await tr->getRange(KeyRangeRef(begin, cdcStreamKeys.end), CLIENT_KNOBS->TOO_MANY); - for (const auto& kv : streams) { - const CDCStreamId streamId = decodeCDCStreamKey(kv.key); - activeStreamIds.insert(streamId); - allocator->observeStreamId(streamId); - } - if (!streams.more) { - break; - } - begin = keyAfter(streams.back().key); - } - - std::unordered_map currentTags; - begin = cdcTagHistoryKeys.begin; - while (begin < cdcTagHistoryKeys.end) { - RangeResult histories = - co_await tr->getRange(KeyRangeRef(begin, cdcTagHistoryKeys.end), CLIENT_KNOBS->TOO_MANY); - for (const auto& kv : histories) { - const CDCTagHistoryEntry history = decodeCDCTagHistoryKey(kv.key); - allocator->observeStreamId(history.streamId); - if (activeStreamIds.contains(history.streamId)) { - currentTags[history.streamId] = history.tag; - } - } - if (!histories.more) { - break; - } - begin = keyAfter(histories.back().key); - } - for (const auto& tagAssignment : currentTags) { - allocator->observeTag(tagAssignment.second); - } -} +namespace { bool retryNativeCdcProxyRequest(Error const& error) { return error.code() == error_code_wrong_shard_server || error.code() == error_code_broken_promise || @@ -258,8 +101,8 @@ bool rewindUnacknowledgedCursorAfterProxyReplacement(CDCCursor* currentPosition, return true; } -// TODO: Have the cluster controller rebalance stream ownership using aggregate CDC proxy throughput and -// update cdcProxyKeys and ClientDBInfo assignments; registration currently chooses any available proxy. +// TODO: Use measured aggregate CDC proxy throughput instead of stream counts when balancing ownership; +// registration currently chooses any available proxy before the controller's opt-in balancing pass. Optional selectAvailableNativeCdcProxy(ClientDBInfo const& clientInfo, Optional previousProxy) { for (const auto& proxy : clientInfo.cdcProxies) { if (!previousProxy.present() || proxy.id() != previousProxy.get()) { @@ -347,10 +190,6 @@ Future getNativeCdcStreamProxy(Database cx, CDCStreamId strea } } -bool nativeCdcNameMatchesStream(Optional const& currentId, CDCStreamId streamId) { - return currentId.present() && decodeCDCStreamNameValue(currentId.get()) == streamId; -} - Future namedNativeCdcStreamStillExists(Database cx, Key name, CDCStreamId streamId) { Transaction tr(cx); while (true) { @@ -456,162 +295,6 @@ Future sampleNativeCdcProxy(CDCProxyInterface proxy, std:: } // namespace -Future registerNativeCdcStream(Database cx, Key name, KeyRange keys, UID proxyId) { - validateNativeCdcStream(name, keys); - - Transaction tr(cx); - while (true) { - Error err; - try { - tr.setOption(FDBTransactionOptions::LOCK_AWARE); - tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); - - const Key nameKey = cdcStreamNameKeyFor(name); - Optional currentId = co_await tr.get(nameKey); - if (currentId.present()) { - const CDCStreamId streamId = decodeCDCStreamNameValue(currentId.get()); - Optional currentKeys = co_await tr.get(cdcStreamKeyFor(streamId)); - if (!currentKeys.present() || decodeCDCStreamKeysValue(currentKeys.get()) != keys) { - throw client_invalid_operation(); - } - if (!(co_await getNativeCdcProxyAssignment(&tr, streamId)).present()) { - CODE_PROBE(true, "Native CDC registration restores missing stream owner", probe::decoration::rare); - const Tag tag = co_await getNativeCdcCurrentTag(&tr, streamId); - Optional sharedTagProxy = co_await getNativeCdcProxyAssignmentForTag(&tr, tag); - CODE_PROBE(sharedTagProxy.present(), - "Native CDC shared-tag streams use one owner", - probe::decoration::rare); - const UID selectedProxy = sharedTagProxy.present() ? sharedTagProxy.get() : proxyId; - tr.set(cdcProxyKeyFor(streamId, selectedProxy), Value()); - if (!sharedTagProxy.present()) { - tr.set(cdcTagOwnerKeyFor(tag), cdcTagOwnerValue(streamId)); - } - signalNativeCdcProxyAssignmentChange(&tr); - co_await tr.commit(); - } - co_return streamId; - } - - // Disabling CDC stops new admission, but existing registrations and - // owner repair must remain available so durable streams can drain. - const bool nativeCdcEnabled = cx->clientInfo->get().nativeCdcEnabled; - const int nativeCdcTagCount = cx->clientInfo->get().nativeCdcTagCount; - validateNativeCdcEnabled(nativeCdcEnabled); - NativeCdcIdentifierAllocator allocator; - co_await observeNativeCdcMetadata(&tr, &allocator); - const auto [streamId, tag] = allocator.allocate(nativeCdcTagCount); - // The read version is a conservative lower bound for tag routing. - // The versionstamped minimum below is the commit version, and stream - // initialization takes their maximum before exposing mutations. - const Version registrationVersion = co_await tr.getReadVersion(); - - tr.set(nameKey, cdcStreamNameValue(streamId)); - tr.set(cdcMaxStreamIdKey, cdcMaxStreamIdValue(streamId)); - tr.set(cdcStreamKeyFor(streamId), cdcStreamKeysValue(keys)); - tr.set(cdcTagHistoryKeyFor(streamId, registrationVersion, tag), Value()); - tr.atomicOp( - cdcMinVersionKeyFor(streamId), cdcVersionstampedMinVersionValue(), MutationRef::SetVersionstampedValue); - Optional sharedTagProxy; - if (allocator.hasStreams(tag)) { - sharedTagProxy = co_await getNativeCdcProxyAssignmentForTag(&tr, tag); - } - const UID selectedProxy = sharedTagProxy.present() ? sharedTagProxy.get() : proxyId; - tr.set(cdcProxyKeyFor(streamId, selectedProxy), Value()); - if (!sharedTagProxy.present()) { - tr.set(cdcTagOwnerKeyFor(tag), cdcTagOwnerValue(streamId)); - } - signalNativeCdcProxyAssignmentChange(&tr); - co_await tr.commit(); - co_return streamId; - } catch (Error& e) { - err = e; - } - co_await tr.onError(err); - } -} - -Future removeNativeCdcStream(Database cx, Key name, CDCStreamId streamId, UID proxyId) { - if (name.empty() || streamId == 0) { - throw client_invalid_operation(); - } - - Transaction tr(cx); - while (true) { - Error err; - try { - tr.setOption(FDBTransactionOptions::LOCK_AWARE); - tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); - - const Key nameKey = cdcStreamNameKeyFor(name); - Optional currentId = co_await tr.get(nameKey); - if (!nativeCdcNameMatchesStream(currentId, streamId)) { - CODE_PROBE(currentId.present(), "Native CDC preserves a replacement stream during removal retry"); - if (currentId.present()) { - TraceEvent("NativeCdcRemovalPreservesReplacement") - .detail("RemovedStreamId", streamId) - .detail("ReplacementStreamId", decodeCDCStreamNameValue(currentId.get())); - } - co_return false; - } - - Optional assignedProxy = co_await getNativeCdcProxyAssignment(&tr, streamId); - if (!assignedProxy.present() || assignedProxy.get() != proxyId) { - CODE_PROBE(true, "Native CDC rejects removal through a stale owner"); - throw wrong_shard_server(); - } - - std::set removedTags; - const KeyRange historyRange = cdcTagHistoryRangeFor(streamId); - Key begin = historyRange.begin; - while (begin < historyRange.end) { - RangeResult history = - co_await tr.getRange(KeyRangeRef(begin, historyRange.end), CLIENT_KNOBS->TOO_MANY); - for (const auto& entry : history) { - removedTags.insert(decodeCDCTagHistoryKey(entry.key).tag); - } - if (!history.more) { - break; - } - begin = keyAfter(history.back().key); - } - - tr.clear(nameKey); - tr.clear(cdcStreamKeyFor(streamId)); - for (const Tag& tag : removedTags) { - const Key ownerKey = cdcTagOwnerKeyFor(tag); - Optional indexedStream = co_await tr.get(ownerKey); - if (indexedStream.present() && decodeCDCTagOwnerValue(indexedStream.get()) == streamId) { - tr.clear(ownerKey); - } - tr.set(cdcRetiredTagPopKeyFor(tag), Value()); - tr.atomicOp(cdcRetiredTagPopVersionKeyFor(tag), - cdcVersionstampedMinVersionValue(), - MutationRef::SetVersionstampedValue); - } - tr.clear(cdcTagHistoryRangeFor(streamId)); - tr.clear(cdcMinVersionKeyFor(streamId)); - tr.clear(cdcProxyRangeFor(streamId)); - if (assignedProxy.present()) { - signalNativeCdcProxyAssignmentChange(&tr); - } - co_await tr.commit(); - CODE_PROBE(!removedTags.empty(), "Native CDC removal records final tagged pop work"); - TraceEvent("NativeCdcStreamRemoved") - .detail("StreamId", streamId) - .detail("ProxyID", proxyId) - .detail("CommitVersion", tr.getCommittedVersion()) - .detail("RetiredTagCount", removedTags.size()); - co_return true; - } catch (Error& e) { - if (e.code() == error_code_wrong_shard_server) { - throw; - } - err = e; - } - co_await tr.onError(err); - } -} - Future> listNativeCdcStreams(Database cx) { Transaction tr(cx); while (true) { @@ -634,12 +317,12 @@ Future> listNativeCdcStreams(Database cx) { begin = keyAfter(page.back().key); } - std::unordered_map streamKeys; + std::unordered_map> streamRanges; begin = cdcStreamKeys.begin; while (begin < cdcStreamKeys.end) { RangeResult page = co_await tr.getRange(KeyRangeRef(begin, cdcStreamKeys.end), CLIENT_KNOBS->TOO_MANY); for (const auto& kv : page) { - streamKeys.emplace(decodeCDCStreamKey(kv.key), decodeCDCStreamKeysValue(kv.value)); + streamRanges.emplace(decodeCDCStreamKey(kv.key), decodeCDCStreamKeysValue(kv.value)); } if (!page.more) { break; @@ -664,11 +347,11 @@ Future> listNativeCdcStreams(Database cx) { std::vector result; result.reserve(names.size()); for (auto& [name, streamId] : names) { - auto keys = streamKeys.find(streamId); + auto ranges = streamRanges.find(streamId); auto minVersion = minVersions.find(streamId); - if (keys != streamKeys.end() && minVersion != minVersions.end()) { - result.push_back( - NativeCdcStreamInfo{ std::move(name), streamId, keys->second, minVersion->second }); + if (ranges != streamRanges.end() && minVersion != minVersions.end()) { + result.push_back(NativeCdcStreamInfo{ + std::move(name), streamId, std::move(ranges->second), minVersion->second }); } } co_return result; @@ -679,51 +362,6 @@ Future> listNativeCdcStreams(Database cx) { } } -Future reassignNativeCdcStreams(Database cx, UID oldProxyId, UID newProxyId) { - if (oldProxyId == newProxyId) { - co_return; - } - - Transaction tr(cx); - while (true) { - Error err; - try { - tr.setOption(FDBTransactionOptions::LOCK_AWARE); - tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); - tr.setOption(FDBTransactionOptions::PRIORITY_SYSTEM_IMMEDIATE); - - bool changed = false; - Key begin = cdcProxyKeys.begin; - while (begin < cdcProxyKeys.end) { - RangeResult assignments = - co_await tr.getRange(KeyRangeRef(begin, cdcProxyKeys.end), CLIENT_KNOBS->TOO_MANY); - for (const auto& assignment : assignments) { - const auto [streamId, proxyId] = decodeCDCProxyKey(assignment.key); - if (proxyId == oldProxyId) { - tr.clear(assignment.key); - tr.set(cdcProxyKeyFor(streamId, newProxyId), Value()); - changed = true; - } - } - if (!assignments.more) { - break; - } - begin = keyAfter(assignments.back().key); - } - - if (changed) { - CODE_PROBE(true, "Native CDC reassigns streams after proxy replacement"); - signalNativeCdcProxyAssignmentChange(&tr); - co_await tr.commit(); - } - co_return; - } catch (Error& e) { - err = e; - } - co_await tr.onError(err); - } -} - Future acknowledgeNativeCdcStream(Database cx, CDCStreamId streamId, Version consumedThrough, @@ -768,8 +406,8 @@ Future acknowledgeNativeCdcStream(Database cx, } } -Future registerNativeCdcStreamClient(Database cx, Key name, KeyRange keys) { - validateNativeCdcStream(name, keys); +Future registerNativeCdcStreamClient(Database cx, Key name, std::vector ranges) { + normalizeNativeCdcStreamRanges(name, ranges); Optional previousProxy; while (true) { Future proxyChanged = cx->clientInfo->onChange(); @@ -799,7 +437,7 @@ Future registerNativeCdcStreamClient(Database cx, Key name, KeyRang CDCProxyInterface proxy = selectedProxy.get(); try { Future> request = - proxy.registerStream.tryGetReply(CDCRegisterStreamRequest(name, keys)); + proxy.registerStream.tryGetReply(CDCRegisterStreamRequest(name, ranges)); // Assignment publications for other streams also change ClientDBInfo. Keep this request alive while its // proxy remains published; abandoning it can let a server-side retry recreate the stream after removal. while (true) { @@ -860,7 +498,7 @@ Future getNativeCdcStatus(Database cx) { stream.info.name = decodeCDCStreamNameKey(entry.key); } for (const auto& entry : metadata[1].get()) { - streams[decodeCDCStreamKey(entry.key)].info.keys = decodeCDCStreamKeysValue(entry.value); + streams[decodeCDCStreamKey(entry.key)].info.ranges = decodeCDCStreamKeysValue(entry.value); } for (const auto& entry : metadata[2].get()) { streams[decodeCDCMinVersionKey(entry.key)].info.minVersion = decodeCDCMinVersionValue(entry.value); @@ -884,8 +522,8 @@ Future getNativeCdcStatus(Database cx) { stream.info.streamId = streamId; std::sort(stream.tags.begin(), stream.tags.end()); stream.tags.erase(std::unique(stream.tags.begin(), stream.tags.end()), stream.tags.end()); - if (stream.info.name.empty() || stream.info.keys.empty() || stream.info.minVersion == invalidVersion || - stream.tags.empty()) { + if (stream.info.name.empty() || stream.info.ranges.empty() || + stream.info.minVersion == invalidVersion || stream.tags.empty()) { result.metadataComplete = false; } for (const Tag& tag : stream.tags) { @@ -1021,8 +659,8 @@ Future NativeCdcConsumer::consumeImpl(ReferencecurrentPosition))); + CDCConsumeReply reply = co_await throwErrorOr( + proxy.consume.tryGetReply(CDCConsumeRequest(self->currentPosition, self->consumerId))); if (reply.lastConsumedVersion == self->currentPosition.lastConsumedVersion && reply.mutations.empty()) { // The server lease bounds abandoned long polls. Renew it transparently so the public consume // operation remains a long poll without accumulating server actors after client cancellation. @@ -1103,40 +741,75 @@ Future NativeCdcConsumer::acknowledge() { return acknowledgeImpl(Reference::addRef(this)); } -TEST_CASE("/NativeCDC/LifecycleAllocation") { - ASSERT(!validNativeCdcTagCount(-1)); - ASSERT(!validNativeCdcTagCount(0)); - ASSERT(validNativeCdcTagCount(1)); - ASSERT(validNativeCdcTagCount(std::numeric_limits::max() + 1u)); - ASSERT(!validNativeCdcTagCount(std::numeric_limits::max() + 2u)); - - NativeCdcIdentifierAllocator allocator; - auto [initialId, initialTag] = allocator.allocate(CLIENT_KNOBS->NATIVE_CDC_TAG_COUNT); - ASSERT_EQ(initialId, 1); - ASSERT_EQ(initialTag, Tag(tagLocalityCDC, 0)); - - allocator.observeStreamId(9); - allocator.observeTag(initialTag); - allocator.observeTag(Tag(tagLocalityCDC, 2)); - auto [nextId, nextTag] = allocator.allocate(CLIENT_KNOBS->NATIVE_CDC_TAG_COUNT); - ASSERT_EQ(nextId, 10); - ASSERT_EQ(nextTag, Tag(tagLocalityCDC, 1)); - - NativeCdcIdentifierAllocator publishedPoolAllocator; - publishedPoolAllocator.observeTag(Tag(tagLocalityCDC, 0)); - // The cluster-controller-published pool is authoritative even when it differs from this process's knob. - auto [publishedPoolId, publishedPoolTag] = publishedPoolAllocator.allocate(1); - ASSERT_EQ(publishedPoolId, 1); - ASSERT_EQ(publishedPoolTag, Tag(tagLocalityCDC, 0)); - - NativeCdcIdentifierAllocator fullPoolAllocator; - for (uint32_t tagId = 0; tagId < static_cast(CLIENT_KNOBS->NATIVE_CDC_TAG_COUNT); ++tagId) { - fullPoolAllocator.observeTag(Tag(tagLocalityCDC, static_cast(tagId))); - } - auto [sharedId, sharedTag] = fullPoolAllocator.allocate(CLIENT_KNOBS->NATIVE_CDC_TAG_COUNT); - ASSERT_EQ(sharedId, 1); - ASSERT_EQ(sharedTag, Tag(tagLocalityCDC, 0)); +TEST_CASE("/NativeCDC/RangeNormalization") { + std::vector ranges{ + KeyRangeRef("x"_sr, "z"_sr), KeyRangeRef("b"_sr, "d"_sr), KeyRangeRef("a"_sr, "b"_sr), + KeyRangeRef("a"_sr, "c"_sr), KeyRangeRef("b"_sr, "c"_sr), KeyRangeRef("x"_sr, "z"_sr) + }; + const std::vector expected{ KeyRangeRef("a"_sr, "d"_sr), KeyRangeRef("x"_sr, "z"_sr) }; + normalizeNativeCdcStreamRanges("orders"_sr, ranges); + ASSERT(ranges == expected); + normalizeNativeCdcStreamRanges("orders"_sr, ranges); + ASSERT(ranges == expected); + + std::vector entireKeyspace{ normalKeys }; + normalizeNativeCdcStreamRanges("all"_sr, entireKeyspace); + ASSERT(entireKeyspace == std::vector{ normalKeys }); + + constexpr int singletonCount = 10; + const int keyLength = CLIENT_KNOBS->VALUE_SIZE_LIMIT / (2LL * singletonCount) + 1; + ASSERT_LT(keyLength, CLIENT_KNOBS->KEY_SIZE_LIMIT); + std::vector singletonRanges; + int64_t endpointBytes = 0; + for (int i = 0; i < singletonCount; ++i) { + const std::string prefix = format("%04d/", i); + ASSERT_GT(keyLength, prefix.size()); + const Key key(StringRef(prefix + std::string(keyLength - prefix.size(), 'x'))); + singletonRanges.push_back(singleKeyRange(key)); + endpointBytes += static_cast(key.size()) + key.size() + 1; + } + ASSERT_GT(endpointBytes, CLIENT_KNOBS->VALUE_SIZE_LIMIT); + ASSERT_LE(cdcStreamKeysValue(singletonRanges).size(), CLIENT_KNOBS->VALUE_SIZE_LIMIT); + const auto expectedSingletons = singletonRanges; + normalizeNativeCdcStreamRanges("singletons"_sr, singletonRanges); + ASSERT(singletonRanges == expectedSingletons); + return Void(); +} +TEST_CASE("/NativeCDC/InvalidRanges") { + auto expectInvalid = [](KeyRef name, std::vector ranges) { + try { + normalizeNativeCdcStreamRanges(name, ranges); + } catch (Error& error) { + ASSERT_EQ(error.code(), error_code_client_invalid_operation); + return; + } + ASSERT(false); + }; + expectInvalid(KeyRef(), { normalKeys }); + expectInvalid("orders"_sr, {}); + expectInvalid("orders"_sr, { KeyRangeRef("a"_sr, "a"_sr) }); + expectInvalid("orders"_sr, { normalKeys, systemKeys }); + expectInvalid("orders"_sr, std::vector(NATIVE_CDC_MAX_RANGES + 1, normalKeys)); + + std::vector maximumCount; + std::vector oversizedMetadata; + const int endpointLength = CLIENT_KNOBS->VALUE_SIZE_LIMIT / (int64_t{ 2 } * NATIVE_CDC_MAX_RANGES); + for (int i = 0; i < NATIVE_CDC_MAX_RANGES; ++i) { + const std::string prefix = format("%04d/", i); + maximumCount.emplace_back(KeyRangeRef(prefix + "a", prefix + "z")); + ASSERT_GT(endpointLength, prefix.size()); + oversizedMetadata.emplace_back(KeyRangeRef(prefix + std::string(endpointLength - prefix.size(), 'a'), + prefix + std::string(endpointLength - prefix.size(), 'z'))); + } + normalizeNativeCdcStreamRanges("orders"_sr, maximumCount); + ASSERT_EQ(maximumCount.size(), NATIVE_CDC_MAX_RANGES); + ASSERT_LE(2 * NATIVE_CDC_MAX_RANGES * endpointLength, CLIENT_KNOBS->VALUE_SIZE_LIMIT); + ASSERT_GT(cdcStreamKeysValue(oversizedMetadata).size(), CLIENT_KNOBS->VALUE_SIZE_LIMIT); + expectInvalid("orders"_sr, oversizedMetadata); + const std::string oversizedBegin(CLIENT_KNOBS->VALUE_SIZE_LIMIT, 'a'); + const std::string oversizedEnd(CLIENT_KNOBS->VALUE_SIZE_LIMIT, 'b'); + expectInvalid("orders"_sr, { KeyRangeRef(oversizedBegin, oversizedEnd) }); return Void(); } diff --git a/fdbclient/NativeCdcInternal.h b/fdbclient/NativeCdcInternal.h index ae52ee32fe1..ec2a2c521de 100644 --- a/fdbclient/NativeCdcInternal.h +++ b/fdbclient/NativeCdcInternal.h @@ -24,15 +24,12 @@ #include "fdbclient/NativeCdc.h" -// Durable metadata operations used by CDC server roles. Registration is -// feature gated; drain and cleanup operations remain available for streams -// persisted before native CDC is disabled. -Future registerNativeCdcStream(Database cx, Key name, KeyRange keys, UID proxyId); -// Persists per-tag final-pop watermarks before removing stream metadata. -Future removeNativeCdcStream(Database cx, Key name, CDCStreamId streamId, UID proxyId); +// Shared admission and metadata identity checks for native CDC operations. +void validateNativeCdcEnabled(bool enabled); +void normalizeNativeCdcStreamRanges(KeyRef const& name, std::vector& ranges); +bool nativeCdcNameMatchesStream(Optional const& currentId, CDCStreamId streamId); + Future> listNativeCdcStreams(Database cx); -// Atomically moves any streams assigned to a failed proxy to its replacement. -Future reassignNativeCdcStreams(Database cx, UID oldProxyId, UID newProxyId); // Persists the exclusive unpopped watermark after consuming through a version. // knownAvailableThrough permits a consumer to acknowledge log data it has // already received before that version is visible at a transaction read version. diff --git a/fdbclient/RYWIterator.cpp b/fdbclient/RYWIterator.cpp index c420a479c89..426940621df 100644 --- a/fdbclient/RYWIterator.cpp +++ b/fdbclient/RYWIterator.cpp @@ -424,6 +424,48 @@ static int getWriteMapCount(WriteMap* p) { return count; } +TEST_CASE("/fdbclient/ExtStringRef/materialization") { + struct MaterializationCase { + StringRef base; + int padding; + StringRef expected; + }; + const MaterializationCase cases[] = { + { ""_sr, 0, ""_sr }, + { ""_sr, 3, "\x00\x00\x00"_sr }, + { "a\x00z"_sr, 0, "a\x00z"_sr }, + { "a\x00z"_sr, 1, "a\x00z\x00"_sr }, + { "a\x00z"_sr, 3, "a\x00z\x00\x00\x00"_sr }, + }; + for (const auto& test : cases) { + Standalone base(test.base); + ExtStringRef extended(base, test.padding); + Arena ownedArena; + Arena borrowArena; + StringRef owned = extended.toArena(ownedArena); + StringRef borrowedOrOwned = extended.toArenaOrRef(borrowArena); + Standalone standalone = extended.toStandaloneStringRef(); + ASSERT(owned == test.expected); + ASSERT(borrowedOrOwned == test.expected); + ASSERT(standalone == test.expected); + if (test.padding == 0) { + ASSERT(borrowedOrOwned.begin() == base.begin()); + ASSERT(borrowArena.getSize() == 0); + } + if (!base.empty()) { + mutateString(base)[0] = 'b'; + ASSERT(owned == test.expected); + ASSERT(standalone == test.expected); + if (test.padding == 0) { + ASSERT(borrowedOrOwned == base); + } else { + ASSERT(borrowedOrOwned == test.expected); + } + } + } + return Void(); +} + TEST_CASE("/fdbclient/WriteMap/emptiness") { Arena arena = Arena(); WriteMap writes = WriteMap(&arena); diff --git a/fdbclient/S3Client.cpp b/fdbclient/S3Client.cpp index f4392aecc44..81205a235ec 100644 --- a/fdbclient/S3Client.cpp +++ b/fdbclient/S3Client.cpp @@ -166,7 +166,7 @@ Reference getEndpoint(const std::string& s3url, return endpoint; } catch (Error& e) { - TraceEvent(SevError, "S3ClientGetEndpointFailed").detail("URL", StringRef(s3url)).detail("Error", e.what()); + TraceEvent(SevError, "S3ClientGetEndpointFailed").error(e).detail("URL", StringRef(s3url)); throw; } } @@ -1112,7 +1112,7 @@ Future listFiles(std::string s3url, int maxDepth) { } } } catch (Error& e) { - TraceEvent(SevError, "S3ClientListFilesError").detail("URL", s3url).detail("Error", e.what()); + TraceEvent(SevError, "S3ClientListFilesError").error(e).detail("URL", s3url); if (e.code() == error_code_backup_invalid_url) { std::cerr << "ERROR: Invalid blobstore URL: " << s3url << std::endl; } else if (e.code() == error_code_backup_auth_missing) { diff --git a/fdbclient/SnapshotCache.h b/fdbclient/SnapshotCache.h index cdd85e14332..e5ad588b6f0 100644 --- a/fdbclient/SnapshotCache.h +++ b/fdbclient/SnapshotCache.h @@ -34,21 +34,13 @@ struct ExtStringRef { Standalone toStandaloneStringRef() const { auto s = makeString(size()); - if (!base.empty()) { - memcpy(mutateString(s), base.begin(), base.size()); - } - memset(mutateString(s) + base.size(), 0, extra_zero_bytes); + copyTo(mutateString(s)); return s; }; StringRef toArenaOrRef(Arena& a) const { if (extra_zero_bytes) { - StringRef dest = StringRef(new (a) uint8_t[size()], size()); - if (!base.empty()) { - memcpy(mutateString(dest), base.begin(), base.size()); - } - memset(mutateString(dest) + base.size(), 0, extra_zero_bytes); - return dest; + return toArena(a); } else { return base; } @@ -62,10 +54,7 @@ struct ExtStringRef { StringRef toArena(Arena& a) const { if (extra_zero_bytes) { StringRef dest = StringRef(new (a) uint8_t[size()], size()); - if (!base.empty()) { - memcpy(mutateString(dest), base.begin(), base.size()); - } - memset(mutateString(dest) + base.size(), 0, extra_zero_bytes); + copyTo(mutateString(dest)); return dest; } else { return StringRef(a, base); @@ -115,6 +104,13 @@ struct ExtStringRef { ExtStringRef keyAfter() const { return ExtStringRef(base, extra_zero_bytes + 1); } private: + void copyTo(uint8_t* dest) const { + if (!base.empty()) { + memcpy(dest, base.begin(), base.size()); + } + memset(dest + base.size(), 0, extra_zero_bytes); + } + friend struct Traceable; StringRef base; int extra_zero_bytes; diff --git a/fdbclient/SpecialKeySpace.cpp b/fdbclient/SpecialKeySpace.cpp index 151113592ae..44f548afae8 100644 --- a/fdbclient/SpecialKeySpace.cpp +++ b/fdbclient/SpecialKeySpace.cpp @@ -1628,25 +1628,27 @@ Future GlobalConfigImpl::getRange(ReadYourWritesTransaction* ryw, RangeResult result; KeyRangeRef modified = KeyRangeRef(kr.begin.removePrefix(getKeyRange().begin), kr.end.removePrefix(getKeyRange().begin)); - std::map> values = ryw->getDatabase()->globalConfig->get(modified); + std::map> values = ryw->getDatabase()->globalConfig->get(modified); for (const auto& [key, config] : values) { Key prefixedKey = key.withPrefix(getKeyRange().begin); - if (config.isValid() && config->value.has_value()) { - if (config->value.type() == typeid(StringRef)) { - result.push_back_deep(result.arena(), - KeyValueRef(prefixedKey, std::any_cast(config->value).toString())); - } else if (config->value.type() == typeid(int64_t)) { - result.push_back_deep(result.arena(), - KeyValueRef(prefixedKey, std::to_string(std::any_cast(config->value)))); - } else if (config->value.type() == typeid(bool)) { - result.push_back_deep(result.arena(), - KeyValueRef(prefixedKey, std::to_string(std::any_cast(config->value)))); - } else if (config->value.type() == typeid(float)) { - result.push_back_deep(result.arena(), - KeyValueRef(prefixedKey, std::to_string(std::any_cast(config->value)))); - } else if (config->value.type() == typeid(double)) { - result.push_back_deep(result.arena(), - KeyValueRef(prefixedKey, std::to_string(std::any_cast(config->value)))); + if (config.isValid() && config->getValue().has_value()) { + if (config->getValue().type() == typeid(StringRef)) { + result.push_back_deep( + result.arena(), KeyValueRef(prefixedKey, std::any_cast(config->getValue()).toString())); + } else if (config->getValue().type() == typeid(int64_t)) { + result.push_back_deep( + result.arena(), + KeyValueRef(prefixedKey, std::to_string(std::any_cast(config->getValue())))); + } else if (config->getValue().type() == typeid(bool)) { + result.push_back_deep( + result.arena(), KeyValueRef(prefixedKey, std::to_string(std::any_cast(config->getValue())))); + } else if (config->getValue().type() == typeid(float)) { + result.push_back_deep( + result.arena(), KeyValueRef(prefixedKey, std::to_string(std::any_cast(config->getValue())))); + } else if (config->getValue().type() == typeid(double)) { + result.push_back_deep( + result.arena(), + KeyValueRef(prefixedKey, std::to_string(std::any_cast(config->getValue())))); } else { ASSERT(false); } diff --git a/fdbclient/StorageServerInterface.cpp b/fdbclient/StorageServerInterface.cpp index e7019c5bbb3..a58e6e78d18 100644 --- a/fdbclient/StorageServerInterface.cpp +++ b/fdbclient/StorageServerInterface.cpp @@ -97,6 +97,10 @@ void StorageServerInterface::initEndpoints() { streams.push_back(getCheckSum.getReceiver()); streams.push_back(bulkdump.getReceiver()); FlowTransport::transport().addEndpoints(streams); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + // streams[0] is `getValue` (base endpoint); streams[1..kNumAdjustedEndpoints] are adjusted endpoints. + ASSERT(streams.size() - 1 == kNumAdjustedEndpoints); + } } // if size + hex of checksum is shorter than value, record that instead of actual value. break-even point is 12 diff --git a/fdbclient/SystemData.cpp b/fdbclient/SystemData.cpp index 0ea881e10d4..7b9e742356b 100644 --- a/fdbclient/SystemData.cpp +++ b/fdbclient/SystemData.cpp @@ -839,18 +839,18 @@ CDCStreamId decodeCDCStreamKey(KeyRef const& key) { return streamId; } -Value cdcStreamKeysValue(KeyRangeRef const& keys) { +Value cdcStreamKeysValue(std::vector const& ranges) { BinaryWriter wr(IncludeVersion(ProtocolVersion::withNativeCdc())); - wr << keys; + wr << ranges; return wr.toValue(); } -KeyRange decodeCDCStreamKeysValue(ValueRef const& value) { - KeyRange keys; +std::vector decodeCDCStreamKeysValue(ValueRef const& value) { + std::vector ranges; BinaryReader reader(value, IncludeVersion()); ASSERT_WE_THINK(reader.protocolVersion().hasNativeCdc()); - reader >> keys; - return keys; + reader >> ranges; + return ranges; } static Key cdcTagHistoryPrefixFor(CDCStreamId streamId) { @@ -885,6 +885,21 @@ CDCTagHistoryEntry decodeCDCTagHistoryKey(KeyRef const& key) { return CDCTagHistoryEntry(streamId, bigEndian64(encodedVersion), tag); } +CDCTagHistoryEntry decodeCDCTagHistoryEntry(KeyRef const& key, ValueRef const& value) { + CDCTagHistoryEntry result = decodeCDCTagHistoryKey(key); + if (!value.empty()) { + if (value.size() != sizeof(Version) + sizeof(uint16_t)) { + throw serialization_failed(); + } + const Version committedVersion = decodeCDCMinVersionValue(value); + if (committedVersion <= result.version) { + throw serialization_failed(); + } + result.version = committedVersion; + } + return result; +} + Key cdcTagOwnerKeyFor(Tag tag) { BinaryWriter wr(Unversioned()); wr.serializeBytes(cdcTagOwnerKeys.begin); @@ -1944,7 +1959,7 @@ TEST_CASE("noSim/SystemData/DataMoveId") { TEST_CASE("/SystemData/NativeCDC") { const Key name = "orders"_sr; const CDCStreamId streamId = 42; - const KeyRange keys(KeyRangeRef("a"_sr, "z"_sr)); + const std::vector ranges{ KeyRangeRef("a"_sr, "c"_sr), KeyRangeRef("x"_sr, "z"_sr) }; const Version minVersion = 123456789; const Tag tag(tagLocalityCDC, 9); const UID proxyId(1, 2); @@ -1953,7 +1968,7 @@ TEST_CASE("/SystemData/NativeCDC") { ASSERT_EQ(decodeCDCStreamNameValue(cdcStreamNameValue(streamId)), streamId); ASSERT_EQ(decodeCDCMaxStreamIdValue(cdcMaxStreamIdValue(streamId)), streamId); ASSERT_EQ(decodeCDCStreamKey(cdcStreamKeyFor(streamId)), streamId); - ASSERT_EQ(decodeCDCStreamKeysValue(cdcStreamKeysValue(keys)), keys); + ASSERT(decodeCDCStreamKeysValue(cdcStreamKeysValue(ranges)) == ranges); const Key tagOwnerKey = cdcTagOwnerKeyFor(tag); ASSERT_EQ(decodeCDCTagOwnerKey(tagOwnerKey), tag); ASSERT(cdcTagOwnerKeys.contains(tagOwnerKey)); @@ -1981,6 +1996,20 @@ TEST_CASE("/SystemData/NativeCDC") { const Key laterTagHistoryKey = cdcTagHistoryKeyFor(streamId, 256, Tag(tagLocalityCDC, 0)); ASSERT(earlierTagHistoryKey < laterTagHistoryKey); ASSERT(cdcTagHistoryRangeFor(streamId).contains(laterTagHistoryKey)); + ASSERT_EQ(decodeCDCTagHistoryEntry(tagHistoryKey, ValueRef()).version, minVersion); + const Value committedBoundary = BinaryWriter::toValue(Versionstamp(minVersion + 20, 3), Unversioned()); + const CDCTagHistoryEntry committedHistory = decodeCDCTagHistoryEntry(tagHistoryKey, committedBoundary); + ASSERT_EQ(committedHistory.version, minVersion + 20); + ASSERT_EQ(committedHistory.tag, tag); + ASSERT_EQ(committedHistory.streamId, streamId); + bool invalidBoundaryRejected = false; + try { + decodeCDCTagHistoryEntry(tagHistoryKey, BinaryWriter::toValue(Versionstamp(minVersion, 0), Unversioned())); + } catch (Error& e) { + ASSERT_EQ(e.code(), error_code_serialization_failed); + invalidBoundaryRejected = true; + } + ASSERT(invalidBoundaryRejected); const Value serializedTagHistory = ObjectWriter::toValue(decodedTagHistory, Unversioned()); const auto deserializedTagHistory = diff --git a/fdbclient/ThreadSafeTransaction.cpp b/fdbclient/ThreadSafeTransaction.cpp index 82428c4ad29..98272a25016 100644 --- a/fdbclient/ThreadSafeTransaction.cpp +++ b/fdbclient/ThreadSafeTransaction.cpp @@ -221,13 +221,13 @@ ThreadFuture ThreadSafeDatabase::createSnapshot(const StringRef& uid, cons }); } -ThreadFuture ThreadSafeDatabase::registerNativeCdcStream(const KeyRef& name, const KeyRangeRef& keys) { +ThreadFuture ThreadSafeDatabase::registerNativeCdcStream(const KeyRef& name, + const std::vector& ranges) { DatabaseContext* db = this->db; Key nameCopy(name); - KeyRange keysCopy(keys); - return onMainThread([db, nameCopy, keysCopy]() -> Future { + return onMainThread([db, nameCopy, ranges]() -> Future { db->checkDeferredError(); - return registerNativeCdcStreamClient(Database(Reference::addRef(db)), nameCopy, keysCopy); + return registerNativeCdcStreamClient(Database(Reference::addRef(db)), nameCopy, ranges); }); } diff --git a/fdbclient/Tracing.cpp b/fdbclient/Tracing.cpp index 47b5ca3f353..f7ad960a6db 100644 --- a/fdbclient/Tracing.cpp +++ b/fdbclient/Tracing.cpp @@ -69,7 +69,7 @@ struct LogfileTracer : ITracer { for (const auto& event : span.events) { TraceEvent(SevInfo, "TracingSpanEvent", span.context.traceID) .detail("Name", event.name) - .detail("Time", event.time); + .detail("SpanEventTime", event.time); for (const auto& [key, value] : event.attributes) { TraceEvent(SevInfo, "TracingSpanEventAttribute", span.context.traceID) .detail("Key", key) diff --git a/fdbclient/include/fdbclient/BackupContainer.h b/fdbclient/include/fdbclient/BackupContainer.h index 1a7c3173440..b8d56d49148 100644 --- a/fdbclient/include/fdbclient/BackupContainer.h +++ b/fdbclient/include/fdbclient/BackupContainer.h @@ -436,7 +436,8 @@ class RangeMapFilters { // Accumulates mutation log value chunks, as both a vector of chunks and as a combined chunk, // in chunk order, and can check the chunk set for completion or intersection with a set // of ranges. -struct AccumulatedMutations { +class AccumulatedMutations { +public: AccumulatedMutations() : lastChunkNumber(-1) {} // Add a KV pair for this mutation chunk set @@ -450,11 +451,19 @@ struct AccumulatedMutations { // that matches the bytes after the header in the combined value in serializedMutations bool isComplete() const; + // Returns the complete serialized payload, or no value if the chunk set is incomplete. + // The returned bytes remain valid until this accumulator is modified or destroyed. + Optional getCompleteMutations() const; + + // The key and value bytes remain owned by the inputs passed to addChunk(). + const std::vector& getChunks() const { return kvs; } + // Returns true if a complete chunk contains any MutationRefs which intersect with any // range in ranges. // It is undefined behavior to run this if isComplete() does not return true. bool matchesAnyRange(const RangeMapFilters& rangeMap) const; +private: std::vector kvs; std::string serializedMutations; int lastChunkNumber; diff --git a/fdbclient/include/fdbclient/CDCProxyInterface.h b/fdbclient/include/fdbclient/CDCProxyInterface.h index 8ed419d0dd6..4ecc7af3d6f 100644 --- a/fdbclient/include/fdbclient/CDCProxyInterface.h +++ b/fdbclient/include/fdbclient/CDCProxyInterface.h @@ -72,17 +72,17 @@ struct CDCRegisterStreamReply { struct CDCRegisterStreamRequest { constexpr static FileIdentifier file_identifier = 1269096; Key name; - KeyRange keys; + std::vector ranges; ReplyPromise reply; CDCRegisterStreamRequest() = default; - CDCRegisterStreamRequest(Key name, KeyRange keys) : name(name), keys(keys) {} + CDCRegisterStreamRequest(Key name, std::vector ranges) : name(name), ranges(std::move(ranges)) {} bool verify() const { return true; } template void serialize(Ar& ar) { - serializer(ar, name, keys, reply); + serializer(ar, name, ranges, reply); } }; @@ -119,15 +119,18 @@ struct CDCConsumeRequest { constexpr static FileIdentifier file_identifier = 8178243; CDCCursor cursor; ReplyPromise reply; + // Stable across one consumer's RPC retries; absent for legacy or direct callers. + Optional consumerId; CDCConsumeRequest() = default; - explicit CDCConsumeRequest(CDCCursor cursor) : cursor(cursor) {} + explicit CDCConsumeRequest(CDCCursor cursor, Optional consumerId = {}) + : cursor(cursor), consumerId(consumerId) {} bool verify() const { return true; } template void serialize(Ar& ar) { - serializer(ar, cursor, reply); + serializer(ar, cursor, reply, consumerId); } }; diff --git a/fdbclient/include/fdbclient/CommitProxyInterface.h b/fdbclient/include/fdbclient/CommitProxyInterface.h index 64ee80d50e1..e497a0e662d 100644 --- a/fdbclient/include/fdbclient/CommitProxyInterface.h +++ b/fdbclient/include/fdbclient/CommitProxyInterface.h @@ -36,6 +36,7 @@ struct CommitProxyInterface { constexpr static FileIdentifier file_identifier = 8954922; + constexpr static int kNumAdjustedEndpoints = 10; enum { LocationAwareLoadBalance = 1 }; enum { AlwaysFresh = 1 }; @@ -65,6 +66,17 @@ struct CommitProxyInterface { NetworkAddress address() const { return commit.getEndpoint().getPrimaryAddress(); } NetworkAddressList addresses() const { return commit.getEndpoint().addresses; } + std::vector getEndpointTokens() const { + // Token at index 0 is the base `commit` endpoint; adjusted endpoints start at 1. + std::vector tokens; + tokens.reserve(kNumAdjustedEndpoints + 1); + tokens.push_back(commit.getEndpoint().token); + for (int i = 1; i <= kNumAdjustedEndpoints; ++i) { + tokens.push_back(commit.getEndpoint().getAdjustedEndpoint(i).token); + } + return tokens; + } + template void serialize(Archive& ar) { serializer(ar, processId, provisional, commit); @@ -85,6 +97,10 @@ struct CommitProxyInterface { PublicRequestStream(commit.getEndpoint().getAdjustedEndpoint(9)); setThrottledShard = RequestStream(commit.getEndpoint().getAdjustedEndpoint(10)); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && g_network && g_network->global(INetwork::enFlowTransport)) { + FlowTransport::transport().interfaceTracker.created( + commit.getEndpoint().getPrimaryAddress(), "CP", getEndpointTokens()); + } } } @@ -103,6 +119,10 @@ struct CommitProxyInterface { streams.push_back(expireIdempotencyId.getReceiver()); streams.push_back(setThrottledShard.getReceiver()); FlowTransport::transport().addEndpoints(streams); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + // streams[0] is `commit` (base endpoint); streams[1..kNumAdjustedEndpoints] are adjusted endpoints. + ASSERT(streams.size() - 1 == kNumAdjustedEndpoints); + } } }; diff --git a/fdbclient/include/fdbclient/DataDistributionConfig.h b/fdbclient/include/fdbclient/DataDistributionConfig.h index 0176cb8c40b..f6d6509de76 100644 --- a/fdbclient/include/fdbclient/DataDistributionConfig.h +++ b/fdbclient/include/fdbclient/DataDistributionConfig.h @@ -37,7 +37,8 @@ struct DDRangeConfig { constexpr static FileIdentifier file_identifier = 9193856; - explicit(false) DDRangeConfig(Optional replicationFactor = {}, Optional teamID = {}) + DDRangeConfig() = default; + explicit DDRangeConfig(Optional replicationFactor, Optional teamID = {}) : replicationFactor(replicationFactor), teamID(teamID) {} Optional replicationFactor; diff --git a/fdbclient/include/fdbclient/DatabaseContext.h b/fdbclient/include/fdbclient/DatabaseContext.h index b4827051bce..2bcf3c15127 100644 --- a/fdbclient/include/fdbclient/DatabaseContext.h +++ b/fdbclient/include/fdbclient/DatabaseContext.h @@ -31,6 +31,7 @@ #include #include #include +#include #pragma once #include "fdbclient/FDBTypes.h" @@ -422,6 +423,10 @@ class DatabaseContext : public ReferenceCounted, public FastAll std::map server_interf; + // Periodically samples FlowTransport per-address connect-failed counts and evicts + // location-cache entries for any address whose count advanced (a dead/flapping peer). + Future locationCachePeerEvictor; + // map from ssid -> tss interface std::unordered_map tssMapping; // map from tssid -> metrics for that tss pair diff --git a/fdbclient/include/fdbclient/GlobalConfig.h b/fdbclient/include/fdbclient/GlobalConfig.h index 3d15d0f1dd9..bd81f8c75e3 100644 --- a/fdbclient/include/fdbclient/GlobalConfig.h +++ b/fdbclient/include/fdbclient/GlobalConfig.h @@ -75,12 +75,15 @@ extern const KeyRef samplingWindow; // Structure used to hold the values stored by global configuration. The arena // is used as memory to store both the key and the value (the value is only // stored in the arena if it is an object; primitives are just copied). -struct ConfigValue : ReferenceCounted { +class ConfigValue : public ReferenceCounted { Arena arena; std::any value; +public: ConfigValue() = default; ConfigValue(Arena&& a, std::any&& v) : arena(a), value(v) {} + + const std::any& getValue() const { return value; } }; class GlobalConfig : NonCopyable { @@ -122,8 +125,8 @@ class GlobalConfig : NonCopyable { // reference which also contains the arena holding the object. As long as // the caller keeps the ConfigValue reference, the value is guaranteed to // be readable. An empty reference is returned if the value does not exist. - Reference get(KeyRef name); - std::map> get(KeyRangeRef range); + Reference get(KeyRef name); + std::map> get(KeyRangeRef range); // For arithmetic value types, returns a copy of the value for the given // key, or the supplied default value if the framework does not know about @@ -134,8 +137,8 @@ class GlobalConfig : NonCopyable { try { auto configValue = get(name); if (configValue.isValid()) { - if (configValue->value.has_value()) { - return std::any_cast(configValue->value); + if (configValue->getValue().has_value()) { + return std::any_cast(configValue->getValue()); } } @@ -195,7 +198,7 @@ class GlobalConfig : NonCopyable { Future _updater; Promise initialized; AsyncTrigger configChanged; - std::unordered_map> data; + std::unordered_map> data; Version lastUpdate; // The key should be a global config string literal key (see the top of this file). std::unordered_map)>> callbacks; diff --git a/fdbclient/include/fdbclient/GrvProxyInterface.h b/fdbclient/include/fdbclient/GrvProxyInterface.h index 3c57708ce75..5523f65a2a5 100644 --- a/fdbclient/include/fdbclient/GrvProxyInterface.h +++ b/fdbclient/include/fdbclient/GrvProxyInterface.h @@ -218,6 +218,7 @@ struct GlobalConfigRefreshRequest { // information of the cluster, and handles proxied GlobalConfig requests. struct GrvProxyInterface { constexpr static FileIdentifier file_identifier = 8743216; + constexpr static int kNumAdjustedEndpoints = 3; enum { LocationAwareLoadBalance = 1 }; enum { AlwaysFresh = 1 }; @@ -240,6 +241,17 @@ struct GrvProxyInterface { NetworkAddress address() const { return getConsistentReadVersion.getEndpoint().getPrimaryAddress(); } NetworkAddressList addresses() const { return getConsistentReadVersion.getEndpoint().addresses; } + std::vector getEndpointTokens() const { + // Token at index 0 is the base `getConsistentReadVersion` endpoint; adjusted endpoints start at 1. + std::vector tokens; + tokens.reserve(kNumAdjustedEndpoints + 1); + tokens.push_back(getConsistentReadVersion.getEndpoint().token); + for (int i = 1; i <= kNumAdjustedEndpoints; ++i) { + tokens.push_back(getConsistentReadVersion.getEndpoint().getAdjustedEndpoint(i).token); + } + return tokens; + } + template void serialize(Archive& ar) { serializer(ar, processId, provisional, getConsistentReadVersion); @@ -250,6 +262,10 @@ struct GrvProxyInterface { getConsistentReadVersion.getEndpoint().getAdjustedEndpoint(2)); refreshGlobalConfig = PublicRequestStream( getConsistentReadVersion.getEndpoint().getAdjustedEndpoint(3)); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && g_network && g_network->global(INetwork::enFlowTransport)) { + FlowTransport::transport().interfaceTracker.created( + getConsistentReadVersion.getEndpoint().getPrimaryAddress(), "GP", getEndpointTokens()); + } } } @@ -260,6 +276,11 @@ struct GrvProxyInterface { streams.push_back(getHealthMetrics.getReceiver()); streams.push_back(refreshGlobalConfig.getReceiver()); FlowTransport::transport().addEndpoints(streams); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + // streams[0] is `getConsistentReadVersion` (base endpoint); streams[1..kNumAdjustedEndpoints] are adjusted + // endpoints. + ASSERT(streams.size() - 1 == kNumAdjustedEndpoints); + } } }; diff --git a/fdbclient/include/fdbclient/IClientApi.h b/fdbclient/include/fdbclient/IClientApi.h index c669966346b..ce44a8ee9ed 100644 --- a/fdbclient/include/fdbclient/IClientApi.h +++ b/fdbclient/include/fdbclient/IClientApi.h @@ -157,7 +157,8 @@ class IDatabase { // Native CDC operations. These values are intentionally independent from // NativeAPI so multi-version client wrappers can forward them without // depending on the native client implementation. - virtual ThreadFuture registerNativeCdcStream(const KeyRef& name, const KeyRangeRef& keys) = 0; + virtual ThreadFuture registerNativeCdcStream(const KeyRef& name, + const std::vector& ranges) = 0; virtual ThreadFuture removeNativeCdcStream(const KeyRef& name) = 0; virtual ThreadFuture> listNativeCdcStreams() = 0; virtual ThreadFuture> createNativeCdcConsumer(const KeyRef& name) = 0; diff --git a/fdbclient/include/fdbclient/KeyRangeMap.h b/fdbclient/include/fdbclient/KeyRangeMap.h index eda30829a2a..8b06de1e8e4 100644 --- a/fdbclient/include/fdbclient/KeyRangeMap.h +++ b/fdbclient/include/fdbclient/KeyRangeMap.h @@ -253,18 +253,18 @@ void insertCoalescedRange(Map, Metric>& } } -} // namespace KeyRangeMapImpl - -template -void CoalescedKeyRangeMap::insert(const KeyRangeRef& keys, const Val& value) { - KeyRangeMapImpl::insertCoalescedRange(this->map, this->mf, mapEnd, keys, value); -} - -template -void CoalescedKeyRangeMap::insert(const KeyRef& key, const Val& value) { +// Coalesces adjacent equal-valued ranges in memory. The transaction sequencing requirements of the +// database-backed krmSetRangeCoalescing operations do not apply to this synchronous update. +template +void insertCoalescedKey(Map, Metric>& map, + const MetricFunc& mf, + const Key& mapEnd, + const KeyRef& key, + const Val& value, + MakeKeyAfter makeKeyAfter) { ASSERT(key < mapEnd); - auto begin = RangeMap::map.lower_bound(key); + auto begin = map.lower_bound(key); auto end = begin; if (end->key == key) ++end; @@ -295,70 +295,40 @@ void CoalescedKeyRangeMap::insert(const KeyRef& key, co insertBegin = true; } - RangeMap::map.erase(begin, end); + map.erase(begin, end); if (insertEnd) { - MapPair p(keyAfter(key), endVal); - RangeMap::map.insert( - p, true, RangeMap::mf(p)); + MapPair p(makeKeyAfter(key), endVal); + map.insert(p, true, mf(p)); } if (insertBegin) { - MapPair p(key, value); - RangeMap::map.insert( - p, true, RangeMap::mf(p)); + MapPair p(key, value); + map.insert(p, true, mf(p)); } } +} // namespace KeyRangeMapImpl + template -void CoalescedKeyRefRangeMap::insert(const KeyRangeRef& keys, const Val& value) { +void CoalescedKeyRangeMap::insert(const KeyRangeRef& keys, const Val& value) { KeyRangeMapImpl::insertCoalescedRange(this->map, this->mf, mapEnd, keys, value); } template -void CoalescedKeyRefRangeMap::insert(const KeyRef& key, const Val& value, Arena& arena) { - ASSERT(key < mapEnd); - - auto begin = RangeMap::map.lower_bound(key); - auto end = begin; - if (end->key == key) - ++end; - - bool insertEnd = false; - bool insertBegin = false; - Val endVal; - - if (!equalsKeyAfter(key, end->key)) { - auto before_end = end; - before_end.decrementNonEnd(); - if (value != before_end->value) { - insertEnd = true; - endVal = before_end->value; - } - } - - if (!insertEnd && end->value == value && end->key != mapEnd) { - ++end; - } +void CoalescedKeyRangeMap::insert(const KeyRef& key, const Val& value) { + KeyRangeMapImpl::insertCoalescedKey( + this->map, this->mf, mapEnd, key, value, [](const KeyRef& key) -> Key { return keyAfter(key); }); +} - if (key == allKeys.begin) { - insertBegin = true; - } else { - auto before_begin = begin; - before_begin.decrementNonEnd(); - if (before_begin->value != value) - insertBegin = true; - } +template +void CoalescedKeyRefRangeMap::insert(const KeyRangeRef& keys, const Val& value) { + KeyRangeMapImpl::insertCoalescedRange(this->map, this->mf, mapEnd, keys, value); +} - RangeMap::map.erase(begin, end); - if (insertEnd) { - MapPair p(keyAfter(key, arena), endVal); - RangeMap::map.insert( - p, true, RangeMap::mf(p)); - } - if (insertBegin) { - MapPair p(key, value); - RangeMap::map.insert( - p, true, RangeMap::mf(p)); - } +template +void CoalescedKeyRefRangeMap::insert(const KeyRef& key, const Val& value, Arena& arena) { + KeyRangeMapImpl::insertCoalescedKey(this->map, this->mf, mapEnd, key, value, [&arena](const KeyRef& key) -> KeyRef { + return keyAfter(key, arena); + }); } #endif diff --git a/fdbclient/include/fdbclient/Knobs.h b/fdbclient/include/fdbclient/Knobs.h index 47468759a89..24b1dc89513 100644 --- a/fdbclient/include/fdbclient/Knobs.h +++ b/fdbclient/include/fdbclient/Knobs.h @@ -56,6 +56,18 @@ class SWIFT_CXX_IMMORTAL_SINGLETON_TYPE ClientKnobs : public KnobsImplcommitProxies/grvProxies immediately so a + // killed proxy's RequestStream is dropped on rotation rather than on the next transaction's + // lazy getCommitProxies()/getGrvProxies(). Off by default; randomized in simulation; + // StalePeerTest forces it on. + bool DBCONTEXT_EAGER_PROXY_UPDATE; double STATUS_IDLE_TIMEOUT; double STATUS_TIMEOUT; double GRPC_CTL_SERVICE_DEFAULT_TIMEOUT; // Default timeout for gRPC FDBCTL service requests @@ -105,6 +117,31 @@ class SWIFT_CXX_IMMORTAL_SINGLETON_TYPE ClientKnobs : public KnobsImpl 0, evictor will assert otherwise. + double LOCATION_CACHE_PEER_EVICTOR_DELAY; + // In the locationCachePeerEvictor sweep, evict an address whose persistent connect-failed + // count advanced by more than this since the previous sweep. + // 0 = any new connect failure counts. + // Value must be >= 0, evictor will assert otherwise. + // + // NOTE: this threshold must be less than the expected number of connect failures + // a dead endpoint generates in one evictor interval, otherwise eviction never triggers. + // Expected failures per interval ~ LOCATION_CACHE_PEER_EVICTOR_DELAY / + // (SERVER_REQUEST_INTERVAL + CONNECTION_MONITOR_TIMEOUT) + int LOCATION_CACHE_PEER_EVICTOR_FAILED_THRESHOLD; + // Number of location cache ranges the address-based invalidation scan processes + // before yielding to the main event loop. + // Value must be >= 1 (evictor will assert otherwise). + // If this value is > LOCATION_CACHE_EVICTION_SIZE, then this will boil down to blocking + // invalidation without any yields in between. + int LOCATION_CACHE_PEER_EVICTOR_SCAN_CHUNK; + int GET_RANGE_SHARD_LIMIT; int WARM_RANGE_SHARD_LIMIT; int STORAGE_METRICS_SHARD_LIMIT; diff --git a/fdbclient/include/fdbclient/MultiVersionTransaction.h b/fdbclient/include/fdbclient/MultiVersionTransaction.h index 084a5697af4..a6dd405a5df 100644 --- a/fdbclient/include/fdbclient/MultiVersionTransaction.h +++ b/fdbclient/include/fdbclient/MultiVersionTransaction.h @@ -95,7 +95,8 @@ struct FdbCApi : public ThreadSafeReferenceCounted { using FDBNativeCdcStreamInfo = struct native_cdc_stream_info { FDBKey name; uint64_t streamId; - FDBKeyRange keyRange; + const FDBKeyRange* ranges; + int rangeCount; int64_t minVersion; }; @@ -157,10 +158,8 @@ struct FdbCApi : public ThreadSafeReferenceCounted { FDBFuture* (*databaseRegisterNativeCdcStream)(FDBDatabase* database, uint8_t const* name, int nameLength, - uint8_t const* beginKey, - int beginKeyLength, - uint8_t const* endKey, - int endKeyLength); + FDBKeyRange const* ranges, + int rangeCount); FDBFuture* (*databaseRemoveNativeCdcStream)(FDBDatabase* database, uint8_t const* name, int nameLength); FDBFuture* (*databaseListNativeCdcStreams)(FDBDatabase* database); FDBFuture* (*databaseCreateNativeCdcConsumer)(FDBDatabase* database, uint8_t const* name, int nameLength); @@ -433,7 +432,7 @@ class DLDatabase : public IDatabase, ThreadSafeReferenceCounted { ThreadFuture rebootWorker(const StringRef& address, bool check, int duration) override; ThreadFuture forceRecoveryWithDataLoss(const StringRef& dcid) override; ThreadFuture createSnapshot(const StringRef& uid, const StringRef& snapshot_command) override; - ThreadFuture registerNativeCdcStream(const KeyRef& name, const KeyRangeRef& keys) override; + ThreadFuture registerNativeCdcStream(const KeyRef& name, const std::vector& ranges) override; ThreadFuture removeNativeCdcStream(const KeyRef& name) override; ThreadFuture> listNativeCdcStreams() override; ThreadFuture> createNativeCdcConsumer(const KeyRef& name) override; @@ -751,7 +750,7 @@ class MultiVersionDatabase final : public IDatabase, ThreadSafeReferenceCounted< ThreadFuture rebootWorker(const StringRef& address, bool check, int duration) override; ThreadFuture forceRecoveryWithDataLoss(const StringRef& dcid) override; ThreadFuture createSnapshot(const StringRef& uid, const StringRef& snapshot_command) override; - ThreadFuture registerNativeCdcStream(const KeyRef& name, const KeyRangeRef& keys) override; + ThreadFuture registerNativeCdcStream(const KeyRef& name, const std::vector& ranges) override; ThreadFuture removeNativeCdcStream(const KeyRef& name) override; ThreadFuture> listNativeCdcStreams() override; ThreadFuture> createNativeCdcConsumer(const KeyRef& name) override; diff --git a/fdbclient/include/fdbclient/NativeAPI.h b/fdbclient/include/fdbclient/NativeAPI.h index 39daa8c67c6..33a9775818c 100644 --- a/fdbclient/include/fdbclient/NativeAPI.h +++ b/fdbclient/include/fdbclient/NativeAPI.h @@ -576,8 +576,8 @@ inline uint64_t getWriteOperationCost(uint64_t bytes) { if (bytes == 0) { return CLIENT_KNOBS->TAG_THROTTLING_RW_FUNGIBILITY_RATIO * CLIENT_KNOBS->TAG_THROTTLING_PAGE_SIZE; } else { - return CLIENT_KNOBS->TAG_THROTTLING_RW_FUNGIBILITY_RATIO * CLIENT_KNOBS->TAG_THROTTLING_PAGE_SIZE * - ((bytes - 1) / CLIENT_KNOBS->TAG_THROTTLING_PAGE_SIZE + 1); + const uint64_t pages = (bytes - 1) / CLIENT_KNOBS->TAG_THROTTLING_PAGE_SIZE + 1; + return CLIENT_KNOBS->TAG_THROTTLING_RW_FUNGIBILITY_RATIO * CLIENT_KNOBS->TAG_THROTTLING_PAGE_SIZE * pages; } } diff --git a/fdbclient/include/fdbclient/NativeCdc.h b/fdbclient/include/fdbclient/NativeCdc.h index e63f6ccf1b7..ac1b1197c14 100644 --- a/fdbclient/include/fdbclient/NativeCdc.h +++ b/fdbclient/include/fdbclient/NativeCdc.h @@ -35,6 +35,7 @@ class NativeCdcConsumer : public ReferenceCounted { Version knownAvailableThrough = invalidVersion; Version lastAcknowledgedVersion; Optional deliveryProxyId; + UID consumerId = deterministicRandom()->randomUniqueID(); bool operationOutstanding = false; public: @@ -53,7 +54,9 @@ class NativeCdcConsumer : public ReferenceCounted { // registration and the remaining operations stay available so existing durable // streams can be drained after the feature is disabled. Requests retry when // stream ownership changes. -Future registerNativeCdcStreamClient(Database cx, Key name, KeyRange keys); +// Ranges form an immutable union. Registration normalizes overlap and adjacency +// so equivalent range sets have the same identity regardless of input order. +Future registerNativeCdcStreamClient(Database cx, Key name, std::vector ranges); Future removeNativeCdcStreamClient(Database cx, Key name); Future> listNativeCdcStreamsClient(Database cx); @@ -99,7 +102,7 @@ enum class NativeCdcRemoveResult { Removed, AlreadyAbsent, StreamReplaced }; // means the registration is gone, not that its retained history is reclaimed. Future removeNativeCdcStreamGuarded(Database cx, Key name, CDCStreamId expectedStreamId); -// Uses the range registered for this name; consumers do not respecify it. A +// Uses the ranges registered for this name; consumers do not respecify them. A // CDCCursor remains a serializable position token and does not hold Database. Future> createNativeCdcConsumer(Database cx, Key name); Reference resumeNativeCdcConsumer(Database cx, CDCCursor position); diff --git a/fdbclient/include/fdbclient/NativeCdcClient.h b/fdbclient/include/fdbclient/NativeCdcClient.h index 7b166bfd6f1..481a08bd586 100644 --- a/fdbclient/include/fdbclient/NativeCdcClient.h +++ b/fdbclient/include/fdbclient/NativeCdcClient.h @@ -31,10 +31,12 @@ // Native CDC value types shared by thread-safe client surfaces and language // bindings. Keep this header independent from NativeAPI so multi-version // client plumbing does not depend on the native client implementation. +constexpr int NATIVE_CDC_MAX_RANGES = 1024; + struct NativeCdcStreamInfo { Key name; CDCStreamId streamId = 0; - KeyRange keys; + std::vector ranges; Version minVersion = invalidVersion; }; diff --git a/fdbclient/include/fdbclient/StorageServerInterface.h b/fdbclient/include/fdbclient/StorageServerInterface.h index 4f290630d00..037b6854954 100644 --- a/fdbclient/include/fdbclient/StorageServerInterface.h +++ b/fdbclient/include/fdbclient/StorageServerInterface.h @@ -88,6 +88,7 @@ struct UpdateCommitCostRequest { struct StorageServerInterface { constexpr static FileIdentifier file_identifier = 15302073; + constexpr static int kNumAdjustedEndpoints = 26; enum { BUSY_ALLOWED = 0, BUSY_FORCE = 1, BUSY_LOCAL = 2 }; enum { LocationAwareLoadBalance = 1 }; @@ -143,6 +144,17 @@ struct StorageServerInterface { UID id() const { return uniqueID; } bool isAcceptingRequests() const { return acceptingRequests; } void startAcceptingRequests() { acceptingRequests = true; } + + std::vector getEndpointTokens() const { + // Token at index 0 is the base `getValue` endpoint; adjusted endpoints start at 1. + std::vector tokens; + tokens.reserve(kNumAdjustedEndpoints + 1); + tokens.push_back(getValue.getEndpoint().token); + for (int i = 1; i <= kNumAdjustedEndpoints; ++i) { + tokens.push_back(getValue.getEndpoint().getAdjustedEndpoint(i).token); + } + return tokens; + } void stopAcceptingRequests() { acceptingRequests = false; } bool isTss() const { return tssPairID.present(); } std::string toString() const { return id().shortString(); } @@ -160,6 +172,10 @@ struct StorageServerInterface { if (Ar::isDeserializing) { initEndpointsFromGetValue(); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && g_network && g_network->global(INetwork::enFlowTransport)) { + FlowTransport::transport().interfaceTracker.created( + getValue.getEndpoint().getPrimaryAddress(), "SS", getEndpointTokens()); + } } } bool operator==(StorageServerInterface const& s) const { return uniqueID == s.uniqueID; } diff --git a/fdbclient/include/fdbclient/StorageServerLoadBalance.h b/fdbclient/include/fdbclient/StorageServerLoadBalance.h index 8849d5c9ee3..1352c43e500 100644 --- a/fdbclient/include/fdbclient/StorageServerLoadBalance.h +++ b/fdbclient/include/fdbclient/StorageServerLoadBalance.h @@ -368,13 +368,13 @@ struct LoadBalanceRequestHooks tssRequestStream(tssData.get().endpoint); Future> fTssResult = tssRequestStream.tryGetReply(request); - model->addActor.send(tssComparison(request, - ssResponse, - fTssResult, - tssData.get(), - stream->getEndpoint().token.first(), - alternatives, - channel)); + model->addBackgroundActor(tssComparison(request, + ssResponse, + fTssResult, + tssData.get(), + stream->getEndpoint().token.first(), + alternatives, + channel)); } } } diff --git a/fdbclient/include/fdbclient/SystemData.h b/fdbclient/include/fdbclient/SystemData.h index cff862a4a6c..a1100ea2f3a 100644 --- a/fdbclient/include/fdbclient/SystemData.h +++ b/fdbclient/include/fdbclient/SystemData.h @@ -293,14 +293,16 @@ extern const KeyRef cdcMaxStreamIdKey; Value cdcMaxStreamIdValue(CDCStreamId streamId); CDCStreamId decodeCDCMaxStreamIdValue(ValueRef const& value); -// "\xff/cdc/keys/[[CDCStreamId]]" := "[[KeyRange]]" +// "\xff/cdc/keys/[[CDCStreamId]]" := "[[vector]]" extern const KeyRangeRef cdcStreamKeys; Key cdcStreamKeyFor(CDCStreamId streamId); CDCStreamId decodeCDCStreamKey(KeyRef const& key); -Value cdcStreamKeysValue(KeyRangeRef const& keys); -KeyRange decodeCDCStreamKeysValue(ValueRef const& value); +Value cdcStreamKeysValue(std::vector const& ranges); +std::vector decodeCDCStreamKeysValue(ValueRef const& value); -// "\xff/cdc/tagHistory/[[CDCStreamId]][[Version]][[Tag]]" := "" +// "\xff/cdc/tagHistory/[[CDCStreamId]][[Version]][[Tag]]" := "" | commit versionstamp +// Empty values use the key's version. A pending live retag stores its exact commit +// boundary in the value; the key retains the transaction read version for ordering. struct CDCTagHistoryEntry { constexpr static FileIdentifier file_identifier = 13091844; @@ -322,6 +324,7 @@ extern const KeyRangeRef cdcTagHistoryKeys; Key cdcTagHistoryKeyFor(CDCStreamId streamId, Version version, Tag tag); KeyRange cdcTagHistoryRangeFor(CDCStreamId streamId); CDCTagHistoryEntry decodeCDCTagHistoryKey(KeyRef const& key); +CDCTagHistoryEntry decodeCDCTagHistoryEntry(KeyRef const& key, ValueRef const& value); // "\xff\x02/cdc/tagOwner/[[Tag]]" := "[[CDCStreamId]]" // Derived lookup hint, not authoritative ownership. Validate the stream is active diff --git a/fdbclient/include/fdbclient/ThreadSafeTransaction.h b/fdbclient/include/fdbclient/ThreadSafeTransaction.h index 6a310631637..6a3d76190aa 100644 --- a/fdbclient/include/fdbclient/ThreadSafeTransaction.h +++ b/fdbclient/include/fdbclient/ThreadSafeTransaction.h @@ -58,7 +58,7 @@ class ThreadSafeDatabase : public IDatabase, public ThreadSafeReferenceCounted forceRecoveryWithDataLoss(const StringRef& dcid) override; ThreadFuture createSnapshot(const StringRef& uid, const StringRef& snapshot_command) override; - ThreadFuture registerNativeCdcStream(const KeyRef& name, const KeyRangeRef& keys) override; + ThreadFuture registerNativeCdcStream(const KeyRef& name, const std::vector& ranges) override; ThreadFuture removeNativeCdcStream(const KeyRef& name) override; ThreadFuture> listNativeCdcStreams() override; ThreadFuture> createNativeCdcConsumer(const KeyRef& name) override; diff --git a/fdbclient/include/fdbclient/VersionedMap.h b/fdbclient/include/fdbclient/VersionedMap.h index 46b04389757..30a8041f93f 100644 --- a/fdbclient/include/fdbclient/VersionedMap.h +++ b/fdbclient/include/fdbclient/VersionedMap.h @@ -143,6 +143,9 @@ class PTreeFinger { PTreeFinger(PTreeFinger&& f) { *this = f; } PTreeFinger& operator=(PTreeFinger const& f) { + if (this == &f) { + return *this; + } size_ = f.size_; bound_sz_ = f.bound_sz_; std::copy(f.entries_, f.entries_ + size_, entries_); @@ -150,6 +153,9 @@ class PTreeFinger { } PTreeFinger& operator=(PTreeFinger&& f) { + if (this == &f) { + return *this; + } size_ = std::exchange(f.size_, 0); bound_sz_ = f.bound_sz_; std::copy(f.entries_, f.entries_ + size_, entries_); diff --git a/fdbrpc/AsyncFileCached.cpp b/fdbrpc/AsyncFileCached.cpp index b8b9b4ce967..20d36127525 100644 --- a/fdbrpc/AsyncFileCached.cpp +++ b/fdbrpc/AsyncFileCached.cpp @@ -265,7 +265,7 @@ Future AsyncFileCached::flush() { if (!f.isReady()) i++; } - ASSERT(flushable.size() <= debug_count); + ASSERT_LE(flushable.size(), debug_count); return waitForAll(unflushed); } diff --git a/fdbrpc/AsyncFileEncrypted.cpp b/fdbrpc/AsyncFileEncrypted.cpp index d70abbf9b1d..aa3039b6c3b 100644 --- a/fdbrpc/AsyncFileEncrypted.cpp +++ b/fdbrpc/AsyncFileEncrypted.cpp @@ -65,6 +65,9 @@ class AsyncFileEncryptedImpl { } static Future read(Reference self, void* data, int length, int64_t offset) { + if (length == 0) { + co_return 0; + } if (self->fileSize == -1) { int64_t rawSize = co_await self->file->size(); self->fileSize = AsyncFileEncrypted::rawToLogicalSize(rawSize, self->encryptionBlockSize); @@ -152,7 +155,7 @@ class AsyncFileEncryptedImpl { AsyncFileEncrypted::AsyncFileEncrypted(Reference file, Mode mode, int encryptionBlockSize) : file(file), mode(mode), currentBlock(0), encryptionBlockSize(encryptionBlockSize) { - ASSERT(encryptionBlockSize > 0); + ASSERT_GT(encryptionBlockSize, 0); firstBlockIV = AsyncFileEncryptedImpl::getFirstBlockIV(file->getFilename()); if (mode == Mode::APPEND_ONLY) { writeBuffer = std::vector(encryptionBlockSize, 0); @@ -173,7 +176,7 @@ int64_t AsyncFileEncrypted::rawToLogicalSize(int64_t rawSize, int blockSize) { const int64_t trailing = rawSize % rawBlockSize; int64_t logical = fullBlocks * blockSize; if (trailing > 0) { - ASSERT(trailing > GCM_TAG_LEN); + ASSERT_GT(trailing, GCM_TAG_LEN); logical += trailing - GCM_TAG_LEN; } return logical; @@ -299,3 +302,34 @@ TEST_CASE("fdbrpc/AsyncFileEncrypted") { } ASSERT(writeBuffer == readBuffer); } + +TEST_CASE("fdbrpc/AsyncFileEncrypted/ZeroLengthRead") { + const int encryptionBlockSize = 4096; + const int bytes = 2 * encryptionBlockSize + 1; + StreamCipherKey::initializeGlobalRandomTestKey(); + int flags = IAsyncFile::OPEN_READWRITE | IAsyncFile::OPEN_CREATE | IAsyncFile::OPEN_ATOMIC_WRITE_AND_CREATE | + IAsyncFile::OPEN_UNBUFFERED | IAsyncFile::OPEN_UNCACHED | IAsyncFile::OPEN_NO_AIO; + Reference rawFile = co_await IAsyncFileSystem::filesystem()->open( + joinPath(params.getDataDir(), "test-encrypted-file-zero-length"), flags, 0600); + std::vector readBuffer(encryptionBlockSize, 0xa5); + const auto untouchedBuffer = readBuffer; + Reference file = + makeReference(rawFile, AsyncFileEncrypted::Mode::READ_ONLY, encryptionBlockSize); + int bytesRead = co_await file->read(readBuffer.data(), 0, 0); + ASSERT_EQ(bytesRead, 0); + ASSERT(readBuffer == untouchedBuffer); + + file = makeReference(rawFile, AsyncFileEncrypted::Mode::APPEND_ONLY, encryptionBlockSize); + std::vector writeBuffer(bytes, 0x5a); + co_await file->write(writeBuffer.data(), bytes, 0); + co_await file->sync(); + file = makeReference(rawFile, AsyncFileEncrypted::Mode::READ_ONLY, encryptionBlockSize); + for (int offset : { 0, 1, encryptionBlockSize, encryptionBlockSize + 1, bytes, bytes + 1 }) { + bytesRead = co_await file->read(readBuffer.data(), 0, offset); + ASSERT_EQ(bytesRead, 0); + ASSERT(readBuffer == untouchedBuffer); + } + bytesRead = co_await file->read(readBuffer.data(), 1, 0); + ASSERT_EQ(bytesRead, 1); + ASSERT_EQ(readBuffer[0], writeBuffer[0]); +} diff --git a/fdbrpc/AsyncFileKAIO.h b/fdbrpc/AsyncFileKAIO.h index 67004e5c952..acdf24e6e97 100644 --- a/fdbrpc/AsyncFileKAIO.h +++ b/fdbrpc/AsyncFileKAIO.h @@ -433,7 +433,7 @@ class AsyncFileKAIO final : public IAsyncFile, public ReferenceCountedowner->lastFileSize != io->owner->nextFileSize) { ++ctx.countPreSubmitTruncate; int64_t truncateSize = io->owner->nextFileSize - io->owner->lastFileSize; - ASSERT(truncateSize > 0); + ASSERT_GT(truncateSize, 0); ctx.preSubmitTruncateBytes += truncateSize; largestTruncate = std::max(largestTruncate, truncateSize); io->owner->truncate(io->owner->nextFileSize); diff --git a/fdbrpc/CountedSectionTest.cpp b/fdbrpc/CountedSectionTest.cpp index 48fa71700bf..571b737b0a0 100644 --- a/fdbrpc/CountedSectionTest.cpp +++ b/fdbrpc/CountedSectionTest.cpp @@ -1,5 +1,5 @@ /* - * CountedSecitonTest.cpp + * CountedSectionTest.cpp * * This source file is part of the FoundationDB open source project * diff --git a/fdbrpc/FailureMonitor.cpp b/fdbrpc/FailureMonitor.cpp index 75579ec0ed9..d8114082bac 100644 --- a/fdbrpc/FailureMonitor.cpp +++ b/fdbrpc/FailureMonitor.cpp @@ -65,7 +65,7 @@ Future IFailureMonitor::onStateEqual(Endpoint const& endpoint, FailureStat } Future IFailureMonitor::onFailedFor(Endpoint const& endpoint, double sustainedFailureDuration, double slope) { - ASSERT(slope < 1.0); + ASSERT_LT(slope, 1.0); return waitForContinuousFailure(this, endpoint, sustainedFailureDuration, slope); } diff --git a/fdbrpc/FileTransfer.cpp b/fdbrpc/FileTransfer.cpp index 479507d7439..ba2712baf88 100644 --- a/fdbrpc/FileTransfer.cpp +++ b/fdbrpc/FileTransfer.cpp @@ -19,6 +19,7 @@ */ #ifdef FLOW_GRPC_ENABLED #include +#include #include "FileTransfer.h" #include "flow/IRandom.h" @@ -144,21 +145,9 @@ std::optional FileTransferClient::DownloadFile(const std::string& filena const std::string& output_filename, bool verify) { - uint32_t expected_crc = 0; - uint32_t expected_size = 0; - { - fdbrpc::GetFileInfoRequest request; - grpc::ClientContext context; - request.set_file_name(filename); - request.set_get_crc_checksum(verify); - request.set_get_size(true); - fdbrpc::GetFileInfoReply response; - auto res = stub_->GetFileInfo(&context, request, &response); - if (!res.ok()) { - return std::nullopt; - } - expected_crc = response.crc_checksum(); - expected_size = response.file_size(); + const auto fileInfo = GetFileInfo(filename, verify); + if (!fileInfo.has_value()) { + return std::nullopt; } fdbrpc::DownloadRequest request; @@ -188,13 +177,13 @@ std::optional FileTransferClient::DownloadFile(const std::string& filena // Close file after writing output_file.close(); - failed = failed || (bytes_read != expected_size); + failed = failed || std::cmp_not_equal(bytes_read, fileInfo->file_size()); // Verify checksum if (!failed && verify) { std::ifstream output_file_reader(output_filename); uint32_t actual_crc = crc32_checksum_ifstream(&output_file_reader); - failed = (actual_crc != expected_crc); + failed = (actual_crc != fileInfo->crc_checksum()); } // Check final gRPC status diff --git a/fdbrpc/FlowGrpc.cpp b/fdbrpc/FlowGrpc.cpp index 186f0a68a12..5e66ddc6c4e 100644 --- a/fdbrpc/FlowGrpc.cpp +++ b/fdbrpc/FlowGrpc.cpp @@ -68,7 +68,7 @@ Future GrpcServer::run() { co_await run_actor_; } catch (Error& err) { if (err.code() != error_code_operation_cancelled) { - TraceEvent(SevError, "GrpcServerRunError").detail("Endpoint", address_).detail("Error", err.name()); + TraceEvent(SevError, "GrpcServerRunError").error(err).detail("Endpoint", address_); throw; } } diff --git a/fdbrpc/FlowTests.cpp b/fdbrpc/FlowTests.cpp index d8d89c6761c..3e3bf77441f 100644 --- a/fdbrpc/FlowTests.cpp +++ b/fdbrpc/FlowTests.cpp @@ -118,18 +118,6 @@ void onReady(FutureStream&& f, Func&& func, ErrFunc&& errFunc) { } } -static Future emptyVoidActor(Uncancellable = Uncancellable()) { - co_return; -} - -static Future emptyActor() { - return Void(); -} - -static Future oneWaitVoidActor(Future f, Uncancellable = Uncancellable()) { - co_await f; -} - static Future oneWaitActor(Future f) { co_await f; } @@ -1027,301 +1015,114 @@ TEST_CASE("/flow/flow/chooseTwoActor") { return Void(); } -TEST_CASE("#flow/flow/perf/actor patterns") { - double start; - int N = 1000000; - - start = timer(); - for (int i = 0; i < N; i++) - emptyVoidActor(); - printf("emptyVoidActor(): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - - start = timer(); - for (int i = 0; i < N; i++) { - emptyActor(); - } - printf("emptyActor(): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - - Promise neverSet; - Future never = neverSet.getFuture(); - Future already = Void(); - - start = timer(); - for (int i = 0; i < N; i++) - oneWaitVoidActor(already); - printf("oneWaitVoidActor(already): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - - /*start = timer(); - for (int i = 0; i < N; i++) - oneWaitVoidActor(never); - printf("oneWaitVoidActor(never): %0.1f M/sec\n", N / 1e6 / (timer() - start));*/ - - { - start = timer(); - for (int i = 0; i < N; i++) { - Future f = oneWaitActor(already); - ASSERT(f.isReady()); - } - printf("oneWaitActor(already): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - start = timer(); - for (int i = 0; i < N; i++) { - Future f = oneWaitActor(never); - ASSERT(!f.isReady()); - } - printf("(cancelled) oneWaitActor(never): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - start = timer(); - for (int i = 0; i < N; i++) { - Promise p; - Future f = oneWaitActor(p.getFuture()); - p.send(Void()); - ASSERT(f.isReady()); - } - printf("oneWaitActor(after): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - std::vector> pipe(N); - std::vector> out(N); - start = timer(); - for (int i = 0; i < N; i++) { - out[i] = oneWaitActor(pipe[i].getFuture()); - } - for (int i = 0; i < N; i++) { - pipe[i].send(Void()); - ASSERT(out[i].isReady()); - } - printf("oneWaitActor(fifo): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - std::vector> pipe(N); - std::vector> out(N); - start = timer(); - for (int i = 0; i < N; i++) { - out[i] = oneWaitActor(pipe[i].getFuture()); - } - for (int i = N - 1; i >= 0; i--) { - pipe[i].send(Void()); - ASSERT(out[i].isReady()); - } - printf("oneWaitActor(lifo): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - start = timer(); - for (int i = 0; i < N; i++) { - Future f = chooseTwoActor(already, already); - ASSERT(f.isReady()); - } - printf("chooseTwoActor(already, already): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - start = timer(); - for (int i = 0; i < N; i++) { - Future f = chooseTwoActor(already, never); - ASSERT(f.isReady()); - } - printf("chooseTwoActor(already, never): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - start = timer(); - for (int i = 0; i < N; i++) { - Future f = chooseTwoActor(never, already); - ASSERT(f.isReady()); - } - printf("chooseTwoActor(never, already): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - start = timer(); - for (int i = 0; i < N; i++) { - Future f = chooseTwoActor(never, never); - ASSERT(!f.isReady()); - } - printf("(cancelled) chooseTwoActor(never, never): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - start = timer(); - for (int i = 0; i < N; i++) { - Promise p; - Future f = chooseTwoActor(p.getFuture(), never); - p.send(Void()); - ASSERT(f.isReady()); - } - printf("chooseTwoActor(after, never): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - std::vector> pipe(N); - std::vector> out(N); - start = timer(); - for (int i = 0; i < N; i++) { - out[i] = chooseTwoActor(pipe[i].getFuture(), never); - } - for (int i = 0; i < N; i++) { - pipe[i].send(Void()); - ASSERT(out[i].isReady()); - } - printf("chooseTwoActor(fifo, never): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - std::vector> pipe(N); - std::vector> out(N); - start = timer(); - for (int i = 0; i < N; i++) { - out[i] = chooseTwoActor(pipe[i].getFuture(), pipe[i].getFuture()); - } - for (int i = 0; i < N; i++) { - pipe[i].send(Void()); - ASSERT(out[i].isReady()); - } - printf("chooseTwoActor(fifo, fifo): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - std::vector> pipe(N); - std::vector> out(N); - start = timer(); - for (int i = 0; i < N; i++) { - out[i] = chooseTwoActor(chooseTwoActor(pipe[i].getFuture(), never), never); - } - for (int i = 0; i < N; i++) { - pipe[i].send(Void()); - ASSERT(out[i].isReady()); - } - printf("chooseTwoActor^2((fifo, never), never): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } +TEST_CASE("/flow/flow/actor patterns/wait") { + Future ready = oneWaitActor(Void()); + ASSERT(ready.isReady() && !ready.isError()); + Promise input; { - start = timer(); - for (int i = 0; i < N; i++) { - Promise p; - Future f = oneWaitActor(chooseTwoActor(p.getFuture(), never)); - p.send(Void()); - ASSERT(f.isReady()); - } - printf("oneWaitActor(chooseTwoActor(after, never)): %0.1f M/sec\n", N / 1e6 / (timer() - start)); + Future cancelled = oneWaitActor(input.getFuture()); + ASSERT(!cancelled.isReady() && input.getFutureReferenceCount() > 0); } + ASSERT(input.getFutureReferenceCount() == 0); { - std::vector> pipe(N); - std::vector> out(N); - start = timer(); - for (int i = 0; i < N; i++) { - out[i] = oneWaitActor(chooseTwoActor(pipe[i].getFuture(), never)); - } - for (int i = 0; i < N; i++) { - pipe[i].send(Void()); - ASSERT(out[i].isReady()); - } - printf("oneWaitActor(chooseTwoActor(fifo, never)): %0.1f M/sec\n", N / 1e6 / (timer() - start)); + Future completed = oneWaitActor(input.getFuture()); + ASSERT(!completed.isReady()); + input.send(Void()); + ASSERT(completed.isReady() && !completed.isError()); } + ASSERT(input.getFutureReferenceCount() == 0); + return Void(); +} +TEST_CASE("/flow/flow/actor patterns/race") { + Promise pending; { - start = timer(); - for (int i = 0; i < N; i++) { - Promise p; - Future f = chooseTwoActor(p.getFuture(), never); - Future a = oneWaitActor(f); - Future b = oneWaitActor(f); - p.send(Void()); - ASSERT(f.isReady()); - } - printf("2xoneWaitActor(chooseTwoActor(after, never)): %0.2f M/sec\n", N / 1e6 / (timer() - start)); + Future ready = Void(); + Future bothReady = chooseTwoActor(ready, ready); + Future firstReady = chooseTwoActor(ready, pending.getFuture()); + Future secondReady = chooseTwoActor(pending.getFuture(), ready); + ASSERT(bothReady.isReady() && !bothReady.isError()); + ASSERT(firstReady.isReady() && !firstReady.isError()); + ASSERT(secondReady.isReady() && !secondReady.isError()); } + ASSERT(pending.getFutureReferenceCount() == 0); { - std::vector> pipe(N); - std::vector> out1(N); - std::vector> out2(N); - start = timer(); - for (int i = 0; i < N; i++) { - Future f = chooseTwoActor(pipe[i].getFuture(), never); - out1[i] = oneWaitActor(f); - out2[i] = oneWaitActor(f); - } - for (int i = 0; i < N; i++) { - pipe[i].send(Void()); - ASSERT(out2[i].isReady()); - } - printf("2xoneWaitActor(chooseTwoActor(fifo, never)): %0.2f M/sec\n", N / 1e6 / (timer() - start)); + Future cancelled = chooseTwoActor(pending.getFuture(), pending.getFuture()); + ASSERT(!cancelled.isReady() && pending.getFutureReferenceCount() > 0); } + ASSERT(pending.getFutureReferenceCount() == 0); { - std::vector> pipe(N); - std::vector> out1(N); - std::vector> out2(N); - start = timer(); - for (int i = 0; i < N; i++) { - Future f = chooseTwoActor(oneWaitActor(pipe[i].getFuture()), never); - out1[i] = oneWaitActor(f); - out2[i] = oneWaitActor(f); - } - for (int i = 0; i < N; i++) { - pipe[i].send(Void()); - ASSERT(out2[i].isReady()); - } - printf("2xoneWaitActor(chooseTwoActor(oneWaitActor(fifo), never)): %0.2f M/sec\n", N / 1e6 / (timer() - start)); + Future sharedInput = chooseTwoActor(pending.getFuture(), pending.getFuture()); + ASSERT(!sharedInput.isReady()); + pending.send(Void()); + ASSERT(sharedInput.isReady() && !sharedInput.isError()); } + ASSERT(pending.getFutureReferenceCount() == 0); + return Void(); +} - { - std::vector> pipe(N); - std::vector> out1(N); - std::vector> out2(N); - start = timer(); - for (int i = 0; i < N; i++) { - g_cheese = pipe[i].getFuture(); - Future f = chooseTwoActor(cheeseWaitActor(), never); - g_cheese = f; - out1[i] = cheeseWaitActor(); - out2[i] = cheeseWaitActor(); +TEST_CASE("/flow/flow/actor patterns/composition") { + const int batchSize = 4; + for (bool lifo : { false, true }) { + Promise never; + std::vector> inputs(batchSize); + { + std::vector> waits(batchSize); + std::vector> sharedRaces(batchSize); + std::vector> nestedRaces(batchSize); + std::vector> firstOutputs(batchSize); + std::vector> secondOutputs(batchSize); + for (int i = 0; i < batchSize; ++i) { + waits[i] = oneWaitActor(inputs[i].getFuture()); + sharedRaces[i] = chooseTwoActor(inputs[i].getFuture(), inputs[i].getFuture()); + nestedRaces[i] = + chooseTwoActor(chooseTwoActor(inputs[i].getFuture(), never.getFuture()), never.getFuture()); + Future fanout = chooseTwoActor(oneWaitActor(inputs[i].getFuture()), never.getFuture()); + firstOutputs[i] = oneWaitActor(fanout); + secondOutputs[i] = oneWaitActor(fanout); + ASSERT(!waits[i].isReady() && !sharedRaces[i].isReady() && !nestedRaces[i].isReady() && + !firstOutputs[i].isReady() && !secondOutputs[i].isReady()); + } + for (int i = 0; i < batchSize; ++i) { + const int index = lifo ? batchSize - 1 - i : i; + inputs[index].send(Void()); + for (int j = 0; j < batchSize; ++j) { + const bool completed = lifo ? j >= index : j <= index; + ASSERT(waits[j].isReady() == completed && sharedRaces[j].isReady() == completed && + nestedRaces[j].isReady() == completed && firstOutputs[j].isReady() == completed && + secondOutputs[j].isReady() == completed); + } + ASSERT(!waits[index].isError() && !sharedRaces[index].isError() && !nestedRaces[index].isError() && + !firstOutputs[index].isError() && !secondOutputs[index].isError()); + } } - for (int i = 0; i < N; i++) { - pipe[i].send(Void()); - ASSERT(out2[i].isReady()); + ASSERT(never.getFutureReferenceCount() == 0); + for (const auto& input : inputs) { + ASSERT(input.getFutureReferenceCount() == 0); } - printf("2xcheeseActor(chooseTwoActor(cheeseActor(fifo), never)): %0.2f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - PromiseStream data; - start = timer(); - Future sum = sumActor(data.getFuture()); - for (int i = 0; i < N; i++) - data.send(1); - data.sendError(end_of_stream()); - ASSERT(sum.get() == N); - printf("sumActor: %0.2f M/sec\n", N / 1e6 / (timer() - start)); } + return Void(); +} +TEST_CASE("/flow/flow/actor patterns/global input") { + Promise original, replacement; { - start = timer(); - std::vector> ps(3); - std::vector> fs(3); - - for (int i = 0; i < N; i++) { - ps.clear(); - ps.resize(3); - for (int j = 0; j < ps.size(); j++) - fs[j] = ps[j].getFuture(); - - Future q = quorum(fs, 2); - for (auto& p : ps) - p.send(Void()); - } - printf("quorum(2/3): %0.2f M/sec\n", N / 1e6 / (timer() - start)); - } - + g_cheese = original.getFuture(); + Future first = cheeseWaitActor(); + g_cheese = replacement.getFuture(); + Future second = cheeseWaitActor(); + g_cheese = Future(); + ASSERT(!first.isReady() && !second.isReady()); + original.send(Void()); + ASSERT(first.isReady() && !first.isError() && !second.isReady()); + replacement.send(Void()); + ASSERT(second.isReady() && !second.isError()); + } + ASSERT(original.getFutureReferenceCount() == 0 && replacement.getFutureReferenceCount() == 0); return Void(); } diff --git a/fdbrpc/FlowTransport.cpp b/fdbrpc/FlowTransport.cpp index f5d998b5474..4b78928baef 100644 --- a/fdbrpc/FlowTransport.cpp +++ b/fdbrpc/FlowTransport.cpp @@ -150,7 +150,7 @@ void EndpointMap::realloc() { void EndpointMap::insertWellKnown(NetworkMessageReceiver* r, const Endpoint::Token& token, TaskPriority priority) { const auto index = token.second(); - ASSERT(index < uint64_t(wellKnownEndpointCount)); + ASSERT_LT(index, uint64_t(wellKnownEndpointCount)); ASSERT(data[index].receiver == nullptr); data[index].receiver = r; data[index].token() = @@ -430,6 +430,10 @@ class TransportData { NetworkAddressCachedString localAddresses; std::vector> listeners; std::unordered_map> peers; + + std::unordered_map persistentConnectFailedCount; + double persistentConnectFailedLastPrune = 0; + // FIXME: explain what the std::pair represent: std::unordered_map> closedPeers; HealthMonitor healthMonitor; @@ -975,13 +979,56 @@ Future connectionKeeper(Reference self, } } catch (Error& e) { ++self->connectFailedCount; + // Track per-address cumulative connect failures + last-failure time. The map + // potentially has unbounded list of peers as they're having connection issues. + // To make the map bounded, evict based on TTL. + // + // Note: this prune only runs here, inside the connect-failure path, so it is + // failure-driven rather than on a timer. If connect failures stop across all + // addresses, the scan does not run again and stale entries can linger past their + // TTL until the next failure on any address. The map is still bounded -- by the + // set of addresses ever contacted, and the next failure prunes the stale ones -- + // so this is not unbounded growth, just not strictly time-based eviction. + { + double ttl = FLOW_KNOBS->PERSISTENT_CONNECT_FAILED_COUNT_TTL; + double tNow = now(); + auto& failCounts = self->transport->persistentConnectFailedCount; + auto& info = failCounts[self->destination]; + info.count++; + info.lastFailed = tNow; + TraceEvent("PersistentConnectFailed") + .suppressFor(5.0) + .detail("PeerAddr", self->destination) + .detail("ConnectFailedTotal", info.count); + if (ttl > 0 && tNow - self->transport->persistentConnectFailedLastPrune >= ttl) { + self->transport->persistentConnectFailedLastPrune = tNow; + for (auto it = failCounts.begin(); it != failCounts.end();) { + if (tNow - it->second.lastFailed >= ttl) { + TraceEvent("PersistentConnectFailedPrune") + .suppressFor(5.0) + .detail("PeerAddr", it->first) + .detail("ConnectFailedTotal", it->second.count); + it = failCounts.erase(it); + } else { + ++it; + } + } + } + } if (e.code() != error_code_connection_failed) { throw; } + TraceEvent("ConnectionTimedOut", conn ? conn->getDebugID() : UID()) .suppressFor(1.0) .detail("PeerAddr", self->destination) - .detail("PeerAddress", self->destination); + .detail("PeerAddress", self->destination) + .detail("PeerReferences", self->peerReferences) + .detail("ReliableEmpty", self->reliable.empty()) + .detail("UnsentEmpty", self->unsent.empty()) + .detail("OutstandingReplies", self->outstandingReplies) + .detail("ConnectFailedCount", self->connectFailedCount) + .detail("Connected", self->connected); throw; } @@ -1130,7 +1177,13 @@ Future connectionKeeper(Reference self, .errorUnsuppressed(e) .suppressFor(1.0) .detail("PeerAddr", self->destination) - .detail("PeerAddress", self->destination); + .detail("PeerAddress", self->destination) + .detail("PeerReferences", self->peerReferences) + .detail("ReliableEmpty", self->reliable.empty()) + .detail("UnsentEmpty", self->unsent.empty()) + .detail("OutstandingReplies", self->outstandingReplies) + .detail("ConnectFailedCount", self->connectFailedCount) + .detail("Connected", self->connected); self->connect.cancel(); self->transport->peers.erase(self->destination); self->transport->orderedAddresses.erase(self->destination); @@ -1290,7 +1343,7 @@ static void deliverNow(TransportData* self, g_currentDeliveryPeerDisconnect = nullptr; }); StringRef data = reader.arenaReadAll(); - ASSERT(data.size() > 8); + ASSERT_GT(data.size(), 8); ArenaObjectReader objReader(std::move(reader.arena()), data, AssumeVersion(reader.protocolVersion())); receiver->receive(objReader); } catch (Error& e) { @@ -1946,6 +1999,225 @@ static Future multiVersionCleanupWorker(TransportData* self) { } } +// ==== InterfaceTracker ==== +// Bookkeeping is gated entirely on FLOW_KNOBS->STALE_PEER_OBSERVABILITY: every +// mutating method early-returns when the knob is off, every accessor returns +// empty/zero. Callers may invoke these unconditionally. + +void InterfaceTracker::created(const NetworkAddress& dstAddr, + const std::string& dstRole, + const std::vector& tokens) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return; + } + auto& entry = map[Key{ dstAddr, dstRole }]; + entry.numCreated += tokens.size(); + int64_t id = nextCreateId++; + entry.createRecords.push_back( + { id, g_network ? g_network->now() : 0.0, platform::get_backtrace(), (int)tokens.size() }); + for (const auto& tok : tokens) { + tokenToInfo[TokenKey{ dstAddr, tok }] = TokenInfo{ dstRole, id }; + } +} + +void InterfaceTracker::peerRefAdded(const NetworkAddress& addr) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return; + } + peerRefCounts[addr].added++; +} + +void InterfaceTracker::peerRefRemovedRaw(const NetworkAddress& addr) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return; + } + peerRefCounts[addr].removed++; +} + +void InterfaceTracker::peerRefRemoved(const NetworkAddress& addr, const UID& token) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return; + } + auto it = tokenToInfo.find(TokenKey{ addr, token }); + if (it != tokenToInfo.end()) { + ASSERT(map.contains(Key{ addr, it->second.role })); + auto& entry = map[Key{ addr, it->second.role }]; + entry.numDeleted++; + for (auto& rec : entry.createRecords) { + if (rec.id == it->second.createId) { + rec.numStreamsDeleted++; + break; + } + } + } +} + +int64_t InterfaceTracker::getDelta(const NetworkAddress& addr, const std::string& role) const { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return 0; + } + auto it = map.find(Key{ addr, role }); + if (it == map.end()) + return 0; + return it->second.numCreated - it->second.numDeleted; +} + +int64_t InterfaceTracker::flowReceiverCreated(const NetworkAddress& addr, const UID& token) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return -1; + } + int64_t id = nextFlowReceiverId++; + flowReceiverRecords[id] = FlowReceiverRecord{ + id, addr, token, g_network ? g_network->now() : 0.0, platform::get_backtrace(), currentCallerTag + }; + return id; +} + +void InterfaceTracker::flowReceiverDestroyed(const NetworkAddress& addr, int64_t id) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return; + } + flowReceiverRecords.erase(id); +} + +int64_t InterfaceTracker::promiseRefAdded(const NetworkAddress& addr, const UID& token) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return -1; + } + int64_t id = nextRefId++; + refRecords[id] = RefRecord{ id, addr, token, g_network ? g_network->now() : 0.0, platform::get_backtrace(), false }; + return id; +} + +void InterfaceTracker::promiseRefReleased(int64_t id) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return; + } + if (id < 0) + return; + refRecords.erase(id); +} + +int64_t InterfaceTracker::futureRefAdded(const NetworkAddress& addr, const UID& token) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return -1; + } + int64_t id = nextRefId++; + refRecords[id] = RefRecord{ id, addr, token, g_network ? g_network->now() : 0.0, platform::get_backtrace(), true }; + return id; +} + +// Clone the (addr, token) of an existing tracked future ref into a brand-new +// tracked ref. Used by FutureStream's copy ctor/assignment: a copy is a +// distinct ref and must get its own id so it is released independently, rather +// than going untracked (which would undercount live future refs). Copy the +// fields out before inserting -- the insert may rehash and invalidate `it`. +int64_t InterfaceTracker::futureRefCopied(int64_t srcId) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return -1; + } + if (srcId < 0) { + return -1; + } + auto it = refRecords.find(srcId); + if (it == refRecords.end()) { + return -1; + } + NetworkAddress addr = it->second.addr; + UID token = it->second.token; + int64_t id = nextRefId++; + refRecords[id] = RefRecord{ id, addr, token, g_network ? g_network->now() : 0.0, platform::get_backtrace(), true }; + return id; +} + +void InterfaceTracker::futureRefReleased(int64_t id) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return; + } + if (id < 0) + return; + refRecords.erase(id); +} + +void InterfaceTracker::prettyPrintLeakedReceivers(const NetworkAddress& srcAddr, + const std::vector& filterAddrs) const { + for (const auto& [id, rec] : flowReceiverRecords) { + for (const auto& filterAddr : filterAddrs) { + if (rec.addr == filterAddr) { + TraceEvent("FlowReceiverLeaked") + .detail("SrcProcess", srcAddr) + .detail("CallerTag", rec.callerTag) + .detail("DstAddress", rec.addr) + .detail("Token", rec.token) + .detail("ReceiverId", rec.id) + .detail("CreateTime", format("%.6f", rec.createTime)) + .detail("Backtrace", rec.backtrace); + } + } + } +} + +void InterfaceTracker::prettyPrintLeakedRefs(const NetworkAddress& srcAddr, + const std::vector& filterAddrs) const { + for (const auto& [id, rec] : refRecords) { + for (const auto& filterAddr : filterAddrs) { + if (rec.addr == filterAddr) { + TraceEvent(rec.isFutureRef ? "FutureRefLeaked" : "PromiseRefLeaked") + .detail("SrcProcess", srcAddr) + .detail("DstAddress", rec.addr) + .detail("Token", rec.token) + .detail("RefId", rec.id) + .detail("RefCreateTime", format("%.6f", rec.time)) + .detail("RefBacktrace", rec.backtrace); + break; + } + } + } +} + +void InterfaceTracker::prettyPrint(const NetworkAddress& srcAddr, + const std::vector& filterAddrs) const { + for (const auto& filterAddr : filterAddrs) { + for (const auto& [key, entry] : map) { + if (key.dstAddress == filterAddr) { + TraceEvent("InterfaceTrackerDump") + .detail("SrcProcess", srcAddr) + .detail("DstAddress", key.dstAddress) + .detail("DstRole", key.dstRole) + .detail("NumCreated", entry.numCreated) + .detail("NumDeleted", entry.numDeleted) + .detail("Delta", entry.numCreated - entry.numDeleted); + if (entry.numCreated > entry.numDeleted) { + for (const auto& rec : entry.createRecords) { + if (rec.numStreamsDeleted < rec.numStreams) { + TraceEvent("InterfaceTrackerLeaked") + .detail("SrcProcess", srcAddr) + .detail("DstAddress", key.dstAddress) + .detail("DstRole", key.dstRole) + .detail("CreateId", rec.id) + .detail("CreateTime", format("%.6f", rec.time)) + .detail("NumStreams", rec.numStreams) + .detail("NumStreamsDeleted", rec.numStreamsDeleted) + .detail("Backtrace", rec.backtrace); + } + } + } + } + } + } + for (const auto& filterAddr : filterAddrs) { + auto it = peerRefCounts.find(filterAddr); + if (it != peerRefCounts.end()) { + TraceEvent("InterfaceTrackerPeerRefRaw") + .detail("SrcProcess", srcAddr) + .detail("DstAddress", filterAddr) + .detail("TotalAdded", it->second.added) + .detail("TotalRemoved", it->second.removed) + .detail("RawDelta", it->second.added - it->second.removed); + } + } +} + FlowTransport::FlowTransport(uint64_t transportId, int maxWellKnownEndpoints, IPAllowList const* allowList) : self(new TransportData(transportId, maxWellKnownEndpoints, allowList)) { self->multiVersionCleanup = multiVersionCleanupWorker(self); @@ -1954,6 +2226,23 @@ FlowTransport::FlowTransport(uint64_t transportId, int maxWellKnownEndpoints, IP self->publicKeys.emplace(keyName, key.toPublic()); } } + g_futureRefReleasedCallback = [](int64_t id) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return; + } + if (g_network && g_network->global(INetwork::enFlowTransport)) { + FlowTransport::transport().interfaceTracker.futureRefReleased(id); + } + }; + g_futureRefCopiedCallback = [](int64_t srcId) -> int64_t { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return -1; + } + if (g_network && g_network->global(INetwork::enFlowTransport)) { + return FlowTransport::transport().interfaceTracker.futureRefCopied(srcId); + } + return -1; + }; } FlowTransport::~FlowTransport() { @@ -1980,7 +2269,11 @@ const std::unordered_map>& FlowTransport::getAll return self->peers; } -std::map>* FlowTransport::getIncompatiblePeers() { +const std::unordered_map& FlowTransport::getPersistentConnectFailedCounts() const { + return self->persistentConnectFailedCount; +} + +std::vector FlowTransport::consumeReportableIncompatiblePeers() { for (auto it = self->incompatiblePeers.begin(); it != self->incompatiblePeers.end();) { if (self->multiVersionConnections.contains(it->second.first)) { it = self->incompatiblePeers.erase(it); @@ -1988,7 +2281,16 @@ std::map>* FlowTransport::getIncompa it++; } } - return &self->incompatiblePeers; + std::vector reportable; + for (auto it = self->incompatiblePeers.begin(); it != self->incompatiblePeers.end();) { + if (now() - it->second.second > FLOW_KNOBS->INCOMPATIBLE_PEER_DELAY_BEFORE_LOGGING) { + reportable.push_back(it->first); + it = self->incompatiblePeers.erase(it); + } else { + it++; + } + } + return reportable; } Future FlowTransport::onIncompatibleChanged() { @@ -2030,6 +2332,7 @@ void FlowTransport::addPeerReference(const Endpoint& endpoint, bool isStream) { } else { peer->peerReferences++; } + interfaceTracker.peerRefAdded(endpoint.getPrimaryAddress()); } void FlowTransport::removePeerReference(const Endpoint& endpoint, bool isStream) { @@ -2038,6 +2341,19 @@ void FlowTransport::removePeerReference(const Endpoint& endpoint, bool isStream) Reference peer = self->getPeer(endpoint.getPrimaryAddress()); if (peer) { peer->peerReferences--; + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + interfaceTracker.peerRefRemovedRaw(endpoint.getPrimaryAddress()); + interfaceTracker.peerRefRemoved(endpoint.getPrimaryAddress(), endpoint.token); + // Per-token backtrace of which code path is releasing this peer ref. + // platform::get_backtrace() is expensive and this fires on every + // removePeerReference; suppressFor caps the per-event rate. + TraceEvent("PeerRefRemovedBacktrace") + .suppressFor(2.0) + .detail("PeerAddr", endpoint.getPrimaryAddress()) + .detail("Token", endpoint.token) + .detail("PeerReferences", peer->peerReferences) + .detail("Backtrace", platform::get_backtrace()); + } if (peer->peerReferences < 0) { TraceEvent(SevError, "InvalidPeerReferences") .detail("References", peer->peerReferences) @@ -2271,7 +2587,7 @@ TEST_CASE("noSim/fdbrpc/FlowTransport/PacketLimitOnSend") { TransportData transport(1, WLTOKEN_FIRST_AVAILABLE, nullptr); NetworkAddress address(IPAddress(0x7f000001), 45000, true, false); Endpoint endpoint(NetworkAddressList{ address, {} }, UID(1, 2)); - Reference peer = makeReference(&transport, address); + auto peer = makeReference(&transport, address); sendPacket(&transport, peer, SerializeSource("ok"_sr), endpoint, false); PacketBuffer* const tail = peer->unsent.getWriteBuffer(); uint32_t acceptedLength; @@ -2310,7 +2626,7 @@ TEST_CASE("noSim/fdbrpc/FlowTransport/PacketLimitOnSend") { sendPacket(&transport, peer, SerializeSource("again"_sr), endpoint, false); ASSERT_GT(tail->bytes_written, previousLength); - Reference emptyPeer = makeReference(&transport, address); + auto emptyPeer = makeReference(&transport, address); bool emptyRejected = false; try { sendPacket(&transport, emptyPeer, SerializeSource(StringRef(oversized)), endpoint, true); diff --git a/fdbrpc/HTTP.cpp b/fdbrpc/HTTP.cpp index fc817b7885f..7de0e52c23d 100644 --- a/fdbrpc/HTTP.cpp +++ b/fdbrpc/HTTP.cpp @@ -267,7 +267,7 @@ Future read_delimited_into_string(Reference conn, size_t pos) { size_t sPos = pos; int lookBack = strlen(delim) - 1; - ASSERT(lookBack >= 0); + ASSERT_GE(lookBack, 0); while (true) { size_t endPos = buf->find(delim, sPos); diff --git a/fdbrpc/HealthMonitor.cpp b/fdbrpc/HealthMonitor.cpp index 95527e3b643..3ca63736878 100644 --- a/fdbrpc/HealthMonitor.cpp +++ b/fdbrpc/HealthMonitor.cpp @@ -32,7 +32,7 @@ void HealthMonitor::purgeOutdatedHistory() { if (p.first < now() - FLOW_KNOBS->HEALTH_MONITOR_CLIENT_REQUEST_INTERVAL_SECS) { auto& count = peerClosedNum[p.second]; --count; - ASSERT(count >= 0); + ASSERT_GE(count, 0); if (count == 0) { peerClosedNum.erase(p.second); } diff --git a/fdbrpc/include/fdbrpc/AsyncFileCached.h b/fdbrpc/include/fdbrpc/AsyncFileCached.h index 59e6d67309e..fdbb6710576 100644 --- a/fdbrpc/include/fdbrpc/AsyncFileCached.h +++ b/fdbrpc/include/fdbrpc/AsyncFileCached.h @@ -162,7 +162,7 @@ class AsyncFileCached final : public IAsyncFile, public ReferenceCounted this->length) { length = int(this->length - offset); - ASSERT(length >= 0); + ASSERT_GE(length, 0); } auto f = read_write_impl(static_cast(data), length, offset); if (f.isReady() && !f.isError()) @@ -446,7 +446,7 @@ struct AFCPage : public EvictablePage, public FastAllocated { } void releaseZeroCopy() { --zeroCopyRefCount; - ASSERT(zeroCopyRefCount >= 0); + ASSERT_GE(zeroCopyRefCount, 0); } Future read(void* data, int length, int offset) { @@ -522,13 +522,13 @@ struct AFCPage : public EvictablePage, public FastAllocated { if (FLOW_KNOBS->FLOW_CACHEDFILE_WRITE_IO_SIZE > 0) { allowance = (pageCache->pageSize + FLOW_KNOBS->FLOW_CACHEDFILE_WRITE_IO_SIZE - 1) / FLOW_KNOBS->FLOW_CACHEDFILE_WRITE_IO_SIZE; // round up - ASSERT(allowance > 0); + ASSERT_GT(allowance, 0); } co_await owner->getRateControl()->getAllowance(allowance); } if (pageOffset + pageCache->pageSize > owner->length) { - ASSERT(pageOffset < owner->length); + ASSERT_LT(pageOffset, owner->length); memset(static_cast(data) + owner->length - pageOffset, 0, pageCache->pageSize - (owner->length - pageOffset)); diff --git a/fdbrpc/include/fdbrpc/AsyncFileReadAhead.h b/fdbrpc/include/fdbrpc/AsyncFileReadAhead.h index 3002286b486..e6323140ed0 100644 --- a/fdbrpc/include/fdbrpc/AsyncFileReadAhead.h +++ b/fdbrpc/include/fdbrpc/AsyncFileReadAhead.h @@ -83,7 +83,7 @@ class AsyncFileReadAheadCache final : public IAsyncFile, public ReferenceCounted // Start blocks up to the read ahead size beyond the last needed block but don't go past the end of the file int lastBlockNumInFile = ((fileSize + f->m_block_size - 1) / f->m_block_size) - 1; - ASSERT(lastBlockNum <= lastBlockNumInFile); + ASSERT_LE(lastBlockNum, lastBlockNumInFile); int lastBlockToStart = std::min(lastBlockNum + f->m_read_ahead_blocks, lastBlockNumInFile); int blockNum{ 0 }; @@ -138,7 +138,7 @@ class AsyncFileReadAheadCache final : public IAsyncFile, public ReferenceCounted } } - ASSERT(wpos == length); + ASSERT_EQ(wpos, length); ASSERT(localCache.empty()); // If the cache is too large then go through the cache in block number order and remove any entries whose future diff --git a/fdbrpc/include/fdbrpc/FlowTransport.h b/fdbrpc/include/fdbrpc/FlowTransport.h index ad68b112b7d..41f756b9167 100644 --- a/fdbrpc/include/fdbrpc/FlowTransport.h +++ b/fdbrpc/include/fdbrpc/FlowTransport.h @@ -26,6 +26,8 @@ #include #include #include +#include +#include #include "fdbrpc/DDSketch.h" #include "flow/genericactors.h" @@ -34,8 +36,111 @@ #include "flow/ProtocolVersion.h" #include "flow/Net2Packet.h" #include "flow/Arena.h" +#include "flow/Platform.h" #include "flow/PKey.h" +// Tracks creation and destruction of interface objects (StorageServerInterface, +// TLogInterface, etc.) per process. Used for debugging stale peer references. +// +// All bookkeeping methods are no-ops when FLOW_KNOBS->STALE_PEER_OBSERVABILITY +// is false. Method bodies live in fdbrpc/FlowTransport.actor.cpp to keep this +// header light. +struct InterfaceTracker { + struct Key { + NetworkAddress dstAddress; + std::string dstRole; + bool operator==(const Key& other) const { return dstAddress == other.dstAddress && dstRole == other.dstRole; } + }; + struct KeyHash { + size_t operator()(const Key& k) const { + return std::hash()(k.dstAddress) ^ std::hash()(k.dstRole); + } + }; + struct Entry { + int64_t numCreated = 0; + int64_t numDeleted = 0; + struct CreateRecord { + int64_t id; + double time; + std::string backtrace; + int numStreams; + int numStreamsDeleted = 0; + }; + std::vector createRecords; + }; + struct TokenKey { + NetworkAddress addr; + UID token; + bool operator==(const TokenKey& other) const { return addr == other.addr && token == other.token; } + }; + struct TokenKeyHash { + size_t operator()(const TokenKey& k) const { return k.addr.hash() ^ k.token.hash(); } + }; + struct TokenInfo { + std::string role; + int64_t createId; + }; + struct PeerRefCount { + int64_t added = 0; + int64_t removed = 0; + }; + struct FlowReceiverRecord { + int64_t id; + NetworkAddress addr; + UID token; + double createTime; + std::string backtrace; + std::string callerTag; + }; + struct RefRecord { + int64_t id; + NetworkAddress addr; + UID token; + double time; + std::string backtrace; + bool isFutureRef = false; + }; + + // Per (address, role): created/deleted interface counts + per-creation backtraces (the Delta source). + std::unordered_map map; + // Per (address, token): which (role, create-record) a stream token belongs to, for matching deletes to creates. + std::unordered_map tokenToInfo; + // Per address: running count of peer references added vs removed (raw peer-ref accounting). + std::unordered_map peerRefCounts; + // Live FlowReceiver records by id: who created each receiver + backtrace, erased on destroy. + std::unordered_map flowReceiverRecords; + // Live promise/future ref records by id: creation backtrace per outstanding ref, erased on release. + std::unordered_map refRecords; + // Monotonic id generators for the records above (id 0 unused; -1 means "not tracked"). + int64_t nextCreateId = 0; + int64_t nextFlowReceiverId = 0; + int64_t nextRefId = 0; + // Optional tag identifying the current call site, attached to new FlowReceiver records. + std::string currentCallerTag; + + void created(const NetworkAddress& dstAddr, const std::string& dstRole, const std::vector& tokens); + + void peerRefAdded(const NetworkAddress& addr); + void peerRefRemovedRaw(const NetworkAddress& addr); + void peerRefRemoved(const NetworkAddress& addr, const UID& token); + + int64_t getDelta(const NetworkAddress& addr, const std::string& role) const; + + int64_t flowReceiverCreated(const NetworkAddress& addr, const UID& token); + void flowReceiverDestroyed(const NetworkAddress& addr, int64_t id); + + int64_t promiseRefAdded(const NetworkAddress& addr, const UID& token); + void promiseRefReleased(int64_t id); + int64_t futureRefAdded(const NetworkAddress& addr, const UID& token); + int64_t futureRefCopied(int64_t srcId); + void futureRefReleased(int64_t id); + + void prettyPrintLeakedReceivers(const NetworkAddress& srcAddr, + const std::vector& filterAddrs) const; + void prettyPrintLeakedRefs(const NetworkAddress& srcAddr, const std::vector& filterAddrs) const; + void prettyPrint(const NetworkAddress& srcAddr, const std::vector& filterAddrs) const; +}; + class IConnection; // Applications own IDs starting at WLTOKEN_FIRST_AVAILABLE and reserve their @@ -201,6 +306,12 @@ struct Peer : public ReferenceCounted { class IPAllowList; +// Per-address connect-failure tracking on TransportData. +struct ConnectFailedInfo { + int64_t count = 0; // cumulative number of failed connect attempts + double lastFailed = 0; // now() of the most recent failure +}; + // FIXME: describe what FlowTransport represents. Is it everything // for a given process? Is it some subset of what a process uses? class FlowTransport : NonCopyable { @@ -237,10 +348,17 @@ class FlowTransport : NonCopyable { // Returns all peers that the FlowTransport is monitoring. const std::unordered_map>& getAllPeers() const; - // Returns the set of all peers that have attempted to connect, but have incompatible protocol versions - std::map>* getIncompatiblePeers(); + // Returns and removes peers whose incompatible protocol versions have persisted long enough to report. + // Peers later recognized as multi-version connections are discarded without reporting. + std::vector consumeReportableIncompatiblePeers(); + + // Returns a per-address cumulative count of connect failures. + // The map lives on TransportData and persists across Peer destruction, + // so it remains a reliable signal for catching short-lived peers that + // all point to the same address + const std::unordered_map& getPersistentConnectFailedCounts() const; - // Returns when getIncompatiblePeers has at least one peer which is incompatible. + // Returns when an incompatible peer has persisted long enough to report. Future onIncompatibleChanged(); // Signal that a peer connection is being used, even if no messages are currently being sent to the peer @@ -318,6 +436,8 @@ class FlowTransport : NonCopyable { // Periodically read JWKS (RFC 7517) public key file to refresh public key set. void watchPublicKeyFile(const std::string& publicKeyFilePath); + InterfaceTracker interfaceTracker; + private: class TransportData* self; }; diff --git a/fdbrpc/include/fdbrpc/LoadBalance.h b/fdbrpc/include/fdbrpc/LoadBalance.h index ec6c96c65c2..184f9e9a108 100644 --- a/fdbrpc/include/fdbrpc/LoadBalance.h +++ b/fdbrpc/include/fdbrpc/LoadBalance.h @@ -268,22 +268,15 @@ struct RequestData : NonCopyable { ASSERT(modelHolder->model); QueueModel* model = modelHolder->model; - if (model->laggingRequestCount > FLOW_KNOBS->MAX_LAGGING_REQUESTS_OUTSTANDING || - model->laggingRequests.isReady()) { - model->laggingRequests.cancel(); - model->laggingRequestCount = 0; - model->addActor = PromiseStream>(); - model->laggingRequests = actorCollection(model->addActor.getFuture(), &model->laggingRequestCount); - } - - // We need to process the lagging request in order to update the queue model - Reference holderCapture = std::move(modelHolder); - auto triedAllOptionsCapture = triedAllOptions; - Future updateModel = map(response, [holderCapture, triedAllOptionsCapture](Reply result) { - checkAndProcessResultImpl(result, holderCapture, AtMostOnce::False, triedAllOptionsCapture); - return Void(); + model->addLaggingRequest([this] { + // We need to process the lagging request in order to update the queue model + Reference holderCapture = std::move(modelHolder); + auto triedAllOptionsCapture = triedAllOptions; + return map(response, [holderCapture, triedAllOptionsCapture](Reply result) { + checkAndProcessResultImpl(result, holderCapture, AtMostOnce::False, triedAllOptionsCapture); + return Void(); + }); }); - model->addActor.send(updateModel); } ~RequestData() { diff --git a/fdbrpc/include/fdbrpc/QueueModel.h b/fdbrpc/include/fdbrpc/QueueModel.h index 90eece2d78d..f6e95dffa7e 100644 --- a/fdbrpc/include/fdbrpc/QueueModel.h +++ b/fdbrpc/include/fdbrpc/QueueModel.h @@ -96,9 +96,6 @@ class QueueModel { double secondMultiplier; double secondBudget; - PromiseStream> addActor; - Future laggingRequests; // requests for which a different recipient already answered - int laggingRequestCount; QueueModel() : secondMultiplier(1.0), secondBudget(0), laggingRequestCount(0) { laggingRequests = actorCollection(addActor.getFuture(), &laggingRequestCount); @@ -106,7 +103,24 @@ class QueueModel { ~QueueModel() { laggingRequests.cancel(); } + void addBackgroundActor(Future actor) { addActor.send(actor); } + + // The lagging actor must be created after an exhausted collection is cancelled. + template + void addLaggingRequest(MakeActor&& makeActor) { + if (laggingRequestCount > FLOW_KNOBS->MAX_LAGGING_REQUESTS_OUTSTANDING || laggingRequests.isReady()) { + laggingRequests.cancel(); + laggingRequestCount = 0; + addActor = PromiseStream>(); + laggingRequests = actorCollection(addActor.getFuture(), &laggingRequestCount); + } + addActor.send(makeActor()); + } + private: + PromiseStream> addActor; + Future laggingRequests; // requests for which a different recipient already answered + int laggingRequestCount; std::unordered_map data; }; diff --git a/fdbrpc/include/fdbrpc/fdbrpc.h b/fdbrpc/include/fdbrpc/fdbrpc.h index f5924bb8b44..6540f3a5615 100644 --- a/fdbrpc/include/fdbrpc/fdbrpc.h +++ b/fdbrpc/include/fdbrpc/fdbrpc.h @@ -39,13 +39,19 @@ class FlowReceiver : public NetworkMessageReceiver, public NonCopyable { Endpoint endpoint; bool m_isLocalEndpoint; bool m_stream; + int64_t m_flowReceiverId; protected: - FlowReceiver() : m_isLocalEndpoint(false), m_stream(false) {} + FlowReceiver() : m_isLocalEndpoint(false), m_stream(false), m_flowReceiverId(-1) {} FlowReceiver(Endpoint const& remoteEndpoint, bool stream) - : endpoint(remoteEndpoint), m_isLocalEndpoint(false), m_stream(stream) { + : endpoint(remoteEndpoint), m_isLocalEndpoint(false), m_stream(stream), m_flowReceiverId(-1) { FlowTransport::transport().addPeerReference(endpoint, m_stream); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && m_stream && endpoint.getPrimaryAddress().isValid() && + endpoint.getPrimaryAddress().isPublic()) { + m_flowReceiverId = FlowTransport::transport().interfaceTracker.flowReceiverCreated( + endpoint.getPrimaryAddress(), endpoint.token); + } } ~FlowReceiver() { @@ -53,6 +59,10 @@ class FlowReceiver : public NetworkMessageReceiver, public NonCopyable { FlowTransport::transport().removeEndpoint(endpoint, this); } else { FlowTransport::transport().removePeerReference(endpoint, m_stream); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && m_flowReceiverId >= 0) { + FlowTransport::transport().interfaceTracker.flowReceiverDestroyed(endpoint.getPrimaryAddress(), + m_flowReceiverId); + } } } @@ -66,6 +76,11 @@ class FlowReceiver : public NetworkMessageReceiver, public NonCopyable { endpoint = remoteEndpoint; m_stream = stream; FlowTransport::transport().addPeerReference(endpoint, m_stream); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && m_stream && endpoint.getPrimaryAddress().isValid() && + endpoint.getPrimaryAddress().isPublic()) { + m_flowReceiverId = FlowTransport::transport().interfaceTracker.flowReceiverCreated( + endpoint.getPrimaryAddress(), endpoint.token); + } } // If already a remote endpoint, returns that. Otherwise makes this @@ -181,6 +196,8 @@ class ReplyPromise final : public ComposedIdentifier { networkSender(Uncancellable(), getFuture(), &sav->getRawEndpoint()); } + // Acquiring each incoming reference before releasing its old reference preserves self-assignment. + // NOLINTNEXTLINE(bugprone-unhandled-self-assignment) void operator=(const ReplyPromise& rhs) { if (rhs.sav) rhs.sav->addPromiseRef(); @@ -597,7 +614,7 @@ class ReplyPromiseStream { // Must be called on the server before sending results on the stream to ratelimit the amount of data outstanding to // the client Future onReady() const { - ASSERT(queue->acknowledgements.bytesLimit > 0); + ASSERT_GT(queue->acknowledgements.bytesLimit, 0); if (queue->acknowledgements.failures.isError()) { return queue->acknowledgements.failures.getError(); } @@ -618,6 +635,8 @@ class ReplyPromiseStream { // client void setByteLimit(int64_t byteLimit) const { queue->acknowledgements.bytesLimit = byteLimit; } + // Acquiring each incoming reference before releasing its old reference preserves self-assignment. + // NOLINTNEXTLINE(bugprone-unhandled-self-assignment) void operator=(const ReplyPromiseStream& rhs) { rhs.queue->addPromiseRef(); if (queue) @@ -908,35 +927,88 @@ class RequestStream { return getReplyUnlessFailedFor(ReplyPromise(), sustainedFailureDuration, sustainedFailureSlope); } - explicit RequestStream(const Endpoint& endpoint) : queue(new NetNotifiedQueue(0, 1, endpoint)) {} + explicit RequestStream(const Endpoint& endpoint) + : queue(new NetNotifiedQueue(0, 1, endpoint)), m_promiseRefTrackingId(-1) { + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && queue->isRemoteEndpoint() && g_network && + g_network->global(INetwork::enFlowTransport)) { + m_promiseRefTrackingId = FlowTransport::transport().interfaceTracker.promiseRefAdded( + endpoint.getPrimaryAddress(), endpoint.token); + } + } SWIFT_CXX_IMPORT_UNSAFE FutureStream getFuture() const { queue->addFutureRef(); - return FutureStream(queue); + int64_t trackingId = -1; + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && queue->isRemoteEndpoint() && g_network && + g_network->global(INetwork::enFlowTransport)) { + const auto& ep = queue->getEndpoint(TaskPriority::DefaultEndpoint); + trackingId = FlowTransport::transport().interfaceTracker.futureRefAdded(ep.getPrimaryAddress(), ep.token); + } + return FutureStream(queue, trackingId); } - RequestStream() : queue(new NetNotifiedQueue(0, 1)) {} + RequestStream() : queue(new NetNotifiedQueue(0, 1)), m_promiseRefTrackingId(-1) {} explicit RequestStream(PeerCompatibilityPolicy policy) : RequestStream() { queue->setPeerCompatibilityPolicy(policy); } - RequestStream(const RequestStream& rhs) : queue(rhs.queue) { queue->addPromiseRef(); } - RequestStream(RequestStream&& rhs) noexcept : queue(rhs.queue) { rhs.queue = 0; } + RequestStream(const RequestStream& rhs) : queue(rhs.queue), m_promiseRefTrackingId(-1) { + queue->addPromiseRef(); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && queue->isRemoteEndpoint() && g_network && + g_network->global(INetwork::enFlowTransport)) { + const auto& ep = queue->getEndpoint(TaskPriority::DefaultEndpoint); + m_promiseRefTrackingId = + FlowTransport::transport().interfaceTracker.promiseRefAdded(ep.getPrimaryAddress(), ep.token); + } + } + RequestStream(RequestStream&& rhs) noexcept : queue(rhs.queue), m_promiseRefTrackingId(rhs.m_promiseRefTrackingId) { + rhs.queue = 0; + // Transfer promise-ref tracking ownership to the moved-to object: clear rhs's id so its + // destructor does not release a tracking record now owned by *this (avoids double-release). + rhs.m_promiseRefTrackingId = -1; + } + // Acquiring each incoming reference before releasing its old reference preserves self-assignment. + // NOLINTNEXTLINE(bugprone-unhandled-self-assignment) void operator=(const RequestStream& rhs) { rhs.queue->addPromiseRef(); - if (queue) + int64_t newTrackingId = -1; + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && rhs.queue->isRemoteEndpoint() && g_network && + g_network->global(INetwork::enFlowTransport)) { + const auto& ep = rhs.queue->getEndpoint(TaskPriority::DefaultEndpoint); + newTrackingId = + FlowTransport::transport().interfaceTracker.promiseRefAdded(ep.getPrimaryAddress(), ep.token); + } + if (queue) { queue->delPromiseRef(); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && m_promiseRefTrackingId >= 0 && g_network && + g_network->global(INetwork::enFlowTransport)) { + FlowTransport::transport().interfaceTracker.promiseRefReleased(m_promiseRefTrackingId); + } + } queue = rhs.queue; + m_promiseRefTrackingId = newTrackingId; } void operator=(RequestStream&& rhs) noexcept { if (queue != rhs.queue) { - if (queue) + if (queue) { queue->delPromiseRef(); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && m_promiseRefTrackingId >= 0 && g_network && + g_network->global(INetwork::enFlowTransport)) { + FlowTransport::transport().interfaceTracker.promiseRefReleased(m_promiseRefTrackingId); + } + } queue = rhs.queue; + m_promiseRefTrackingId = rhs.m_promiseRefTrackingId; rhs.queue = 0; + rhs.m_promiseRefTrackingId = -1; } } ~RequestStream() { - if (queue) + if (queue) { queue->delPromiseRef(); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && m_promiseRefTrackingId >= 0 && g_network && + g_network->global(INetwork::enFlowTransport)) { + FlowTransport::transport().interfaceTracker.promiseRefReleased(m_promiseRefTrackingId); + } + } // queue = (NetNotifiedQueue*)0xdeadbeef; } @@ -959,6 +1031,11 @@ class RequestStream { private: NetNotifiedQueue* queue; + // InterfaceTracker id for this stream's promise-ref, or -1 when not tracked. It stays -1 + // whenever STALE_PEER_OBSERVABILITY is off (the gated *RefAdded blocks never run), so the + // plain transfers of this member in the copy/move/assignment paths are no-ops in that case + // and need no knob guard; only the tracker calls (promiseRefAdded/promiseRefReleased) are gated. + int64_t m_promiseRefTrackingId; }; // Public request streams require T::verify() and reject unauthorized messages. diff --git a/fdbrpc/sim2.cpp b/fdbrpc/sim2.cpp index 2f8a15a1f0e..4eb6d8f4108 100644 --- a/fdbrpc/sim2.cpp +++ b/fdbrpc/sim2.cpp @@ -2968,7 +2968,8 @@ Future waitUntilDiskReady(Reference diskParameters, int64_ if (diskParameters->nextOperation < now()) diskParameters->nextOperation = now(); - diskParameters->nextOperation += (1.0 / diskParameters->iops) + (size / diskParameters->bandwidth); + diskParameters->nextOperation += + (1.0 / diskParameters->iops) + (static_cast(size) / diskParameters->bandwidth); double randomLatency; if (sync) { diff --git a/fdbrpc/tests/CMakeLists.txt b/fdbrpc/tests/CMakeLists.txt index a0ba2a2dbba..5a46faf07e6 100644 --- a/fdbrpc/tests/CMakeLists.txt +++ b/fdbrpc/tests/CMakeLists.txt @@ -13,3 +13,16 @@ if(WITH_GRPC) endif() add_flow_target(EXECUTABLE NAME fdbrpc_transport_bench SRCS fdbrpc_bench.cpp) target_link_libraries(fdbrpc_transport_bench PUBLIC flow fdbrpc boost_target_program_options) + +add_flow_target(EXECUTABLE NAME fdbrpc_network_test SRCS networktest_main.cpp networktest.cpp NetworkTest.h) +target_link_libraries(fdbrpc_network_test PRIVATE flow fdbrpc fmt::fmt) + +if(FDB_REGISTER_UNIT_TESTS AND Python3_EXECUTABLE) + add_dependencies(unit_tests fdbrpc_network_test) + add_test(NAME unit/fdbrpc_network_test/native + COMMAND ${Python3_EXECUTABLE} ${CMAKE_CURRENT_SOURCE_DIR}/networktest_smoke.py $) + set_tests_properties(unit/fdbrpc_network_test/native PROPERTIES + LABELS "unit;native;fdbrpc_network_test" + TIMEOUT 90 + ENVIRONMENT "${SANITIZER_OPTIONS}") +endif() diff --git a/fdbserver/include/fdbserver/NetworkTest.h b/fdbrpc/tests/NetworkTest.h similarity index 58% rename from fdbserver/include/fdbserver/NetworkTest.h rename to fdbrpc/tests/NetworkTest.h index ae51501c22b..f0cd7a92812 100644 --- a/fdbserver/include/fdbserver/NetworkTest.h +++ b/fdbrpc/tests/NetworkTest.h @@ -18,14 +18,15 @@ * limitations under the License. */ -#ifndef FDBSERVER_NETWORKTEST_H -#define FDBSERVER_NETWORKTEST_H +#ifndef FDBRPC_TESTS_NETWORKTEST_H +#define FDBRPC_TESTS_NETWORKTEST_H #pragma once -#include "fdbclient/FDBTypes.h" #include "fdbrpc/fdbrpc.h" #include "flow/FileIdentifier.h" +constexpr int WLTOKEN_NETWORKTEST = WLTOKEN_FIRST_AVAILABLE; + struct NetworkTestInterface { RequestStream test; NetworkTestInterface() = default; @@ -35,9 +36,9 @@ struct NetworkTestInterface { struct NetworkTestReply { constexpr static FileIdentifier file_identifier = 14465374; - Value value; + Standalone value; NetworkTestReply() = default; - explicit NetworkTestReply(Value value) : value(value) {} + explicit NetworkTestReply(Standalone value) : value(value) {} template void serialize(Ar& ar) { serializer(ar, value); @@ -46,11 +47,11 @@ struct NetworkTestReply { struct NetworkTestRequest { constexpr static FileIdentifier file_identifier = 4146513; - Key key; + Standalone key; uint32_t replySize; ReplyPromise reply; NetworkTestRequest() = default; - NetworkTestRequest(Key key, uint32_t replySize) : key(key), replySize(replySize) {} + NetworkTestRequest(Standalone key, uint32_t replySize) : key(key), replySize(replySize) {} template void serialize(Ar& ar) { serializer(ar, key, replySize, reply); @@ -61,4 +62,33 @@ Future networkTestServer(); Future networkTestClient(std::string const& testServers); +class NetworkTestIntRange { +public: + NetworkTestIntRange() = default; + NetworkTestIntRange(int low, int high); + + int get() const; + int maximum() const { return max; } + std::string toString() const; + +private: + int min = 0; + int max = 0; +}; + +struct P2PNetworkTestOptions { + std::vector listenerAddresses; + std::vector remoteAddresses; + int connectionsOut = 1; + NetworkTestIntRange requestBytes{ 50, 100 }; + NetworkTestIntRange replyBytes{ 500, 1000 }; + NetworkTestIntRange requests{ 10, 10000 }; + NetworkTestIntRange idleMilliseconds; + NetworkTestIntRange waitReadMilliseconds; + NetworkTestIntRange waitWriteMilliseconds; + double targetDuration = 0.0; +}; + +Future networkTestP2P(P2PNetworkTestOptions options, bool oneshot); + #endif diff --git a/fdbserver/networktest.cpp b/fdbrpc/tests/networktest.cpp similarity index 81% rename from fdbserver/networktest.cpp rename to fdbrpc/tests/networktest.cpp index d8f6473cbc6..c6739106a14 100644 --- a/fdbserver/networktest.cpp +++ b/fdbrpc/tests/networktest.cpp @@ -19,17 +19,16 @@ */ #include "fmt/format.h" -#include "fdbserver/NetworkTest.h" +#include "NetworkTest.h" #include "flow/ActorCollection.h" #include "flow/CoroUtils.h" #include "flow/Knobs.h" -#include "flow/UnitTest.h" +#include #include +#include #include "flow/IConnection.h" -constexpr int WLTOKEN_NETWORKTEST = WLTOKEN_FIRST_AVAILABLE; - struct LatencyStats { using sample = double; double x = 0; @@ -75,7 +74,7 @@ class NetworkTestServer { while (true) { NetworkTestRequest req = co_await interf.test.getFuture(); LatencyStats::sample sample = latency.tick(); - req.reply.send(NetworkTestReply(Value(std::string(req.replySize, '.')))); + req.reply.send(NetworkTestReply(Standalone(std::string(req.replySize, '.')))); latency.tock(sample); sent++; } @@ -246,31 +245,20 @@ Future networkTestClient(std::string const& testServers) { co_await waitForAll(clients); } -struct RandomIntRange { - int min; - int max; - - explicit(false) RandomIntRange(int low = 0, int high = 0) : min(low), max(high) {} - - // Accepts strings of the form "min:max" or "N" - // where N will be used for both min and max - explicit(false) RandomIntRange(std::string str) { - StringRef high = str; - StringRef low = high.eat(":"); - if (high.empty()) { - high = low; - } - min = low.empty() ? 0 : atol(low.toString().c_str()); - max = high.empty() ? 0 : atol(high.toString().c_str()); - if (min > max) { - std::swap(min, max); - } +NetworkTestIntRange::NetworkTestIntRange(int low, int high) : min(std::min(low, high)), max(std::max(low, high)) { + // Sampling uses an exclusive upper bound one greater than max. + if (min < 0 || max == std::numeric_limits::max()) { + throw invalid_option_value(); } +} - int get() const { return (max == 0) ? 0 : nondeterministicRandom()->randomInt(min, max + 1); } +int NetworkTestIntRange::get() const { + return (max == 0) ? 0 : nondeterministicRandom()->randomInt(min, max + 1); +} - std::string toString() const { return format("%d:%d", min, max); } -}; +std::string NetworkTestIntRange::toString() const { + return format("%d:%d", min, max); +} struct P2PNetworkTest { // Addresses to listen on @@ -280,17 +268,17 @@ struct P2PNetworkTest { // Number of outgoing connections to maintain int connectionsOut; // Message size range to send on outgoing established connections - RandomIntRange requestBytes; + NetworkTestIntRange requestBytes; // Message size to reply with on incoming established connections - RandomIntRange replyBytes; + NetworkTestIntRange replyBytes; // Number of requests/replies per session - RandomIntRange requests; + NetworkTestIntRange requests; // Delay after message send and receive are complete before closing connection - RandomIntRange idleMilliseconds; + NetworkTestIntRange idleMilliseconds; // Random delay before socket reads - RandomIntRange waitReadMilliseconds; + NetworkTestIntRange waitReadMilliseconds; // Random delay before socket writes - RandomIntRange waitWriteMilliseconds; + NetworkTestIntRange waitWriteMilliseconds; double targetDuration; double startTime; double globalStartTime; @@ -330,20 +318,11 @@ struct P2PNetworkTest { P2PNetworkTest() = default; - P2PNetworkTest(std::string listenerAddresses, - std::string remoteAddresses, - int connectionsOut, - RandomIntRange sendMsgBytes, - RandomIntRange recvMsgBytes, - RandomIntRange requests, - RandomIntRange idleMilliseconds, - RandomIntRange waitReadMilliseconds, - RandomIntRange waitWriteMilliseconds, - double targetDuration, - bool oneshot) - : connectionsOut(connectionsOut), requestBytes(sendMsgBytes), replyBytes(recvMsgBytes), requests(requests), - idleMilliseconds(idleMilliseconds), waitReadMilliseconds(waitReadMilliseconds), - waitWriteMilliseconds(waitWriteMilliseconds), targetDuration(targetDuration), oneshot(oneshot) { + P2PNetworkTest(const P2PNetworkTestOptions& options, bool oneshot) + : remotes(options.remoteAddresses), connectionsOut(options.connectionsOut), requestBytes(options.requestBytes), + replyBytes(options.replyBytes), requests(options.requests), idleMilliseconds(options.idleMilliseconds), + waitReadMilliseconds(options.waitReadMilliseconds), waitWriteMilliseconds(options.waitWriteMilliseconds), + targetDuration(options.targetDuration), oneshot(oneshot) { bytesSent = 0; bytesReceived = 0; sessionsIn = 0; @@ -351,16 +330,10 @@ struct P2PNetworkTest { connectErrors = 0; acceptErrors = 0; sessionErrors = 0; - msgBuffer = makeString(std::max(sendMsgBytes.max, recvMsgBytes.max)); + msgBuffer = makeString(std::max(requestBytes.maximum(), replyBytes.maximum())); - if (!remoteAddresses.empty()) { - remotes = NetworkAddress::parseList(remoteAddresses); - } - - if (!listenerAddresses.empty()) { - for (auto a : NetworkAddress::parseList(listenerAddresses)) { - listeners.push_back(INetworkConnections::net()->listen(a)); - } + for (auto address : options.listenerAddresses) { + listeners.push_back(INetworkConnections::net()->listen(address)); } } @@ -622,43 +595,13 @@ struct P2PNetworkTest { // Each instance // - listens on 0 or more listenerAddresses // - maintains 0 or more connectionsOut at a time, each to a random choice from remoteAddresses -// Address lists are a string of comma-separated IP:port[:tls] strings. -// -// The other arguments can be specified as "fixedValue" or "minValue:maxValue". // Each outgoing connection will live for a random requests count. // Each request will // - send a random requestBytes sized message // - wait for a random replyBytes sized response. // The client will close the connection after a random idleMilliseconds. // Reads and writes can optionally preceded by random delays, waitReadMilliseconds and waitWriteMilliseconds. -TEST_CASE(":/network/p2ptest") { - P2PNetworkTest p2p(params.get("listenerAddresses").orDefault(""), - params.get("remoteAddresses").orDefault(""), - params.getInt("connectionsOut").orDefault(1), - params.get("requestBytes").orDefault("50:100"), - params.get("replyBytes").orDefault("500:1000"), - params.get("requests").orDefault("10:10000"), - params.get("idleMilliseconds").orDefault("0"), - params.get("waitReadMilliseconds").orDefault("0"), - params.get("waitWriteMilliseconds").orDefault("0"), - params.getDouble("targetDuration").orDefault(0.0), - false); - - co_await p2p.run(); -} - -TEST_CASE(":/network/p2poneshottest") { - P2PNetworkTest p2p(params.get("listenerAddresses").orDefault(""), - params.get("remoteAddresses").orDefault(""), - params.getInt("connectionsOut").orDefault(1), - params.get("requestBytes").orDefault("50:100"), - params.get("replyBytes").orDefault("500:1000"), - params.get("requests").orDefault("10:10000"), - params.get("idleMilliseconds").orDefault("0"), - params.get("waitReadMilliseconds").orDefault("0"), - params.get("waitWriteMilliseconds").orDefault("0"), - params.getDouble("targetDuration").orDefault(0.0), - true); - +Future networkTestP2P(P2PNetworkTestOptions options, bool oneshot) { + P2PNetworkTest p2p(options, oneshot); co_await p2p.run(); } diff --git a/fdbrpc/tests/networktest.md b/fdbrpc/tests/networktest.md new file mode 100644 index 00000000000..901638e7ad9 --- /dev/null +++ b/fdbrpc/tests/networktest.md @@ -0,0 +1,67 @@ +# Network diagnostics + +`fdbrpc_network_test` runs RPC and socket diagnostics using Flow and fdbrpc. +Build it with: + +```sh +cmake --build build --target fdbrpc_network_test +``` + +The executable is `build/bin/fdbrpc_network_test`. It is a development tool; +copy it to each participating host when testing across machines. + +## RPC request/reply + +Start a server, then run a client in another terminal: + +```sh +build/bin/fdbrpc_network_test --mode server --public-address 127.0.0.1:4500 +build/bin/fdbrpc_network_test --mode client --testservers 127.0.0.1:4500 \ + --knob_network_test_script_mode=true +``` + +The client supports comma-separated server addresses. Flow knobs include +`network_test_request_size`, `network_test_reply_size`, +`network_test_client_count`, and `network_test_request_count`. Script mode reports +one measurement after a warmup interval and then finishes. Without a request +limit or script mode, the client runs until stopped. The server runs until stopped. + +`--listen-address` overrides the bind address while `--public-address` specifies +the advertised endpoint. Use `--help` for TLS and tracing options. + +## Raw connections and TLS + +P2P mode can listen, connect, or do both. For example, a bounded loopback run: + +```sh +build/bin/fdbrpc_network_test --mode p2p \ + --test_listenerAddresses=127.0.0.1:4501 \ + --test_remoteAddresses=127.0.0.1:4501 \ + --test_connectionsOut=2 --test_targetDuration=5 +``` + +The `--test_*` options control payload size ranges, requests per +connection, and delays before reads, writes, and connection close. A duration +of zero means run until stopped. `--mode p2p-oneshot` tests one connection and +handshake without sending the traffic workload; run its listener and connector +as separate processes. + +Append `:tls` to addresses to enable TLS and supply `--tls_certificate_file`, +`--tls_key_file`, `--tls_ca_file`, and `--tls_verify_peers` as appropriate. See +[`contrib/mtlsbenchmark`](../../contrib/mtlsbenchmark/readme.md) for a two-process +TLS example with handshake knobs. + +## Smoke test + +The target's CTest entry runs bounded loopback checks for RPC traffic, P2P +sessions, handshake-only mode, and invalid options: + +```sh +ctest --test-dir build -R '^unit/fdbrpc_network_test/native$' --output-on-failure +``` + +To run the same checks with temporary TLS certificates (requires OpenSSL): + +```sh +python3 fdbrpc/tests/networktest_smoke.py build/bin/fdbrpc_network_test --tls +``` diff --git a/fdbrpc/tests/networktest_main.cpp b/fdbrpc/tests/networktest_main.cpp new file mode 100644 index 00000000000..92ccbcc801d --- /dev/null +++ b/fdbrpc/tests/networktest_main.cpp @@ -0,0 +1,399 @@ +/* + * networktest_main.cpp + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "NetworkTest.h" +#include "SimpleOpt/SimpleOpt.h" +#include "fdbrpc/FlowTransport.h" +#include "fdbrpc/Net2FileSystem.h" +#include "flow/ArgParseUtil.h" +#include "flow/BooleanParam.h" +#include "flow/Knobs.h" +#include "flow/Platform.h" +#include "flow/TLSConfig.h" +#include "flow/Trace.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +FDB_BOOLEAN_PARAM(Randomize); +FDB_BOOLEAN_PARAM(IsSimulated); + +namespace { + +enum Option { + OPT_HELP, + OPT_MODE, + OPT_TESTSERVERS, + OPT_PUBLIC_ADDRESS, + OPT_LISTEN_ADDRESS, + OPT_P2P_OPTION, + OPT_KNOB, + OPT_TRACE_DIR, +}; + +CSimpleOpt::SOption options[] = { { OPT_HELP, "-h", SO_NONE }, + { OPT_HELP, "--help", SO_NONE }, + { OPT_MODE, "-m", SO_REQ_SEP }, + { OPT_MODE, "--mode", SO_REQ_SEP }, + { OPT_TESTSERVERS, "--testservers", SO_REQ_SEP }, + { OPT_PUBLIC_ADDRESS, "-p", SO_REQ_SEP }, + { OPT_PUBLIC_ADDRESS, "--public-address", SO_REQ_SEP }, + { OPT_LISTEN_ADDRESS, "-l", SO_REQ_SEP }, + { OPT_LISTEN_ADDRESS, "--listen-address", SO_REQ_SEP }, + { OPT_P2P_OPTION, "--test-", SO_REQ_SEP }, + { OPT_KNOB, "--knob-", SO_REQ_SEP }, + { OPT_TRACE_DIR, "--trace-dir", SO_REQ_SEP }, + { OPT_TRACE_DIR, "--logdir", SO_REQ_SEP }, + TLS_OPTION_FLAGS, + SO_END_OF_OPTIONS }; + +struct Options { + std::string mode; + std::string testServers; + std::vector publicAddresses; + std::vector listenAddresses; + P2PNetworkTestOptions p2pOptions; + bool hasP2POptions = false; + TLSConfig tlsConfig{ TLSEndpointType::SERVER }; + std::string traceDir = "."; + bool showHelp = false; +}; + +void printUsage(const char* program) { + printf("Usage: %s --mode MODE [OPTIONS]\n" + "\n" + "Modes:\n" + " server Serve RPC network-test requests\n" + " client Send RPC requests to --testservers ADDRESS[,ADDRESS...]\n" + " p2p Exercise raw connections and traffic\n" + " p2p-oneshot Exercise connection handshakes without traffic\n" + "\n" + "Options:\n" + " -p, --public-address ADDR RPC server public IP:PORT[:tls]\n" + " -l, --listen-address ADDR RPC server bind address (default: public)\n" + " --testservers ADDRS RPC server addresses, or nanosleep\n" + " --test_NAME VALUE P2P parameter (e.g. listenerAddresses,\n" + " remoteAddresses, targetDuration, connectionsOut)\n" + " --knob_NAME VALUE Override a Flow knob\n" + " --trace-dir DIR Trace directory (default: .)\n" + " -h, --help Show this help\n" + "\n" + "Public/listen addresses accept comma-separated lists or repeated options,\n" + "with at most two addresses. A listen address may be 'public'.\n" + "Option names accept either hyphens or underscores.\n" + "\n%s", + program, + TLS_HELP); +} + +void appendAddresses(std::vector& addresses, const char* text) { + std::string remaining(text); + for (;;) { + const auto comma = remaining.find(','); + addresses.push_back(remaining.substr(0, comma)); + if (comma == std::string::npos) { + return; + } + remaining.erase(0, comma + 1); + } +} + +Optional parseNonnegativeInt(std::string_view text, int maximum = std::numeric_limits::max()) { + int value; + const auto [end, error] = std::from_chars(text.data(), text.data() + text.size(), value); + if (error != std::errc() || end != text.data() + text.size() || value < 0 || value > maximum) { + return {}; + } + return value; +} + +Optional parseRange(std::string_view text) { + const auto colon = text.find(':'); + const auto low = parseNonnegativeInt(text.substr(0, colon), std::numeric_limits::max() - 1); + const auto high = colon == std::string_view::npos + ? low + : parseNonnegativeInt(text.substr(colon + 1), std::numeric_limits::max() - 1); + if (!low.present() || !high.present()) { + return {}; + } + return NetworkTestIntRange(low.get(), high.get()); +} + +bool parseP2POption(P2PNetworkTestOptions& options, const std::string& name, const std::string& value) { + if (name == "listenerAddresses" || name == "remoteAddresses") { + std::vector addresses; + if (!value.empty()) { + addresses = NetworkAddress::parseList(value); + if (addresses.empty() || !std::all_of(addresses.begin(), addresses.end(), [](const auto& address) { + return address.isValid(); + })) { + return false; + } + } + (name == "listenerAddresses" ? options.listenerAddresses : options.remoteAddresses) = std::move(addresses); + return true; + } + if (name == "connectionsOut") { + const auto count = parseNonnegativeInt(value); + if (!count.present()) { + return false; + } + options.connectionsOut = count.get(); + return true; + } + if (name == "targetDuration") { + try { + size_t end; + const auto duration = std::stod(value, &end); + if (end == value.size() && std::isfinite(duration) && duration >= 0) { + options.targetDuration = duration; + return true; + } + } catch (const std::exception&) { + } + return false; + } + NetworkTestIntRange* range = nullptr; + if (name == "requestBytes") { + range = &options.requestBytes; + } else if (name == "replyBytes") { + range = &options.replyBytes; + } else if (name == "requests") { + range = &options.requests; + } else if (name == "idleMilliseconds") { + range = &options.idleMilliseconds; + } else if (name == "waitReadMilliseconds") { + range = &options.waitReadMilliseconds; + } else if (name == "waitWriteMilliseconds") { + range = &options.waitWriteMilliseconds; + } + const auto parsed = parseRange(value); + if (!range || !parsed.present()) { + return false; + } + *range = parsed.get(); + return true; +} + +bool parseArgs(int argc, char** argv, Options& result, FlowKnobs& knobs) { + CSimpleOpt args(argc, argv, options, SO_O_EXACT | SO_O_HYPHEN_TO_UNDERSCORE); + while (args.Next()) { + if (args.LastError() != SO_SUCCESS) { + fprintf(stderr, "ERROR: Invalid or incomplete option '%s'\n", args.OptionText()); + return false; + } + switch (args.OptionId()) { + case OPT_HELP: + result.showHelp = true; + return true; + case OPT_MODE: + result.mode = args.OptionArg(); + break; + case OPT_TESTSERVERS: + result.testServers = args.OptionArg(); + break; + case OPT_PUBLIC_ADDRESS: + appendAddresses(result.publicAddresses, args.OptionArg()); + break; + case OPT_LISTEN_ADDRESS: + appendAddresses(result.listenAddresses, args.OptionArg()); + break; + case OPT_P2P_OPTION: { + auto name = extractPrefixedArgument("--test", args.OptionSyntax()); + if (!name.present() || name.get().empty()) { + return false; + } + if (!parseP2POption(result.p2pOptions, name.get(), args.OptionArg())) { + fprintf(stderr, "ERROR: Invalid P2P option --test_%s=%s\n", name.get().c_str(), args.OptionArg()); + return false; + } + result.hasP2POptions = true; + break; + } + case OPT_KNOB: { + auto name = extractPrefixedArgument("--knob", args.OptionSyntax()); + if (!name.present() || name.get().empty()) { + return false; + } + const auto value = knobs.parseKnobValue(name.get(), args.OptionArg()); + const bool set = std::visit( + [&](const auto& parsed) { + if constexpr (std::is_same_v, NoKnobFound>) { + return false; + } else { + return knobs.setKnob(name.get(), parsed); + } + }, + value); + if (!set) { + fprintf(stderr, "ERROR: Unknown Flow knob '%s'\n", name.get().c_str()); + return false; + } + break; + } + case OPT_TRACE_DIR: + result.traceDir = args.OptionArg(); + break; + case TLSConfig::OPT_TLS_PLUGIN: + break; + case TLSConfig::OPT_TLS_CERTIFICATES: + result.tlsConfig.setCertificatePath(args.OptionArg()); + break; + case TLSConfig::OPT_TLS_KEY: + result.tlsConfig.setKeyPath(args.OptionArg()); + break; + case TLSConfig::OPT_TLS_CA_FILE: + result.tlsConfig.setCAPath(args.OptionArg()); + break; + case TLSConfig::OPT_TLS_PASSWORD: + result.tlsConfig.setPassword(args.OptionArg()); + break; + case TLSConfig::OPT_TLS_VERIFY_PEERS: + result.tlsConfig.addVerifyPeers(args.OptionArg()); + break; + case TLSConfig::OPT_TLS_DISABLE_PLAINTEXT_CONNECTION: + result.tlsConfig.setDisablePlainTextConnection(true); + break; + } + } + if (args.FileCount() != 0 || + (result.mode != "client" && result.mode != "server" && result.mode != "p2p" && result.mode != "p2p-oneshot")) { + fprintf(stderr, "ERROR: Expected --mode client, server, p2p, or p2p-oneshot\n"); + return false; + } + if (result.mode == "server") { + if (result.publicAddresses.empty() || result.publicAddresses.size() > 2 || + (!result.listenAddresses.empty() && result.listenAddresses.size() != result.publicAddresses.size())) { + fprintf(stderr, "ERROR: Server requires one or two public addresses and matching listen addresses\n"); + return false; + } + result.listenAddresses.resize(result.publicAddresses.size(), "public"); + } else if (!result.publicAddresses.empty() || !result.listenAddresses.empty()) { + fprintf(stderr, "ERROR: --public-address and --listen-address require --mode server\n"); + return false; + } + if ((result.mode == "client") != !result.testServers.empty()) { + fprintf(stderr, "ERROR: --testservers is required for client mode and is only valid in client mode\n"); + return false; + } + if (result.mode == "p2p" || result.mode == "p2p-oneshot") { + if (result.p2pOptions.listenerAddresses.empty() && + (result.p2pOptions.remoteAddresses.empty() || result.p2pOptions.connectionsOut == 0)) { + fprintf(stderr, "ERROR: P2P mode requires a listener or a remote with positive connectionsOut\n"); + return false; + } + return true; + } + if (result.hasP2POptions) { + fprintf(stderr, "ERROR: --test_NAME parameters require a P2P mode\n"); + return false; + } + return true; +} + +Future stopNetworkAfter(Future work) { + try { + co_await work; + } catch (Error&) { + g_network->stop(); + throw; + } + g_network->stop(); +} + +} // namespace + +int main(int argc, char** argv) { + try { + platformInit(); + Error::init(); + setvbuf(stdout, nullptr, _IOLBF, BUFSIZ); + setvbuf(stderr, nullptr, _IOLBF, BUFSIZ); + setThreadLocalDeterministicRandomSeed(platform::getRandomSeed()); + // Network and trace globals retain these knobs through process shutdown. + auto* knobs = new FlowKnobs(Randomize::False, IsSimulated::False); + FLOW_KNOBS = knobs; + Options opts; + if (!parseArgs(argc, argv, opts, *knobs)) { + printUsage(argv[0]); + return 1; + } + if (opts.showHelp) { + printUsage(argv[0]); + return 0; + } + + TraceEvent::setNetworkThread(); + g_network = newNet2(opts.tlsConfig, false, true); + g_network->addStopCallback(Net2FileSystem::stop); + Net2FileSystem::newFileSystem(); + FlowTransport::createInstance(false, 1, WLTOKEN_NETWORKTEST + 1); + openTraceFile({}, 10 << 20, 10 << 20, opts.traceDir, "networktest"); + g_network->initTLS(); + g_network->initMetrics(); + FlowTransport::transport().initMetrics(); + + std::vector> work; + if (opts.mode == "server") { + for (size_t i = 0; i < opts.publicAddresses.size(); ++i) { + const auto publicAddress = NetworkAddress::parse(opts.publicAddresses[i]); + const auto listenAddress = opts.listenAddresses[i] == "public" + ? publicAddress + : NetworkAddress::parse(opts.listenAddresses[i]); + if (!publicAddress.isValid() || !listenAddress.isValid() || + publicAddress.isTLS() != listenAddress.isTLS()) { + fprintf(stderr, "ERROR: Public/listen addresses must be valid and use matching TLS settings\n"); + return 1; + } + auto listenError = FlowTransport::transport().bind(publicAddress, listenAddress); + if (listenError.isReady()) { + listenError.get(); + } + work.push_back(listenError); + printf("Listener: %s\n", listenAddress.toString().c_str()); + } + work.push_back(networkTestServer()); + } else if (opts.mode == "client") { + work.push_back(networkTestClient(opts.testServers)); + } else { + work.push_back(networkTestP2P(opts.p2pOptions, opts.mode == "p2p-oneshot")); + } + Future done = stopNetworkAfter(waitForAny(work)); + g_network->run(); + flushTraceFileVoid(); + done.get(); + return 0; + } catch (Error& e) { + fprintf(stderr, "ERROR: Network test failed: %s (%d)\n", e.what(), e.code()); + } catch (std::exception& e) { + fprintf(stderr, "ERROR: Network test failed: %s\n", e.what()); + } + flushTraceFileVoid(); + return 1; +} diff --git a/fdbrpc/tests/networktest_smoke.py b/fdbrpc/tests/networktest_smoke.py new file mode 100644 index 00000000000..956286d620b --- /dev/null +++ b/fdbrpc/tests/networktest_smoke.py @@ -0,0 +1,250 @@ +#!/usr/bin/env python3 +"""Exercise the standalone network diagnostic over bounded loopback connections.""" + +import argparse +import contextlib +import re +import socket +import subprocess +import tempfile +import time +from pathlib import Path + + +def unused_port(): + with socket.socket() as sock: + sock.bind(("127.0.0.1", 0)) + return sock.getsockname()[1] + + +class SmokeTest: + def __init__(self, executable, root): + self.executable = executable + self.root = root + self.deadline = time.monotonic() + 75 + self.tls_options = [] + self.tls_suffix = "" + + def timeout(self, seconds): + remaining = self.deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError("network test smoke exceeded its deadline") + return min(seconds, remaining) + + @contextlib.contextmanager + def process(self, name, arguments): + directory = self.root / name + directory.mkdir() + output = directory / "output.log" + with output.open("w") as log: + process = subprocess.Popen( + [self.executable, *arguments, *self.tls_options], + cwd=directory, + stdout=log, + stderr=subprocess.STDOUT, + ) + try: + yield process, output + except BaseException: + print("{} output:\n{}".format(name, output.read_text())) + raise + finally: + if process.poll() is None: + process.terminate() + try: + process.wait(timeout=2) + except subprocess.TimeoutExpired: + process.kill() + process.wait(timeout=2) + + def finish(self, process, output, seconds=15, success=True): + result = process.wait(timeout=self.timeout(seconds)) + text = output.read_text() + if (result == 0) != success: + raise AssertionError("unexpected exit status {}:\n{}".format(result, text)) + return text + + def ready(self, process, output): + deadline = time.monotonic() + self.timeout(10) + while time.monotonic() < deadline: + if process.poll() is not None: + raise AssertionError("listener exited before becoming ready") + if "Listener: " in output.read_text(): + return + time.sleep(0.05) + raise TimeoutError("listener did not report readiness") + + def address(self, port): + return "127.0.0.1:{}{}".format(port, self.tls_suffix) + + def enable_tls(self): + cert = self.root / "cert.pem" + key = self.root / "key.pem" + config = self.root / "openssl.cnf" + config.write_text( + "[req]\n" + "distinguished_name=subject\n" + "x509_extensions=extensions\n" + "prompt=no\n" + "[subject]\n" + "CN=networktest-smoke\n" + "[extensions]\n" + "basicConstraints=critical,CA:TRUE\n" + "keyUsage=critical,digitalSignature,keyEncipherment,keyCertSign\n" + "extendedKeyUsage=serverAuth,clientAuth\n" + "subjectAltName=IP:127.0.0.1\n" + ) + subprocess.run( + [ + "openssl", + "req", + "-x509", + "-newkey", + "rsa:2048", + "-nodes", + "-days", + "1", + "-config", + str(config), + "-keyout", + str(key), + "-out", + str(cert), + ], + check=True, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + timeout=self.timeout(15), + ) + self.tls_suffix = ":tls" + self.tls_options = [ + "--tls_certificate_file", + str(cert), + "--tls_key_file", + str(key), + "--tls_ca_file", + str(cert), + "--tls_verify_peers", + "Root.CN=networktest-smoke", + ] + + def rpc(self): + port = unused_port() + address = self.address(port) + with self.process("rpc-server", ["--mode", "server", "-p", address]) as ( + server, + server_output, + ): + self.ready(server, server_output) + with self.process( + "rpc-client", + [ + "--mode", + "client", + "--testservers", + address, + "--knob_network_test_script_mode=true", + ], + ) as (client, output): + text = self.finish(client, output) + rates = re.findall(r"(?m)^([0-9]+(?:\.[0-9]+)?)\t", text) + assert any(float(rate) > 0 for rate in rates), text + assert server.poll() is None, "RPC server exited during traffic" + + def p2p(self): + address = self.address(unused_port()) + with self.process( + "p2p", + [ + "--mode", + "p2p", + "--test_listenerAddresses=" + address, + "--test_remoteAddresses=" + address, + "--test_connectionsOut=2", + "--test_requestBytes=32:48", + "--test_replyBytes=96:64", + "--test_requests=2:3", + "--test_idleMilliseconds=1:0", + "--test_waitReadMilliseconds=1:2", + "--test_waitWriteMilliseconds=0:1", + "--test_targetDuration=2", + ], + ) as (process, output): + text = self.finish(process, output) + for expected in ( + "2 outgoing connections", + "Request size: 32:48", + "Response size: 64:96", + "Requests per outgoing session: 2:3", + "Delay before socket read: 1:2", + "Delay before socket write: 0:1", + "Delay before session close: 0:1", + ): + assert expected in text, text + for direction in ("in", "out"): + rates = re.findall(r"([0-9.]+)/s completed sessions " + direction, text) + assert any(float(rate) > 0 for rate in rates), text + errors = re.findall(r"Total Errors (\d+)", text) + assert errors and all(int(count) == 0 for count in errors), text + + def handshake(self): + port = unused_port() + address = self.address(port) + with self.process( + "handshake-server", + ["--mode", "p2p-oneshot", "--test_listenerAddresses=" + address], + ) as (server, server_output): + self.ready(server, server_output) + with self.process( + "handshake-client", + ["--mode", "p2p-oneshot", "--test_remoteAddresses=" + address], + ) as (client, output): + text = self.finish(client, output) + assert re.search(r"Client: connected to .*handshake done", text), text + text = self.finish(server, server_output, seconds=20) + assert re.search(r"Server: connected from .*handshake done", text), text + assert "handshake error" not in text, text + + def invalid_arguments(self): + server = ["--mode", "server", "-p", self.address(unused_port())] + p2p = ["--mode", "p2p", "--test_listenerAddresses=" + self.address(unused_port())] + cases = [ + [], + ["--mode", "p2p"], + p2p + ["--test_unknown=1"], + p2p + ["--test_connectionsOut=-1"], + p2p + ["--test_connectionsOut=invalid"], + p2p + ["--test_requestBytes=1::2"], + p2p + ["--test_replyBytes=2147483647"], + p2p + ["--test_targetDuration=nan"], + server + ["--knob_network_test_script_mode=invalid"], + server + ["--knob_not_a_network_test_knob=1"], + server + ["--not-a-network-test-option"], + ] + for index, arguments in enumerate(cases): + with self.process("invalid-{}".format(index), arguments) as ( + process, + output, + ): + text = self.finish(process, output, seconds=5, success=False) + assert "ERROR:" in text, text + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("executable", type=lambda value: str(Path(value).resolve())) + parser.add_argument("--tls", action="store_true", help="use temporary TLS fixtures") + args = parser.parse_args() + with tempfile.TemporaryDirectory(prefix="fdbrpc-network-smoke-") as directory: + smoke = SmokeTest(args.executable, Path(directory)) + smoke.invalid_arguments() + if args.tls: + smoke.enable_tls() + smoke.rpc() + smoke.p2p() + smoke.handshake() + print("network test smoke passed ({})".format("TLS" if args.tls else "plain TCP")) + + +if __name__ == "__main__": + main() diff --git a/fdbserver/CMakeLists.txt b/fdbserver/CMakeLists.txt index 704975e28d0..306b3d935da 100644 --- a/fdbserver/CMakeLists.txt +++ b/fdbserver/CMakeLists.txt @@ -26,6 +26,7 @@ if(WITH_ROCKSDB) endif() add_subdirectory(core) +add_subdirectory(checkpoint) add_subdirectory(kvstore) add_subdirectory(logsystem) add_subdirectory(mocks3) @@ -193,6 +194,7 @@ target_link_libraries(fdbserver PRIVATE fdbserver_storageserver fdbserver_tester fdbserver_tlog + fdbserver_checkpoint fdbserver_core) if (WITH_ROCKSDB) add_dependencies(fdbserver rocksdb) diff --git a/fdbserver/SimulatedCluster.cpp b/fdbserver/SimulatedCluster.cpp index 090b41777bb..04c83ea11ef 100644 --- a/fdbserver/SimulatedCluster.cpp +++ b/fdbserver/SimulatedCluster.cpp @@ -608,7 +608,7 @@ class TestConfig : public BasicTestConfig { } TestConfig() = default; - explicit(false) TestConfig(const BasicTestConfig& config) : BasicTestConfig(config) {} + explicit TestConfig(const BasicTestConfig& config) : BasicTestConfig(config) {} }; template @@ -874,7 +874,7 @@ Future simulatedFDBDRebooter(ReferenceisSimulated() && e.code() != error_code_io_timeout && (bool)g_network->global(INetwork::enASIOTimedOut)) { TraceEvent(SevError, "IOTimeoutErrorSuppressed") - .detail("ErrorCode", e.code()) + .detail("ObservedErrorCode", e.code()) .detail("RandomId", randomId) .backtrace(); } diff --git a/fdbserver/backupworker/BackupWorker.cpp b/fdbserver/backupworker/BackupWorker.cpp index 1ba6b8bfc14..54fe2f1be61 100644 --- a/fdbserver/backupworker/BackupWorker.cpp +++ b/fdbserver/backupworker/BackupWorker.cpp @@ -18,6 +18,7 @@ * limitations under the License. */ +#include "BackupWorkerPause.h" #include "fdbclient/BackupAgent.h" #include "fdbclient/BackupFileFormat.h" #include "fdbclient/BackupContainer.h" @@ -1013,8 +1014,12 @@ Future checkRemoved(Reference const> db, LogEpoch r } } -static Future monitorWorkerPause(BackupData* self) { - auto tr = makeReference(self->cx); +Future monitorBackupPause(Database cx, + UID workerId, + AsyncVar* pauseState, + const char* pausedEvent, + const char* resumedEvent) { + auto tr = makeReference(cx); Future watch; while (true) { @@ -1026,9 +1031,9 @@ static Future monitorWorkerPause(BackupData* self) { Optional value = co_await tr->get(backupPausedKey); bool paused = value.present() && value.get() == "1"_sr; - if (self->paused.get() != paused) { - TraceEvent(paused ? "BackupWorkerPaused" : "BackupWorkerResumed", self->myId).log(); - self->paused.set(paused); + if (pauseState->get() != paused) { + TraceEvent(paused ? pausedEvent : resumedEvent, workerId).log(); + pauseState->set(paused); } watch = tr->watch(backupPausedKey); @@ -1067,7 +1072,11 @@ Future backupWorker(BackupInterface interf, if (req.recruitedEpoch == req.backupEpoch && req.tag.id == 0) { addActor.send(monitorBackupProgress(&self)); } - addActor.send(monitorWorkerPause(&self)); + addActor.send(monitorBackupPause(self.cx, + self.myId, + &self.paused, + /*pausedEvent=*/"BackupWorkerPaused", + /*resumedEvent=*/"BackupWorkerResumed")); // If the worker is on an old epoch and all backups starts a version >= the endVersion bool exitEarly = co_await shouldBackupWorkerExitEarly(&self); diff --git a/fdbserver/backupworker/BackupWorkerPause.h b/fdbserver/backupworker/BackupWorkerPause.h new file mode 100644 index 00000000000..f8786ab8907 --- /dev/null +++ b/fdbserver/backupworker/BackupWorkerPause.h @@ -0,0 +1,31 @@ +/* + * BackupWorkerPause.h + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#pragma once + +#include "fdbclient/NativeAPI.h" +#include "flow/genericactors.h" + +// The pause state and event names must outlive the returned future. +Future monitorBackupPause(Database cx, + UID workerId, + AsyncVar* pauseState, + const char* pausedEvent, + const char* resumedEvent); diff --git a/fdbserver/backupworker/RangePartitionedBackupWorker.cpp b/fdbserver/backupworker/RangePartitionedBackupWorker.cpp index 4e48f5699cb..e28c14b86ce 100644 --- a/fdbserver/backupworker/RangePartitionedBackupWorker.cpp +++ b/fdbserver/backupworker/RangePartitionedBackupWorker.cpp @@ -18,6 +18,7 @@ * limitations under the License. */ +#include "BackupWorkerPause.h" #include "fdbclient/BackupAgent.h" #include "fdbclient/BackupFileFormat.h" #include "fdbclient/BackupContainer.h" @@ -1051,36 +1052,6 @@ Future setBackupKeys(RangePartitionedBackupData* self, std::map monitorWorkerPause(RangePartitionedBackupData* self) { - auto tr = makeReference(self->cx); - Future watch; - - while (true) { - Error err; - try { - tr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); - tr->setOption(FDBTransactionOptions::LOCK_AWARE); - tr->setOption(FDBTransactionOptions::PRIORITY_SYSTEM_IMMEDIATE); - - Optional value = co_await tr->get(backupPausedKey); - bool paused = value.present() && value.get() == "1"_sr; - if (self->paused.get() != paused) { - TraceEvent(paused ? "RangePartitionedBWPaused" : "RangePartitionedBWResumed", self->myId).log(); - self->paused.set(paused); - } - - watch = tr->watch(backupPausedKey); - co_await tr->commit(); - co_await watch; - tr->reset(); - continue; - } catch (Error& e) { - err = e; - } - co_await tr->onError(err); - } -} - Future monitorRangePartitionedBackupProgress(RangePartitionedBackupData* self) { Future interval; @@ -1313,7 +1284,11 @@ Future rangePartitionedBackupWorker(BackupInterface interf, addActor.send(monitorRangePartitionedBackupProgress(&self)); } - addActor.send(monitorWorkerPause(&self)); + addActor.send(monitorBackupPause(self.cx, + self.myId, + &self.paused, + /*pausedEvent=*/"RangePartitionedBWPaused", + /*resumedEvent=*/"RangePartitionedBWResumed")); // Must be sent before processPartitionMap so logSystem is populated before the partition-map peek. addActor.send(monitorLogSystemFromDbInfo(db, &self)); diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index ea906b165fb..0ac3cb132c2 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -31,6 +31,7 @@ #include "fdbclient/Knobs.h" #include "fdbclient/SystemData.h" #include "NativeCdcInternal.h" +#include "fdbserver/core/NativeCdcMetadata.h" #include "fdbserver/cdcproxy/CDCProxy.h" #include "fdbserver/core/Knobs.h" #include "fdbserver/core/LogProtocolMessage.h" @@ -53,7 +54,7 @@ namespace { // Snapshot from one durable metadata read, used while initializing or validating one stream. struct CDCStreamReadState { - Optional keys; + Optional> ranges; Version minVersion = invalidVersion; Version readVersion = invalidVersion; // Each Version is the inclusive lower bound for log versions routed to its paired tag; the next entry's @@ -72,10 +73,84 @@ struct CDCTagInterval { : tag(tag), begin(begin), end(end), bufferedThrough(begin - 1) {} }; +struct CDCBufferedTag; + +FDB_BOOLEAN_PARAM(HasMutations); + +// Speculative buffering never proves a client cursor and never creates another read-ahead credit. +class CDCStreamReadAhead { + enum class State { Idle, Armed, Claimed }; + State state = State::Idle; + Version issuedReplyThrough = invalidVersion; + Version creditThrough = invalidVersion; + CDCBufferedTag const* claimedTag = nullptr; + +public: + bool provesCursor(Version cursor, Version minVersion) const { + return cursor <= std::max(issuedReplyThrough, minVersion - 1); + } + bool issueReply(Version through, Version bufferedThrough, Version minVersion, HasMutations hasMutations) { + const bool advanced = !provesCursor(through, minVersion); + if (state == State::Armed && creditThrough != bufferedThrough) { + cancel(); + } + issuedReplyThrough = std::max(issuedReplyThrough, through); + if (advanced && hasMutations && through == bufferedThrough && state == State::Idle) { + state = State::Armed; + creditThrough = through; + return true; + } + return false; + } + bool armedFor(Version through) const { return state == State::Armed && creditThrough == through; } + bool claim(CDCBufferedTag const* tag, Version through) { + if (!armedFor(through)) { + return false; + } + state = State::Claimed; + claimedTag = tag; + return true; + } + bool claimedBy(CDCBufferedTag const* tag) const { return state == State::Claimed && claimedTag == tag; } + void finish(CDCBufferedTag const* tag) { + if (claimedBy(tag)) { + cancel(); + } + } + void cancel() { + state = State::Idle; + claimedTag = nullptr; + } +}; + +// A transport retry supersedes only requests from the same logical consumer. +class CDCConsumeLease : public ReferenceCounted { + Optional consumerId; + Promise superseded; + +public: + explicit CDCConsumeLease(Optional consumerId) : consumerId(consumerId) {} + + bool belongsTo(Optional other) const { + return consumerId.present() && consumerId.get().isValid() && consumerId == other; + } + void supersede() { superseded.send(Void()); } + + Future waitForReply(Future reply) { + // Coroutine parameters can outlive completion while the caller retains its result future. + ScopeExit cancelReply([&reply]() { reply.cancel(); }); + auto result = co_await race(reply, superseded.getFuture()); + if (result.index() == 1) { + throw request_maybe_delivered(); + } + co_return std::get<0>(std::move(result)); + } +}; + // Proxy-owned state for one assigned stream. In-flight actors may retain it after active becomes false. struct CDCBufferedStream : ReferenceCounted { CDCStreamId streamId; - Optional keys; + Optional> ranges; bool active = true; bool initialized = false; bool initializationPausedForTesting = false; @@ -83,9 +158,12 @@ struct CDCBufferedStream : ReferenceCounted { bool bufferLimitExceeded = false; Version minVersion = invalidVersion; Version bufferedThrough = invalidVersion; + Version metadataReadVersion = invalidVersion; int64_t bufferedBytes = 0; int readDemand = 0; - int activeConsumes = 0; + std::vector> tagAssignments; + Reference activeConsume; + CDCStreamReadAhead readAhead; std::vector tagIntervals; std::deque> mutations; AsyncTrigger changed; @@ -103,6 +181,7 @@ struct CDCBufferedBatch { struct CDCBufferedTag : ReferenceCounted { Tag tag; bool active = true; + int64_t nextPassReservation = 0; std::set streamIds; AsyncTrigger refresh; AsyncTrigger stopped; @@ -110,6 +189,52 @@ struct CDCBufferedTag : ReferenceCounted { explicit CDCBufferedTag(Tag tag) : tag(tag) {} }; +bool hasCDCReadInterest(Reference const& stream, Reference const& tag) { + return stream->readDemand > 0 || stream->readAhead.claimedBy(tag.getPtr()); +} + +Optional nextCDCPrefetchVersion(Reference const& stream, + Reference const& tag) { + if (!stream->active || !stream->initialized || stream->bufferLimitExceeded || + !stream->readAhead.armedFor(stream->bufferedThrough)) { + return {}; + } + const Version next = std::max(stream->minVersion, stream->bufferedThrough + 1); + if (next > stream->metadataReadVersion) { + return {}; + } + for (auto const& interval : stream->tagIntervals) { + if (interval.begin <= next && next < interval.end) { + return interval.tag == tag->tag ? Optional(next) : Optional(); + } + } + return {}; +} + +// A tag has one buffering actor. Its claim retains the exact stream objects, never a replacement with the same ID. +class CDCReadAheadPass : NonCopyable { + Reference tag; + std::vector> streams; + +public: + explicit CDCReadAheadPass(Reference tag) : tag(tag) {} + ~CDCReadAheadPass() { + for (auto const& stream : streams) { + stream->readAhead.finish(tag.getPtr()); + } + } + bool empty() const { return streams.empty(); } + void claim(Reference stream, Version begin) { + const auto next = nextCDCPrefetchVersion(stream, tag); + // A later frontier keeps its credit for a pass that can actually reach it. + if (next.present() && next.get() == begin && stream->readAhead.claim(tag.getPtr(), stream->bufferedThrough)) { + streams.push_back(stream); + } + } +}; + +FDB_BOOLEAN_PARAM(Prefetch); + // One stream's frontier and estimated materialization cost while selecting work for a single tag-buffering pass. struct CDCBufferCandidate { CDCStreamId streamId; @@ -240,6 +365,12 @@ Version selectedCDCConsumeReplyThrough(CDCConsumeReplySelection const& selection return selection.firstExcludedVersion.present() ? selection.firstExcludedVersion.get() - 1 : bufferedThrough; } +Version boundedCDCConsumeReplyThrough(Version lastConsumedVersion, Version readVersion, Version bufferedThrough) { + // A later cutover may change the tag for versions beyond this metadata snapshot. An already delivered cursor + // remains valid even when a subsequent metadata read temporarily trails it. + return std::max(lastConsumedVersion, std::min(readVersion, bufferedThrough)); +} + // A transactionally consistent durable-watermark snapshot consumed by one acknowledged-data pop pass. struct CDCPopState { std::unordered_map minVersions; @@ -254,6 +385,10 @@ bool hasCompleteLogSystemConfig(LogSystemConfig const& config) { enum class CDCBufferTagPassResult { RETRY, WAIT_FOR_COMMIT, STOP }; +double remainingCommitWait(double passStart, double waitInterval, double currentTime) { + return std::max(0.0, passStart + waitInterval - currentTime); +} + Optional calculateBufferPassLimits(int64_t bufferBytes, int64_t maximumPeekBytes, int64_t retainedReplyCount) { @@ -365,6 +500,7 @@ Version retiredTagPopTarget(Version retiredVersion, Optional safePopVer } class CDCProxy { + friend class CDCProxyPrefetchTest; UID id; Database cx; Reference const> dbInfo; @@ -404,13 +540,21 @@ class CDCProxy { void clearBufferedMutations(Reference stream); void addBufferedBatch(Reference stream, CDCBufferedBatch batch); void reconcileStreamMinVersion(Reference stream, Version minVersion); - void markTagStreamsBufferLimitExceeded(Reference tag, Version begin); + void reconcileStreamMetadata(Reference stream, CDCStreamReadState const& metadata); + void markTagStreamsBufferLimitExceeded(Reference tag, Version begin, int replyByteLimit); void markTagStreamsRawReplyBudgetExceeded(Reference tag, Version begin, int64_t retainedReplyCount); + void attachStreamToTags(Reference stream); + void detachStreamFromTags(CDCStreamId streamId, std::vector const& intervals); void detachStreamFromTags(Reference stream); void deactivateStream(Reference stream); void refreshStreamTags(Reference stream); - Optional nextTagReadVersionForStream(Reference tag, Reference stream); + void changeStreamReadDemand(Reference stream, int delta); + bool tagDemandChangeNeedsRefresh(Reference tag, Reference stream, int delta); + Optional nextTagReadVersionForStream(Reference tag, + Reference stream, + bool ignoreReadInterest = false); Optional nextTagReadVersion(Reference tag); + Optional nextTagPrefetchVersion(Reference tag); void advanceTagBufferedThrough(Reference tag, Version bufferedThrough, std::unordered_set const& selectedStreamIds); @@ -438,15 +582,24 @@ class CDCProxy { CDCCommittedPrefix const& prefix, int64_t preferredBufferedBatch, int64_t hardBufferedBatchLimit); - Future materializeBufferSelection(Reference tag, - Reference cursor, - Version throughVersion, - CDCBufferSelection const& selection, - int64_t rawPeekReservation, - FlowLock::Releaser& reservation, - int64_t bufferLimit); + CDCBufferTagPassResult materializeBufferSelection(Reference tag, + Reference cursor, + Version throughVersion, + CDCBufferSelection const& selection, + int64_t rawPeekReservation, + FlowLock::Releaser& reservation, + int64_t bufferLimit, + Future invalidated); Future rotateContendedPeek(); - Future bufferTagPass(Reference tag, Version begin); + Future bufferTagPass(Reference tag, Version begin, Prefetch prefetch); + Future bufferTagCursor(Reference tag, + Version begin, + Reference cursor, + Future logSystemChanged, + Prefetch prefetch); + CDCProxy() + : logSystem(makeReference>>()), + bufferLock(SERVER_KNOBS->CDC_PROXY_BUFFER_BYTES), actors(false) {} Future bufferTag(Reference tag); Future initializeStream(Reference stream); Future waitForBufferedVersion(Reference stream, Version version); @@ -454,6 +607,7 @@ class CDCProxy { Future monitorAcknowledgedDataPops(); void reconcileStreams(); Future consume(CDCConsumeRequest request); + Future consumeReply(Reference stream, CDCCursor cursor); Future acknowledge(CDCAckRequest request); Future registerStream(CDCRegisterStreamRequest request); Future removeStream(CDCRemoveStreamRequest request); @@ -477,29 +631,35 @@ class CDCProxy { Future run(CDCProxyInterface proxy, uint64_t recoveryCount); }; -Optional clipCDCMutation(MutationRef const& mutation, KeyRangeRef const& keys) { +template +void visitClippedCDCMutations(MutationRef const& mutation, std::vector const& ranges, Visitor&& visitor) { + // Canonical stream ranges are ordered and disjoint, so their ends are strictly increasing. + auto range = std::upper_bound(ranges.begin(), ranges.end(), mutation.param1, [](KeyRef key, KeyRange const& range) { + return key < range.end; + }); if (isSingleKeyMutation((MutationRef::Type)mutation.type)) { - if (keys.contains(mutation.param1)) { - return mutation; + if (range != ranges.end() && range->contains(mutation.param1)) { + visitor(mutation); } } else if (mutation.type == MutationRef::ClearRange) { - KeyRangeRef intersection = keys & KeyRangeRef(mutation.param1, mutation.param2); - if (!intersection.empty()) { - return MutationRef(MutationRef::ClearRange, intersection.begin, intersection.end); + for (; range != ranges.end() && range->begin < mutation.param2; ++range) { + const KeyRangeRef intersection = *range & KeyRangeRef(mutation.param1, mutation.param2); + if (!intersection.empty()) { + visitor(MutationRef(MutationRef::ClearRange, intersection.begin, intersection.end)); + } } } else { ASSERT(false); } - return Optional(); } -FDB_BOOLEAN_PARAM(PrioritizeConsume); +FDB_BOOLEAN_PARAM(PrioritizeDrain); AsyncResult readCDCStreamState(Database cx, CDCStreamId streamId, UID expectedProxyId, - bool requireKeys, - PrioritizeConsume prioritizeConsume = PrioritizeConsume::False) { + bool requireRanges, + PrioritizeDrain prioritizeDrain = PrioritizeDrain::False) { if (streamId == 0) { throw client_invalid_operation(); } @@ -510,22 +670,22 @@ AsyncResult readCDCStreamState(Database cx, try { tr.setOption(FDBTransactionOptions::READ_LOCK_AWARE); tr.setOption(FDBTransactionOptions::READ_SYSTEM_KEYS); - if (prioritizeConsume) { + if (prioritizeDrain) { // Draining committed CDC data must continue while ordinary transaction admission is throttled. tr.setOption(FDBTransactionOptions::PRIORITY_SYSTEM_IMMEDIATE); } - Future> keysFuture = tr.get(cdcStreamKeyFor(streamId)); + Future> rangesFuture = tr.get(cdcStreamKeyFor(streamId)); Future> minVersionFuture = tr.get(cdcMinVersionKeyFor(streamId)); Future assignedProxiesFuture = tr.getRange(cdcProxyRangeFor(streamId), 2); KeyRange tagHistoryRange = cdcTagHistoryRangeFor(streamId); Future historyFuture = tr.getRange(tagHistoryRange, CLIENT_KNOBS->TOO_MANY); CDCStreamReadState result; - Optional keysValue = co_await keysFuture; - if (keysValue.present()) { - result.keys = decodeCDCStreamKeysValue(keysValue.get()); - } else if (requireKeys) { + Optional rangesValue = co_await rangesFuture; + if (rangesValue.present()) { + result.ranges = decodeCDCStreamKeysValue(rangesValue.get()); + } else if (requireRanges) { throw client_invalid_operation(); } @@ -546,7 +706,7 @@ AsyncResult readCDCStreamState(Database cx, while (begin < tagHistoryRange.end) { RangeResult history = co_await historyFuture; for (KeyValueRef const& kv : history) { - const CDCTagHistoryEntry historyEntry = decodeCDCTagHistoryKey(kv.key); + const CDCTagHistoryEntry historyEntry = decodeCDCTagHistoryEntry(kv.key, kv.value); ASSERT_WE_THINK(historyEntry.streamId == streamId); ASSERT_WE_THINK(historyEntry.tag.locality == tagLocalityCDC); tagAssignments.emplace_back(historyEntry.version, historyEntry.tag); @@ -605,6 +765,9 @@ void CDCProxy::refreshLogSystem() { lastLogSystemConfig = info.logSystemConfig; } if (logSystemChanged) { + for (auto const& [streamId, stream] : streams) { + stream->readAhead.cancel(); + } popLogSystemChanged.trigger(); } } @@ -677,7 +840,7 @@ void CDCProxy::addBufferedBatch(Reference stream, CDCBuffered totalBufferedMutationBytes += batch.bufferedBytes; } -void CDCProxy::markTagStreamsBufferLimitExceeded(Reference tag, Version begin) { +void CDCProxy::markTagStreamsBufferLimitExceeded(Reference tag, Version begin, int replyByteLimit) { for (const CDCStreamId streamId : tag->streamIds) { auto stream = streams.find(streamId); if (stream == streams.end() || !stream->second->active) { @@ -692,7 +855,7 @@ void CDCProxy::markTagStreamsBufferLimitExceeded(Reference tag, .detail("Tag", tag->tag) .detail("StreamId", streamId) .detail("BeginVersion", begin) - .detail("RawPeekLimit", SERVER_KNOBS->MAXIMUM_PEEK_BYTES); + .detail("RawPeekLimit", replyByteLimit); stream->second->bufferLimitExceeded = true; stream->second->changed.trigger(); } @@ -719,9 +882,44 @@ void CDCProxy::markTagStreamsRawReplyBudgetExceeded(Reference ta } } -void updateStreamBufferedThrough(Reference stream) { - Version bufferedThrough = stream->minVersion - 1; - for (const auto& interval : stream->tagIntervals) { +Optional firstIncompleteTagInterval(CDCBufferedStream const& stream) { + for (size_t i = 0; i < stream.tagIntervals.size(); ++i) { + if (stream.tagIntervals[i].bufferedThrough < stream.tagIntervals[i].end - 1) { + return i; + } + } + return Optional(); +} + +Optional eligibleTagReadInterval(CDCBufferedStream const& stream, Tag const& tag) { + if (!stream.active || !stream.initialized || stream.bufferLimitExceeded) { + return Optional(); + } + const Optional first = firstIncompleteTagInterval(stream); + if (!first.present()) { + return Optional(); + } + const auto& interval = stream.tagIntervals[first.get()]; + // A later interval cannot release its buffered data until the unread prefix has been delivered. Letting it + // compete for capacity can fill the buffer with that undeliverable tail and prevent the prefix from ever reading. + if (interval.tag != tag || std::max(interval.begin, interval.bufferedThrough + 1) > stream.metadataReadVersion) { + return Optional(); + } + return first; +} + +bool canBufferTagVersion(CDCBufferedStream const& stream, Tag const& tag, Version version) { + const Optional eligible = eligibleTagReadInterval(stream, tag); + if (!eligible.present() || version > stream.metadataReadVersion) { + return false; + } + const auto& interval = stream.tagIntervals[eligible.get()]; + return interval.begin <= version && version < interval.end && version > interval.bufferedThrough; +} + +Version contiguousStreamBufferedThrough(CDCBufferedStream const& stream) { + Version bufferedThrough = stream.minVersion - 1; + for (const auto& interval : stream.tagIntervals) { if (interval.begin > bufferedThrough + 1) { break; } @@ -733,12 +931,30 @@ void updateStreamBufferedThrough(Reference stream) { break; } } + return bufferedThrough; +} + +void updateStreamBufferedThrough(Reference stream) { + const Version bufferedThrough = contiguousStreamBufferedThrough(*stream); if (bufferedThrough > stream->bufferedThrough) { stream->bufferedThrough = bufferedThrough; stream->changed.trigger(); } } +bool advanceStreamTagBufferedThrough(Reference stream, Tag const& tag, Version throughVersion) { + const Optional eligible = eligibleTagReadInterval(*stream, tag); + if (!eligible.present()) { + return false; + } + auto& interval = stream->tagIntervals[eligible.get()]; + interval.bufferedThrough = + std::max(interval.bufferedThrough, std::min({ throughVersion, stream->metadataReadVersion, interval.end - 1 })); + const bool intervalCompleted = interval.bufferedThrough == interval.end - 1; + updateStreamBufferedThrough(stream); + return intervalCompleted; +} + void advanceStreamMinVersion(Reference stream, Version minVersion) { stream->minVersion = std::max(stream->minVersion, minVersion); for (auto& interval : stream->tagIntervals) { @@ -750,7 +966,89 @@ void advanceStreamMinVersion(Reference stream, Version minVer updateStreamBufferedThrough(stream); } +struct CDCStreamMetadataUpdate { + bool historyChanged = false; + bool readVersionAdvanced = false; + int64_t releasedBytes = 0; +}; + +CDCStreamMetadataUpdate reconcileBufferedStreamMetadata(Reference stream, + CDCStreamReadState const& metadata) { + CDCStreamMetadataUpdate update; + ASSERT(stream->mutations.empty() || stream->mutations.back().version <= stream->metadataReadVersion); + stream->minVersion = std::max(stream->minVersion, metadata.minVersion); + std::vector previousIntervals; + if (metadata.readVersion >= stream->metadataReadVersion) { + update.readVersionAdvanced = metadata.readVersion > stream->metadataReadVersion; + stream->metadataReadVersion = metadata.readVersion; + stream->ranges = metadata.ranges; + update.historyChanged = stream->tagAssignments != metadata.tagAssignments; + if (update.historyChanged) { + previousIntervals = std::move(stream->tagIntervals); + stream->tagIntervals.clear(); + for (size_t i = 0; i < metadata.tagAssignments.size(); ++i) { + const Version begin = std::max(stream->minVersion, metadata.tagAssignments[i].first); + const Version end = i + 1 < metadata.tagAssignments.size() ? metadata.tagAssignments[i + 1].first + : std::numeric_limits::max(); + if (begin >= end) { + continue; + } + CDCTagInterval interval(metadata.tagAssignments[i].second, begin, end); + for (const auto& previous : previousIntervals) { + if (previous.tag == interval.tag && previous.begin <= begin && begin < previous.end) { + interval.bufferedThrough = + std::max(interval.bufferedThrough, std::min(previous.bufferedThrough, end - 1)); + } + } + stream->tagIntervals.push_back(interval); + } + stream->tagAssignments = metadata.tagAssignments; + } + } + + for (auto& interval : stream->tagIntervals) { + interval.bufferedThrough = + std::max(interval.bufferedThrough, std::min(stream->minVersion - 1, interval.end - 1)); + } + // Buffered mutations were admitted by an earlier metadata snapshot. A later cutover cannot change their + // routing, and history cleanup only removes acknowledged intervals, so only the acknowledged prefix expires. + while (!stream->mutations.empty() && stream->mutations.front().version < stream->minVersion) { + update.releasedBytes += estimatedCDCConsumeVersionBytes(stream->mutations.front()); + stream->mutations.pop_front(); + } + stream->bufferedBytes -= update.releasedBytes; + ASSERT_GE(stream->bufferedBytes, 0); + stream->bufferedThrough = contiguousStreamBufferedThrough(*stream); + stream->initialized = true; + return update; +} + +void CDCProxy::reconcileStreamMetadata(Reference stream, CDCStreamReadState const& metadata) { + const std::vector previousIntervals = stream->tagIntervals; + const Version previousBufferedThrough = stream->bufferedThrough; + const Optional previousReadInterval = firstIncompleteTagInterval(*stream); + const CDCStreamMetadataUpdate update = reconcileBufferedStreamMetadata(stream, metadata); + if (update.historyChanged) { + CODE_PROBE(!previousIntervals.empty(), "CDC proxy refreshes same-owner tag history"); + detachStreamFromTags(stream->streamId, previousIntervals); + attachStreamToTags(stream); + } + ASSERT_GE(bufferedBytes, update.releasedBytes); + bufferedBytes -= update.releasedBytes; + if (update.releasedBytes > 0) { + bufferLock.release(update.releasedBytes); + } + if (update.historyChanged || update.readVersionAdvanced || + previousReadInterval != firstIncompleteTagInterval(*stream)) { + refreshStreamTags(stream); + } + if (update.historyChanged || previousBufferedThrough != stream->bufferedThrough) { + stream->changed.trigger(); + } +} + void CDCProxy::reconcileStreamMinVersion(Reference stream, Version minVersion) { + const Optional previousReadInterval = firstIncompleteTagInterval(*stream); advanceStreamMinVersion(stream, minVersion); while (!stream->mutations.empty() && stream->mutations.front().version < minVersion) { const int64_t releasedBytes = @@ -762,16 +1060,35 @@ void CDCProxy::reconcileStreamMinVersion(Reference stream, Ve stream->mutations.pop_front(); } ASSERT_GE(stream->bufferedBytes, 0); + if (previousReadInterval != firstIncompleteTagInterval(*stream)) { + refreshStreamTags(stream); + } } -void CDCProxy::detachStreamFromTags(Reference stream) { +void CDCProxy::attachStreamToTags(Reference stream) { for (const auto& interval : stream->tagIntervals) { + auto tag = tags.find(interval.tag); + if (tag == tags.end()) { + auto newTag = makeReference(interval.tag); + tag = tags.emplace(interval.tag, newTag).first; + tag->second->streamIds.insert(stream->streamId); + actors.add(bufferTag(newTag)); + } else { + CODE_PROBE(true, "CDC proxy shares a tag reader across streams"); + tag->second->streamIds.insert(stream->streamId); + tag->second->refresh.trigger(); + } + } +} + +void CDCProxy::detachStreamFromTags(CDCStreamId streamId, std::vector const& intervals) { + for (const auto& interval : intervals) { auto tag = tags.find(interval.tag); if (tag == tags.end()) { continue; } Reference bufferedTag = tag->second; - bufferedTag->streamIds.erase(stream->streamId); + bufferedTag->streamIds.erase(streamId); if (bufferedTag->streamIds.empty()) { bufferedTag->active = false; tags.erase(tag); @@ -782,10 +1099,15 @@ void CDCProxy::detachStreamFromTags(Reference stream) { } } +void CDCProxy::detachStreamFromTags(Reference stream) { + detachStreamFromTags(stream->streamId, stream->tagIntervals); +} + void CDCProxy::deactivateStream(Reference stream) { CODE_PROBE(stream->readDemand > 0, "CDC proxy wakes pending consume when stream is unassigned"); CODE_PROBE(true, "CDC proxy drops removed or reassigned stream state"); stream->active = false; + stream->readAhead.cancel(); stream->changed.trigger(); detachStreamFromTags(stream); clearBufferedMutations(stream); @@ -800,22 +1122,96 @@ void CDCProxy::refreshStreamTags(Reference stream) { } } -Optional CDCProxy::nextTagReadVersionForStream(Reference tag, - Reference stream) { - if (!stream->active || !stream->initialized || stream->bufferLimitExceeded || stream->readDemand == 0) { - return Optional(); - } - Optional begin; - for (const auto& interval : stream->tagIntervals) { - if (interval.tag != tag->tag) { +bool CDCProxy::tagDemandChangeNeedsRefresh(Reference tag, + Reference stream, + int delta) { + const bool beforeInterest = hasCDCReadInterest(stream, tag); + const bool afterInterest = stream->readDemand + delta > 0 || stream->readAhead.claimedBy(tag.getPtr()); + Optional before; + Optional after; + for (const CDCStreamId streamId : tag->streamIds) { + auto found = streams.find(streamId); + if (found == streams.end()) { continue; } - const Version next = std::max(interval.begin, interval.bufferedThrough + 1); - if (next < interval.end && (!begin.present() || next < begin.get())) { - begin = next; + const bool changedStream = found->second.getPtr() == stream.getPtr(); + const auto candidate = nextTagReadVersionForStream(tag, found->second, changedStream); + if (!candidate.present()) { + continue; + } + if ((!changedStream || beforeInterest) && (!before.present() || candidate.get() < before.get())) { + before = candidate; + } + if ((!changedStream || afterInterest) && (!after.present() || candidate.get() < after.get())) { + after = candidate; } } - return begin; + // Absent frontiers must still wake a dormant reader; only equal, present work preserves its peek. + return !before.present() || !after.present() || before.get() != after.get(); +} + +void CDCProxy::changeStreamReadDemand(Reference stream, int delta) { + ASSERT(delta == 1 || delta == -1); + ASSERT_GE(stream->readDemand + delta, 0); + auto refreshCurrentTag = [this](Reference const& tag, bool refresh) { + auto found = tags.find(tag->tag); + if (found != tags.end() && (found->second.getPtr() != tag.getPtr() || refresh)) { + found->second->refresh.trigger(); + } + }; + // A stream with one tag needs neither scratch container on each consume wait. + if (stream->tagIntervals.size() <= 1) { + Reference tag; + bool refresh = false; + if (!stream->tagIntervals.empty()) { + auto found = tags.find(stream->tagIntervals.front().tag); + if (found != tags.end()) { + tag = found->second; + refresh = tagDemandChangeNeedsRefresh(tag, stream, delta); + } + } + stream->readDemand += delta; + if (tag) { + refreshCurrentTag(tag, refresh); + } + return; + } + struct TagDemandSnapshot { + Reference tag; + bool refresh; + }; + std::vector snapshots; + std::unordered_set seenTags; + const size_t maxSnapshots = std::min(stream->tagIntervals.size(), tags.size()); + snapshots.reserve(maxSnapshots); + seenTags.reserve(maxSnapshots); + for (const auto& interval : stream->tagIntervals) { + auto found = tags.find(interval.tag); + if (found != tags.end() && seenTags.insert(interval.tag).second) { + snapshots.push_back({ found->second, tagDemandChangeNeedsRefresh(found->second, stream, delta) }); + } + } + stream->readDemand += delta; + // A claimed prefetch (or another consumer) may already cover the same frontier, avoiding a restart. + // Trigger callbacks can synchronously replace tags or change stream membership. Decide before triggering, + // retain no map iterators across callbacks, and never apply an old object's equality proof to a replacement. + for (const auto& snapshot : snapshots) { + refreshCurrentTag(snapshot.tag, snapshot.refresh); + } +} + +Optional CDCProxy::nextTagReadVersionForStream(Reference tag, + Reference stream, + bool ignoreReadInterest) { + if (!ignoreReadInterest && !hasCDCReadInterest(stream, tag)) { + return Optional(); + } + const Optional eligible = eligibleTagReadInterval(*stream, tag->tag); + if (!eligible.present()) { + return Optional(); + } + const auto& interval = stream->tagIntervals[eligible.get()]; + return std::max(interval.begin, interval.bufferedThrough + 1); } Optional CDCProxy::nextTagReadVersion(Reference tag) { @@ -833,6 +1229,20 @@ Optional CDCProxy::nextTagReadVersion(Reference tag) { return begin; } +Optional CDCProxy::nextTagPrefetchVersion(Reference tag) { + Optional begin; + for (const CDCStreamId streamId : tag->streamIds) { + auto stream = streams.find(streamId); + if (stream != streams.end()) { + const auto next = nextCDCPrefetchVersion(stream->second, tag); + if (next.present() && (!begin.present() || next.get() < begin.get())) { + begin = next; + } + } + } + return begin; +} + void CDCProxy::advanceTagBufferedThrough(Reference tag, Version bufferedThrough, std::unordered_set const& selectedStreamIds) { @@ -842,16 +1252,13 @@ void CDCProxy::advanceTagBufferedThrough(Reference tag, continue; } auto stream = streams.find(streamId); - if (stream == streams.end() || !stream->second->active || stream->second->readDemand == 0) { + if (stream == streams.end() || !stream->second->active || !hasCDCReadInterest(stream->second, tag)) { continue; } - for (auto& interval : stream->second->tagIntervals) { - if (interval.tag == tag->tag && bufferedThrough >= interval.begin) { - interval.bufferedThrough = - std::max(interval.bufferedThrough, std::min(bufferedThrough, interval.end - 1)); - } + Reference bufferedStream = stream->second; + if (advanceStreamTagBufferedThrough(bufferedStream, tag->tag, bufferedThrough)) { + refreshStreamTags(bufferedStream); } - updateStreamBufferedThrough(stream->second); } } @@ -864,7 +1271,8 @@ void CDCProxy::markPoppedTagStreamsTooOld(Reference tag, Version } for (const auto& interval : stream->second->tagIntervals) { const Version next = std::max(interval.begin, interval.bufferedThrough + 1); - if (interval.tag == tag->tag && next < interval.end && next < popped) { + if (interval.tag == tag->tag && next < interval.end && next <= stream->second->metadataReadVersion && + next < popped) { tooOldStreams.push_back(stream->second); break; } @@ -913,24 +1321,14 @@ void CDCProxy::visitBufferedMutations(Reference tag, continue; } auto stream = streams.find(streamId); - if (stream == streams.end() || !stream->second->active || stream->second->readDemand == 0 || - !stream->second->keys.present()) { - continue; - } - const bool coversVersion = - std::any_of(stream->second->tagIntervals.begin(), - stream->second->tagIntervals.end(), - [tag, messageVersion](const auto& interval) { - return interval.tag == tag->tag && interval.begin <= messageVersion && - messageVersion < interval.end && messageVersion > interval.bufferedThrough; - }); - if (!coversVersion) { + if (stream == streams.end() || !hasCDCReadInterest(stream->second, tag) || + !stream->second->ranges.present() || + !canBufferTagVersion(*stream->second, tag->tag, messageVersion)) { continue; } - Optional clipped = clipCDCMutation(mutation, stream->second->keys.get()); - if (clipped.present()) { - visitor(stream->second, messageVersion, clipped.get()); - } + visitClippedCDCMutations(mutation, stream->second->ranges.get(), [&](MutationRef const& clipped) { + visitor(stream->second, messageVersion, clipped); + }); } } cursor->nextMessage(); @@ -1047,13 +1445,14 @@ Future CDCProxy::rotateContendedPeek() { co_await delay(SERVER_KNOBS->BLOCKING_PEEK_TIMEOUT); } -Future CDCProxy::materializeBufferSelection(Reference tag, - Reference cursor, - Version throughVersion, - CDCBufferSelection const& selection, - int64_t rawPeekReservation, - FlowLock::Releaser& reservation, - int64_t bufferLimit) { +CDCBufferTagPassResult CDCProxy::materializeBufferSelection(Reference tag, + Reference cursor, + Version throughVersion, + CDCBufferSelection const& selection, + int64_t rawPeekReservation, + FlowLock::Releaser& reservation, + int64_t bufferLimit, + Future invalidated) { const int64_t materializationReservation = reservation.remaining - rawPeekReservation; ASSERT_GE(materializationReservation, 0); if (selection.selectedBytes <= materializationReservation) { @@ -1061,22 +1460,18 @@ Future CDCProxy::materializeBufferSelection(ReferenceonChange(), - tag->stopped.onTrigger(), - tag->refresh.onTrigger()); - if (exactCapacity.index() == 1 || exactCapacity.index() == 3) { - co_return CDCBufferTagPassResult::RETRY; - } - if (exactCapacity.index() == 2) { - co_return CDCBufferTagPassResult::STOP; - } - reservation.remaining += additionalBytes; - recordBufferUsage(); + // Two readers can exhaust the budget with initial reservations and then both wait for an expansion. + // Drop this cursor and reservation before reacquiring the full amount in one request. + tag->nextPassReservation = rawPeekReservation + selection.selectedBytes; + ASSERT_LE(tag->nextPassReservation, bufferLimit); + return CDCBufferTagPassResult::RETRY; } if (!tag->active) { - co_return CDCBufferTagPassResult::STOP; + return CDCBufferTagPassResult::STOP; + } + // Materialize only while the selected read window is still valid. + if (invalidated.isReady()) { + return CDCBufferTagPassResult::RETRY; } std::unordered_map batches = @@ -1085,11 +1480,10 @@ Future CDCProxy::materializeBufferSelection(Reference CDCProxy::materializeBufferSelection(ReferencenextPassReservation = 0; advanceTagBufferedThrough(tag, throughVersion, selection.selectedStreamIds); // Every raw cursor arena is covered by rawPeekReservation only for this pass. Reopen from the shared minimum // after releasing it so no cursor response remains live outside the proxy memory budget. - co_return CDCBufferTagPassResult::RETRY; + return CDCBufferTagPassResult::RETRY; } -Future CDCProxy::bufferTagPass(Reference tag, Version begin) { - const int64_t bufferLimit = SERVER_KNOBS->CDC_PROXY_BUFFER_BYTES; +Future CDCProxy::bufferTagPass(Reference tag, + Version begin, + Prefetch prefetch) { Reference consumer = logSystem->get(); Future logSystemChanged = logSystem->onChange(); // CDC ReplayMultiCursor instances disable constructor prefetch, so constructing this cursor cannot issue a peek // before the proxy has reserved memory for every reply arena that its replicated read may retain. Reference cursor = consumer->peekSingle(id, begin, tag->tag, {}); - cursor->setReplyByteLimit(SERVER_KNOBS->MAXIMUM_PEEK_BYTES); + co_return co_await bufferTagCursor(tag, begin, std::move(cursor), logSystemChanged, prefetch); +} + +Future CDCProxy::bufferTagCursor(Reference tag, + Version begin, + Reference cursor, + Future logSystemChanged, + Prefetch prefetch) { + const int64_t bufferLimit = SERVER_KNOBS->CDC_PROXY_BUFFER_BYTES; + CDCReadAheadPass readAhead(tag); + if (prefetch) { + for (const CDCStreamId streamId : tag->streamIds) { + auto stream = streams.find(streamId); + if (stream != streams.end()) { + readAhead.claim(stream->second, begin); + } + } + } + if (prefetch && readAhead.empty()) { + co_return CDCBufferTagPassResult::RETRY; + } + Future prefetchDeadline = prefetch ? delay(SERVER_KNOBS->BLOCKING_PEEK_TIMEOUT) : Never(); + Future tagChanged = tag->refresh.onTrigger(); + Future invalidated = logSystemChanged || tagChanged || tag->stopped.onTrigger() || prefetchDeadline; const int64_t retainedReplyCount = cursor->getMaxRetainedReplyCount(); - Optional limits = - calculateBufferPassLimits(bufferLimit, SERVER_KNOBS->MAXIMUM_PEEK_BYTES, retainedReplyCount); + // Leave one reply-sized window for filtered mutations instead of rejecting a topology whose maximum-sized + // replies would consume the entire buffer. Every retained raw arena remains covered by the reservation. + const int replyByteLimit = + std::min(SERVER_KNOBS->MAXIMUM_PEEK_BYTES, bufferLimit / (retainedReplyCount + 1)); + Optional limits = calculateBufferPassLimits(bufferLimit, replyByteLimit, retainedReplyCount); if (!limits.present()) { markTagStreamsRawReplyBudgetExceeded(tag, begin, retainedReplyCount); co_return CDCBufferTagPassResult::RETRY; } + cursor->setReplyByteLimit(replyByteLimit); + CODE_PROBE(replyByteLimit < SERVER_KNOBS->MAXIMUM_PEEK_BYTES, + "CDC proxy sizes raw replies to fit replicated reads within its buffer"); const int64_t rawPeekReservation = limits.get().rawReplyBytes; const int64_t hardBufferedBatchLimit = limits.get().hardBufferedBytes; const int64_t preferredBufferedBatch = limits.get().preferredBufferedBytes; - const int64_t passReservation = limits.get().reservationBytes; + const int64_t passReservation = std::max(limits.get().reservationBytes, tag->nextPassReservation); + if (prefetch && (bufferLock.waiters() != 0 || bufferLock.available() < passReservation)) { + co_return CDCBufferTagPassResult::RETRY; + } if (bufferLock.available() < passReservation) { CODE_PROBE(true, "CDC proxy applies shared buffer backpressure"); peekCapacityContended.trigger(); @@ -1140,7 +1568,7 @@ Future CDCProxy::bufferTagPass(Reference auto capacity = co_await race(bufferLock.take(TaskPriority::TLogPeekReply, passReservation), logSystemChanged, tag->stopped.onTrigger(), - tag->refresh.onTrigger()); + tagChanged); if (capacity.index() == 1 || capacity.index() == 3) { co_return CDCBufferTagPassResult::RETRY; } @@ -1150,7 +1578,7 @@ Future CDCProxy::bufferTagPass(Reference FlowLock::Releaser reservation(bufferLock, passReservation); recordBufferUsage(); // If capacity and a generation change became ready together, discard the cursor built from the old topology. - if (logSystemChanged.isReady()) { + if (invalidated.isReady()) { co_return CDCBufferTagPassResult::RETRY; } if (!cursor->hasMessage()) { @@ -1160,9 +1588,10 @@ Future CDCProxy::bufferTagPass(Reference auto result = co_await race(cursor->getMore(TaskPriority::TLogPeekReply), logSystemChanged, tag->stopped.onTrigger(), - tag->refresh.onTrigger(), - rotateContendedPeek()); - if (result.index() == 1 || result.index() == 3 || result.index() == 4) { + tagChanged, + rotateContendedPeek(), + prefetchDeadline); + if (result.index() == 1 || result.index() == 3 || result.index() == 4 || result.index() == 5) { co_return CDCBufferTagPassResult::RETRY; } if (result.index() == 2) { @@ -1172,10 +1601,19 @@ Future CDCProxy::bufferTagPass(Reference if (e.code() != error_code_cdc_tlog_peek_reply_too_large) { throw; } - markTagStreamsBufferLimitExceeded(tag, begin); + if (invalidated.isReady()) { + co_return CDCBufferTagPassResult::RETRY; + } + markTagStreamsBufferLimitExceeded(tag, begin, replyByteLimit); co_return CDCBufferTagPassResult::RETRY; } } + if (!tag->active) { + co_return CDCBufferTagPassResult::STOP; + } + if (invalidated.isReady()) { + co_return CDCBufferTagPassResult::RETRY; + } // A newly constructed replay cursor can already contain messages, especially after log-generation // changes. Initialize its reader even when getMore() was unnecessary. cursor->setProtocolVersion(g_network->protocolVersion()); @@ -1197,20 +1635,25 @@ Future CDCProxy::bufferTagPass(Reference CODE_PROBE(cursor->hasMessage() && cursor->version().version > committedThrough, "CDC proxy waits for peeked mutations to become committed"); if (throughVersion < begin) { - co_return CDCBufferTagPassResult::WAIT_FOR_COMMIT; + co_return prefetch ? CDCBufferTagPassResult::RETRY : CDCBufferTagPassResult::WAIT_FOR_COMMIT; } selection = selectBufferCandidatesForTag(tag, prefix, preferredBufferedBatch, hardBufferedBatchLimit); } if (selection.selectedStreamIds.empty()) { co_return CDCBufferTagPassResult::RETRY; } - co_return co_await materializeBufferSelection( - tag, cursor, throughVersion, selection, rawPeekReservation, reservation, bufferLimit); + co_return materializeBufferSelection( + tag, cursor, throughVersion, selection, rawPeekReservation, reservation, bufferLimit, invalidated); } Future CDCProxy::bufferTag(Reference tag) { while (tag->active) { Optional begin = nextTagReadVersion(tag); + Prefetch prefetch = Prefetch::False; + if (!begin.present()) { + begin = nextTagPrefetchVersion(tag); + prefetch = Prefetch::True; + } if (!begin.present()) { auto waitForDemand = co_await race(tag->stopped.onTrigger(), tag->refresh.onTrigger()); if (waitForDemand.index() == 0) { @@ -1227,17 +1670,20 @@ Future CDCProxy::bufferTag(Reference tag) { continue; } - const CDCBufferTagPassResult result = co_await bufferTagPass(tag, begin.get()); + const double passStart = now(); + const CDCBufferTagPassResult result = co_await bufferTagPass(tag, begin.get(), prefetch); if (result == CDCBufferTagPassResult::STOP) { co_return; } if (result == CDCBufferTagPassResult::WAIT_FOR_COMMIT) { - // The cursor may already hold a speculative message, so getMore() would complete immediately without - // refreshing its committed frontier. Drop that arena and reopen after one blocking-peek interval. - auto waitForCommit = co_await race(delay(SERVER_KNOBS->BLOCKING_PEEK_TIMEOUT), - logSystem->onChange(), - tag->stopped.onTrigger(), - tag->refresh.onTrigger()); + // Older TLogs can immediately return a speculative message, for which getMore() would not refresh + // the committed frontier. Retain their bounded fallback, but do not add a second blocking-peek + // interval when a TLog already waited for commit progress. The zero-delay case still yields. + auto waitForCommit = + co_await race(delay(remainingCommitWait(passStart, SERVER_KNOBS->BLOCKING_PEEK_TIMEOUT, now())), + logSystem->onChange(), + tag->stopped.onTrigger(), + tag->refresh.onTrigger()); if (waitForCommit.index() == 2) { co_return; } @@ -1260,32 +1706,8 @@ Future CDCProxy::initializeStream(Reference stream) { CODE_PROBE(true, "CDC proxy discards stale stream initialization"); co_return; } - stream->keys = metadata.keys; - stream->minVersion = metadata.minVersion; - stream->bufferedThrough = metadata.minVersion - 1; - for (size_t i = 0; i < metadata.tagAssignments.size(); ++i) { - const Version begin = std::max(metadata.minVersion, metadata.tagAssignments[i].first); - const Version end = i + 1 < metadata.tagAssignments.size() ? metadata.tagAssignments[i + 1].first - : std::numeric_limits::max(); - if (begin < end) { - stream->tagIntervals.emplace_back(metadata.tagAssignments[i].second, begin, end); - } - } - stream->initialized = true; + reconcileStreamMetadata(stream, metadata); stream->changed.trigger(); - for (const auto& interval : stream->tagIntervals) { - auto tag = tags.find(interval.tag); - if (tag == tags.end()) { - auto newTag = makeReference(interval.tag); - tag = tags.emplace(interval.tag, newTag).first; - tag->second->streamIds.insert(stream->streamId); - actors.add(bufferTag(newTag)); - } else { - CODE_PROBE(true, "CDC proxy shares a tag reader across streams"); - tag->second->streamIds.insert(stream->streamId); - tag->second->refresh.trigger(); - } - } } catch (Error& e) { if (e.code() == error_code_client_invalid_operation || e.code() == error_code_wrong_shard_server) { clearBufferedMutations(stream); @@ -1327,7 +1749,7 @@ AsyncResult readPopState(Database cx) { RangeResult histories = co_await tr.getRange(KeyRangeRef(begin, cdcTagHistoryKeys.end), CLIENT_KNOBS->TOO_MANY); for (const auto& kv : histories) { - const CDCTagHistoryEntry history = decodeCDCTagHistoryKey(kv.key); + const CDCTagHistoryEntry history = decodeCDCTagHistoryEntry(kv.key, kv.value); auto minimum = result.minVersions.find(history.streamId); if (minimum == result.minVersions.end()) { continue; @@ -1551,13 +1973,8 @@ Future CDCProxy::waitForBufferedVersion(Reference strea co_return; } - ++stream->readDemand; - refreshStreamTags(stream); - ScopeExit releaseReadDemand([this, stream]() { - ASSERT_GT(stream->readDemand, 0); - --stream->readDemand; - refreshStreamTags(stream); - }); + changeStreamReadDemand(stream, 1); + ScopeExit releaseReadDemand([this, stream]() { changeStreamReadDemand(stream, -1); }); while (stream->active && !stream->bufferLimitExceeded && stream->bufferedThrough < version) { co_await stream->changed.onTrigger(); } @@ -1589,95 +2006,124 @@ Future CDCProxy::consume(CDCConsumeRequest request) { if (!stream->active) { throw wrong_shard_server(); } - if (stream->activeConsumes > 0) { - // A stream has one durable acknowledgement frontier, so concurrent logical consumers cannot be - // isolated. Reject overlapping server requests rather than duplicating an entire reply arena. - CODE_PROBE(true, "CDC proxy rejects concurrent consumers for one stream"); - throw client_invalid_operation(); - } - ++stream->activeConsumes; - ScopeExit releaseStreamConsume([stream]() { - ASSERT_GT(stream->activeConsumes, 0); - --stream->activeConsumes; - }); - const CDCStreamReadState metadata = - co_await readCDCStreamState(cx, request.cursor.streamId, id, true, PrioritizeConsume::True); - CODE_PROBE(stream->minVersion < metadata.minVersion, "Native CDC consume reconciles a durable acknowledgement"); - reconcileStreamMinVersion(stream, metadata.minVersion); - if (request.cursor.lastConsumedVersion > stream->bufferedThrough) { - // A cursor is trusted only when this owner has delivered through it or when it is covered by the durable - // acknowledgement watermark used to initialize bufferedThrough. This prevents a fabricated cursor from - // making the proxy retain every intervening tagged mutation while trying to reach an unproven position. - if (request.cursor.lastConsumedVersion > metadata.readVersion) { - CODE_PROBE(true, "CDC proxy rejects a consume cursor beyond its transaction read version"); - } else { - CODE_PROBE(true, "CDC proxy rejects an unproven consume cursor"); + if (stream->activeConsume.isValid()) { + if (!stream->activeConsume->belongsTo(request.consumerId)) { + CODE_PROBE(true, "CDC proxy rejects concurrent consumers for one stream"); + throw client_invalid_operation(); } - throw client_invalid_operation(); + CODE_PROBE(true, "CDC proxy supersedes a consume after transport retry"); + auto previous = stream->activeConsume; + previous->supersede(); + } + auto lease = makeReference(request.consumerId); + stream->activeConsume = lease; + ScopeExit releaseStreamConsume([stream, lease]() { + if (stream->activeConsume == lease) { + stream->activeConsume.clear(); + } + }); + CDCConsumeReply reply = co_await lease->waitForReply(consumeReply(stream, request.cursor)); + // Record proof before send(), whose callbacks may run synchronously. Empty or capped replies do not + // extend the speculative horizon; neither does replaying an already issued cursor. + const bool armed = stream->readAhead.issueReply(reply.lastConsumedVersion, + stream->bufferedThrough, + stream->minVersion, + HasMutations(!reply.mutations.empty())); + request.reply.send(reply); + if (armed) { + refreshStreamTags(stream); } - - Version begin = request.cursor.lastConsumedVersion == invalidVersion ? stream->minVersion - : request.cursor.lastConsumedVersion + 1; - if (begin < stream->minVersion) { - throw transaction_too_old(); + } catch (Error& e) { + if (e.code() == error_code_actor_cancelled) { + throw; } + request.reply.sendError(e); + } +} - auto buffered = - co_await race(waitForBufferedVersion(stream, begin), delay(SERVER_KNOBS->CDC_PROXY_CONSUME_POLL_TIMEOUT)); - if (buffered.index() == 1) { - CODE_PROBE(true, "CDC proxy expires an idle consume lease"); - CDCConsumeReply reply; - reply.lastConsumedVersion = request.cursor.lastConsumedVersion; - request.reply.send(reply); - co_return; - } - if (stream->tooOld) { - throw transaction_too_old(); - } - if (stream->bufferLimitExceeded) { - throw server_overloaded(); - } - if (!stream->active) { - throw wrong_shard_server(); +Future CDCProxy::consumeReply(Reference stream, CDCCursor cursor) { + const CDCStreamReadState metadata = + co_await readCDCStreamState(cx, cursor.streamId, id, true, PrioritizeDrain::True); + if (stream->tooOld) { + throw transaction_too_old(); + } + if (!stream->active) { + throw wrong_shard_server(); + } + CODE_PROBE(stream->minVersion < metadata.minVersion, "Native CDC consume reconciles a durable acknowledgement"); + reconcileStreamMetadata(stream, metadata); + if (!stream->readAhead.provesCursor(cursor.lastConsumedVersion, stream->minVersion)) { + // Prefetched data is not proof of delivery. A cursor must have been issued in a reply or covered by a + // durable acknowledgement; otherwise it could skip unread data and manufacture more read-ahead work. + if (cursor.lastConsumedVersion > metadata.readVersion) { + CODE_PROBE(true, "CDC proxy rejects a consume cursor beyond its transaction read version"); + } else { + CODE_PROBE(true, "CDC proxy rejects an unproven consume cursor"); } + throw client_invalid_operation(); + } + + Version begin = cursor.lastConsumedVersion == invalidVersion ? stream->minVersion : cursor.lastConsumedVersion + 1; + if (begin < stream->minVersion) { + throw transaction_too_old(); + } + if (begin > std::max(cursor.lastConsumedVersion, metadata.readVersion)) { CDCConsumeReply reply; - CDCConsumeReplySelection selection; - for (const auto& versioned : stream->mutations) { - if (versioned.version < begin) { - continue; - } - if (versioned.version > stream->bufferedThrough) { - break; - } - if (!selectCDCConsumeReplyVersion(&selection, - begin, - versioned.version, - estimatedCDCConsumeVersionBytes(versioned), - SERVER_KNOBS->CDC_PROXY_CONSUME_REPLY_BYTES)) { - break; - } - // Retain the already-accounted stream arena instead of copying mutation payloads for every reply. - reply.arena.dependsOn(versioned.arena()); - reply.mutations.push_back(reply.arena, VersionedMutationsRef(versioned.version, versioned.mutations)); + reply.lastConsumedVersion = cursor.lastConsumedVersion; + co_return reply; + } + + auto buffered = + co_await race(waitForBufferedVersion(stream, begin), delay(SERVER_KNOBS->CDC_PROXY_CONSUME_POLL_TIMEOUT)); + if (buffered.index() == 1) { + CODE_PROBE(true, "CDC proxy expires an idle consume lease"); + CDCConsumeReply reply; + reply.lastConsumedVersion = cursor.lastConsumedVersion; + co_return reply; + } + if (stream->tooOld) { + throw transaction_too_old(); + } + if (stream->bufferLimitExceeded) { + throw server_overloaded(); + } + if (!stream->active) { + throw wrong_shard_server(); + } + + CDCConsumeReply reply; + CDCConsumeReplySelection selection; + const Version replyThrough = + boundedCDCConsumeReplyThrough(cursor.lastConsumedVersion, metadata.readVersion, stream->bufferedThrough); + for (const auto& versioned : stream->mutations) { + if (versioned.version < begin) { + continue; } - if (selection.firstVersionTooLarge) { - CODE_PROBE( - true, "CDC proxy rejects one consume version larger than its reply budget", probe::decoration::rare); - TraceEvent(SevWarn, "CDCProxyConsumeVersionExceedsReplyLimit", id) - .detail("StreamId", stream->streamId) - .detail("Version", begin) - .detail("ReplyLimit", SERVER_KNOBS->CDC_PROXY_CONSUME_REPLY_BYTES); - throw server_overloaded(); - } - reply.lastConsumedVersion = selectedCDCConsumeReplyThrough(selection, stream->bufferedThrough); - request.reply.send(reply); - } catch (Error& e) { - if (e.code() == error_code_actor_cancelled) { - throw; + if (versioned.version > replyThrough) { + break; } - request.reply.sendError(e); + if (!selectCDCConsumeReplyVersion(&selection, + begin, + versioned.version, + estimatedCDCConsumeVersionBytes(versioned), + SERVER_KNOBS->CDC_PROXY_CONSUME_REPLY_BYTES)) { + break; + } + // Retain the already-accounted stream arena instead of copying mutation payloads for every reply. + reply.arena.dependsOn(versioned.arena()); + reply.mutations.push_back(reply.arena, VersionedMutationsRef(versioned.version, versioned.mutations)); + } + if (selection.firstVersionTooLarge) { + CODE_PROBE(true, "CDC proxy rejects one consume version larger than its reply budget", probe::decoration::rare); + TraceEvent(SevWarn, "CDCProxyConsumeVersionExceedsReplyLimit", id) + .detail("StreamId", stream->streamId) + .detail("Version", begin) + .detail("ReplyLimit", SERVER_KNOBS->CDC_PROXY_CONSUME_REPLY_BYTES); + throw server_overloaded(); } + reply.lastConsumedVersion = selectedCDCConsumeReplyThrough(selection, replyThrough); + co_return reply; } Future CDCProxy::acknowledge(CDCAckRequest request) { @@ -1685,7 +2131,8 @@ Future CDCProxy::acknowledge(CDCAckRequest request) { if (request.version < 0 || request.version >= std::numeric_limits::max() - 1) { throw client_invalid_operation(); } - const CDCStreamReadState metadata = co_await readCDCStreamState(cx, request.streamId, id, false); + const CDCStreamReadState metadata = + co_await readCDCStreamState(cx, request.streamId, id, false, PrioritizeDrain::True); if (metadata.minVersion <= request.version) { throw client_invalid_operation(); } @@ -1705,7 +2152,7 @@ Future CDCProxy::acknowledge(CDCAckRequest request) { // Reconcile the new owner's in-memory frontier to that already verified watermark. const Version minVersion = metadata.minVersion; CODE_PROBE(stream->minVersion < minVersion, "CDC proxy reconciles a durable stream acknowledgement"); - reconcileStreamMinVersion(stream, minVersion); + reconcileStreamMetadata(stream, metadata); requestAcknowledgedDataPop(); request.reply.send(Void()); } catch (Error& e) { @@ -1718,7 +2165,7 @@ Future CDCProxy::acknowledge(CDCAckRequest request) { Future CDCProxy::registerStream(CDCRegisterStreamRequest request) { try { - const CDCStreamId streamId = co_await registerNativeCdcStream(cx, request.name, request.keys, id); + const CDCStreamId streamId = co_await registerNativeCdcStream(cx, request.name, request.ranges, id); request.reply.send(CDCRegisterStreamReply(streamId)); } catch (Error& e) { if (e.code() == error_code_actor_cancelled) { @@ -1809,7 +2256,7 @@ Future CDCProxy::serveStatusRequests(FutureStreambufferedThrough; streamStatus.bufferedBytes = stream->bufferedBytes; streamStatus.readDemand = stream->readDemand; - streamStatus.activeConsumeRequests = stream->activeConsumes; + streamStatus.activeConsumeRequests = stream->activeConsume.isValid() ? 1 : 0; streamStatus.tooOld = stream->tooOld; streamStatus.bufferLimitExceeded = stream->bufferLimitExceeded; } @@ -2021,24 +2468,811 @@ Future cdcProxyServer(CDCProxyInterface proxy, } } -TEST_CASE("/NativeCDC/ProxyMutationFiltering") { - const KeyRangeRef keys("c"_sr, "m"_sr); +TEST_CASE("/NativeCDC/ConsumeLeaseSupersession") { + const UID consumerId(1, 2); + auto lease = makeReference(consumerId); + ASSERT(lease->belongsTo(consumerId)); + ASSERT(!lease->belongsTo(UID(3, 4))); + ASSERT(!lease->belongsTo(Optional())); + ASSERT(!makeReference(UID())->belongsTo(UID())); + + Promise pendingReply; + Future original = lease->waitForReply(pendingReply.getFuture()); + ASSERT(!original.isReady()); + lease->supersede(); + ASSERT(original.isReady() && original.isError()); + ASSERT_EQ(original.getError().code(), error_code_request_maybe_delivered); + return Void(); +} + +namespace { + +class CDCPrefetchTestCursor final : public IReplayPeekCursor, public ReferenceCounted { + Standalone payload; + Optional input; + Future ready; + Version messageVersion; + LogMessageVersion position; + bool fetched = false; + bool done = false; + bool containsMutation; + int mutationCount; + int consumedMutations = 0; + int fetches = 0; + Promise fetchStarted; + +public: + explicit CDCPrefetchTestCursor(Future ready, + bool containsMutation = true, + Version version = 100, + int mutationCount = 1) + : ready(ready), messageVersion(version), position(version), containsMutation(containsMutation), + mutationCount(mutationCount) { + BinaryWriter writer(AssumeVersion(g_network->protocolVersion())); + writer << MutationRef(MutationRef::SetValue, "k"_sr, "value"_sr); + payload = writer.toValue(); + } + int fetchCount() const { return fetches; } + Future onFetchStarted() { return fetchStarted.getFuture(); } + void setProtocolVersion(ProtocolVersion version) override { + input = ArenaReader(payload.arena(), payload, AssumeVersion(version)); + } + bool hasMessage() const override { return fetched && !done && containsMutation; } + VectorRef getTags() const override { return {}; } + Arena& arena() override { return payload.arena(); } + ArenaReader* reader() override { return &input.get(); } + StringRef getMessage() override { return payload; } + StringRef getMessageWithTags() override { return payload; } + void nextMessage() override { + if (++consumedMutations < mutationCount) { + setProtocolVersion(input.get().protocolVersion()); + return; + } + done = true; + position = LogMessageVersion(messageVersion + 1); + } + Future getMore(TaskPriority taskID) override { + ++fetches; + if (fetchStarted.canBeSet()) { + fetchStarted.send(Void()); + } + co_await ready; + fetched = true; + if (!containsMutation) { + position = LogMessageVersion(messageVersion + 1); + } + co_return; + } + bool isExhausted() const override { return fetched && !hasMessage(); } + LogMessageVersion const& version() const override { return position; } + Version popped() const override { return 0; } + Version getMinKnownCommittedVersion() const override { return messageVersion; } + int64_t getMaxRetainedReplyCount() const override { return 1; } + void setReplyByteLimit(int limitBytes) override { ASSERT_GT(limitBytes, int64_t(payload.size()) * mutationCount); } + Optional getPrimaryPeekLocation() const override { return {}; } + Optional getCurrentPeekLocation() const override { return {}; } + Version getMaxKnownVersion() const override { return messageVersion; } + Reference cloneNoMore() override { + auto clone = makeReference(Void(), containsMutation, messageVersion, mutationCount); + clone->position = position; + clone->consumedMutations = consumedMutations; + clone->fetched = fetched; + clone->done = done; + return clone; + } + void advanceTo(LogMessageVersion next) override { + if (next > position) { + consumedMutations = mutationCount; + done = true; + position = LogMessageVersion(messageVersion + 1); + } + } + void addref() override { ReferenceCounted::addref(); } + void delref() override { ReferenceCounted::delref(); } +}; + +class CDCProxyPrefetchTest { + CDCProxy proxy; + Reference tag = makeReference(Tag(tagLocalityCDC, 0)); + + Reference addStream(CDCStreamId id) { + auto stream = makeReference(id); + stream->initialized = true; + stream->minVersion = 1; + stream->bufferedThrough = 99; + stream->metadataReadVersion = 1000; + stream->ranges = std::vector{ KeyRangeRef("a"_sr, "z"_sr) }; + stream->tagIntervals.emplace_back(tag->tag, 1, 200); + stream->tagIntervals.back().bufferedThrough = 99; + ASSERT(stream->readAhead.issueReply(99, 99, 1, HasMutations::True)); + proxy.streams[id] = stream; + proxy.tags[tag->tag] = tag; + tag->streamIds.insert(id); + return stream; + } + +public: + static Future sameFrontierDemand(bool release, bool expire = false) { + CDCProxyPrefetchTest test; + auto stream = test.addStream(1); + Future waiter; + if (release) { + waiter = test.proxy.waitForBufferedVersion(stream, 100); + ASSERT_EQ(stream->readDemand, 1); + } + Promise ready; + auto cursor = makeReference(ready.getFuture()); + auto fetchStarted = cursor->onFetchStarted(); + auto work = test.proxy.bufferTagCursor(test.tag, 100, cursor, Never(), Prefetch::True); + auto start = co_await race(fetchStarted, work); + ASSERT_EQ(start.index(), 0); + ASSERT_EQ(cursor->fetchCount(), 1); + ASSERT(stream->readAhead.claimedBy(test.tag.getPtr())); + if (release) { + waiter.cancel(); // Exercise the actual waiter's ScopeExit, not a direct count mutation. + } else { + waiter = test.proxy.waitForBufferedVersion(stream, 100); + } + co_await delay(0); + ASSERT_EQ(stream->readDemand, release ? 0 : 1); + ASSERT(!work.isReady()); + ASSERT(stream->readAhead.claimedBy(test.tag.getPtr())); + if (expire) { + // Removing same-frontier demand must not promote the pass or renew its finite deadline. + ASSERT(release); + co_await work; + ASSERT_EQ(stream->bufferedThrough, 99); + ASSERT(stream->mutations.empty()); + } else { + ready.send(Void()); + co_await work; + if (!release) { + co_await waiter; + } + ASSERT_EQ(stream->bufferedThrough, 100); + ASSERT_EQ(stream->mutations.size(), 1); + ASSERT(!stream->readAhead.provesCursor(100, stream->minVersion)); + } + ASSERT_EQ(cursor->fetchCount(), 1); + ASSERT_EQ(stream->readDemand, 0); + ASSERT(!stream->readAhead.claimedBy(test.tag.getPtr())); + ASSERT(!test.proxy.nextTagPrefetchVersion(test.tag).present()); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), test.proxy.bufferedBytes); + test.proxy.clearBufferedMutations(stream); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), 0); + co_return; + } + + static Future sharedDemand(Version next) { + CDCProxyPrefetchTest test; + auto first = test.addStream(1); + auto second = test.addStream(2); + second->readAhead.cancel(); + second->bufferedThrough = next - 1; + second->tagIntervals.back().bufferedThrough = next - 1; + Promise ready; + auto cursor = makeReference(ready.getFuture()); + auto fetchStarted = cursor->onFetchStarted(); + auto work = test.proxy.bufferTagCursor(test.tag, 100, cursor, Never(), Prefetch::True); + auto start = co_await race(fetchStarted, work); + ASSERT_EQ(start.index(), 0); + co_await delay(0); + ASSERT_EQ(cursor->fetchCount(), 1); + auto waiter = test.proxy.waitForBufferedVersion(second, next); + co_await delay(0); + if (next < 100) { + co_await work; + ASSERT_EQ(test.proxy.nextTagReadVersion(test.tag).get(), next); + ASSERT_EQ(first->bufferedThrough, 99); + ASSERT_EQ(second->bufferedThrough, next - 1); + ASSERT(first->mutations.empty()); + ASSERT(second->mutations.empty()); + waiter.cancel(); + } else { + ASSERT(!work.isReady()); + ASSERT_EQ(test.proxy.nextTagReadVersion(test.tag).get(), 100); + ready.send(Void()); + co_await work; + ASSERT_EQ(first->bufferedThrough, 100); + ASSERT_EQ(first->mutations.size(), 1); + if (next == 100) { + co_await waiter; + ASSERT_EQ(second->bufferedThrough, 100); + ASSERT_EQ(second->mutations.size(), 1); + } else { + ASSERT(!waiter.isReady()); + ASSERT_EQ(second->bufferedThrough, next - 1); + ASSERT(second->mutations.empty()); + waiter.cancel(); + } + } + ASSERT_EQ(second->readDemand, 0); + ASSERT(!first->readAhead.claimedBy(test.tag.getPtr())); + ASSERT(!first->readAhead.armedFor(first->bufferedThrough)); + ASSERT(!first->readAhead.provesCursor(100, first->minVersion)); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), test.proxy.bufferedBytes); + test.proxy.clearBufferedMutations(first); + test.proxy.clearBufferedMutations(second); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), 0); + co_return; + } + + static Future lastDemandLeaves() { + CDCProxyPrefetchTest test; + auto stream = test.addStream(1); + stream->readAhead.cancel(); + auto awakened = test.tag->refresh.onTrigger(); + auto waiter = test.proxy.waitForBufferedVersion(stream, 100); + ASSERT(awakened.isReady()); // No interest -> real demand still wakes a dormant tag. + auto cursor = makeReference(Never()); + auto fetchStarted = cursor->onFetchStarted(); + auto work = test.proxy.bufferTagCursor(test.tag, 100, cursor, Never(), Prefetch::False); + auto start = co_await race(fetchStarted, work); + ASSERT_EQ(start.index(), 0); + co_await delay(0); + ASSERT_EQ(cursor->fetchCount(), 1); + ASSERT(!work.isReady()); + waiter.cancel(); + co_await work; + ASSERT_EQ(stream->readDemand, 0); + ASSERT(!test.proxy.nextTagReadVersion(test.tag).present()); + ASSERT_EQ(stream->bufferedThrough, 99); + ASSERT(stream->mutations.empty()); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), 0); + co_return; + } - Optional inRange = clipCDCMutation(MutationRef(MutationRef::SetValue, "d"_sr, "value"_sr), keys); - ASSERT(inRange.present()); - ASSERT_EQ(inRange.get().param1, "d"_sr); + static Future absentFrontierDemand() { + CDCProxyPrefetchTest test; + auto stream = test.addStream(1); + stream->readAhead.cancel(); + stream->tagIntervals.back().end = 100; + auto acquired = test.tag->refresh.onTrigger(); + auto waiter = test.proxy.waitForBufferedVersion(stream, 100); + ASSERT(acquired.isReady()); // Two absent frontiers must not count as equal, eligible work. + ASSERT(!test.proxy.nextTagReadVersion(test.tag).present()); + auto released = test.tag->refresh.onTrigger(); + waiter.cancel(); + ASSERT(released.isReady()); + ASSERT_EQ(stream->readDemand, 0); + co_return; + } + + static Future replaceTagOnRefresh(CDCProxy* proxy, + Reference first, + Reference oldTag, + Reference replacement) { + co_await first->refresh.onTrigger(); + oldTag->active = false; + oldTag->stopped.trigger(); + proxy->tags[replacement->tag] = replacement; + co_return; + } + + static Future demandRefreshReplacement() { + CDCProxyPrefetchTest test; + auto stream = test.addStream(1); + stream->readAhead.cancel(); + stream->bufferedThrough = 98; + stream->tagIntervals[0].end = 100; + stream->tagIntervals[0].bufferedThrough = 98; + auto second = makeReference(Tag(tagLocalityCDC, 1)); + stream->tagIntervals.emplace_back(second->tag, 100, 200); + stream->tagIntervals.back().bufferedThrough = 99; + auto other = test.addStream(2); + test.tag->streamIds.erase(2); + other->tagIntervals[0].tag = second->tag; + second->streamIds = { 1, 2 }; + test.proxy.tags[second->tag] = second; + CDCReadAheadPass pass(second); + pass.claim(other, 100); + ASSERT_EQ(test.proxy.nextTagReadVersion(second).get(), 100); + auto replacement = makeReference(second->tag); + replacement->streamIds = second->streamIds; + auto notified = replacement->refresh.onTrigger(); + auto replace = replaceTagOnRefresh(&test.proxy, test.tag, second, replacement); + auto waiter = test.proxy.waitForBufferedVersion(stream, 99); + // The first notification runs the replacement coroutine synchronously. The equal second frontier + // was observed on the old object, so it cannot suppress a notification to the replacement. + ASSERT(replace.isReady()); + ASSERT(notified.isReady()); + ASSERT(test.proxy.tags.at(second->tag).getPtr() == replacement.getPtr()); + waiter.cancel(); + ASSERT_EQ(stream->readDemand, 0); + co_return; + } + + static Future publish() { + CDCProxyPrefetchTest test; + auto first = test.addStream(1); + auto second = test.addStream(2); + auto dormant = test.addStream(3); + dormant->readAhead.cancel(); + auto cursor = makeReference(Void()); + ASSERT(!test.proxy.nextTagReadVersion(test.tag).present()); + ASSERT_EQ(test.proxy.nextTagPrefetchVersion(test.tag).get(), 100); + co_await test.proxy.bufferTagCursor(test.tag, 100, cursor, Never(), Prefetch::True); + ASSERT_EQ(cursor->fetchCount(), 1); + for (auto const& stream : { first, second }) { + ASSERT_EQ(stream->bufferedThrough, 100); + ASSERT_EQ(stream->mutations.size(), 1); + ASSERT(!stream->readAhead.claimedBy(test.tag.getPtr())); + ASSERT(!stream->readAhead.provesCursor(100, stream->minVersion)); + ASSERT(!stream->readAhead.armedFor(100)); + } + ASSERT_EQ(dormant->bufferedThrough, 99); + ASSERT(dormant->mutations.empty()); + ASSERT_EQ(dormant->bufferedBytes, 0); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), test.proxy.bufferedBytes); + ASSERT_EQ(test.proxy.bufferedBytes, first->bufferedBytes + second->bufferedBytes); + ASSERT_GT(test.proxy.bufferedBytes, 0); + ASSERT(!test.proxy.nextTagPrefetchVersion(test.tag).present()); + test.proxy.clearBufferedMutations(first); + test.proxy.clearBufferedMutations(second); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), 0); + co_return; + } + + static Future staggeredSharedTag() { + CDCProxyPrefetchTest test; + auto first = test.addStream(1); + auto second = test.addStream(2); + second->bufferedThrough = 199; + second->tagIntervals.back().bufferedThrough = 199; + second->tagIntervals.back().end = 300; + ASSERT(second->readAhead.issueReply(199, 199, second->minVersion, HasMutations::True)); + + auto firstCursor = makeReference(Void()); + ASSERT_EQ(test.proxy.nextTagPrefetchVersion(test.tag).get(), 100); + co_await test.proxy.bufferTagCursor(test.tag, 100, firstCursor, Never(), Prefetch::True); + ASSERT_EQ(firstCursor->fetchCount(), 1); + ASSERT_EQ(first->bufferedThrough, 100); + ASSERT_EQ(first->mutations.size(), 1); + ASSERT_EQ(second->bufferedThrough, 199); + ASSERT(second->mutations.empty()); + ASSERT(second->readAhead.armedFor(199)); + ASSERT_EQ(test.proxy.nextTagPrefetchVersion(test.tag).get(), 200); + + // The next speculative pass can require the entire buffer under buggified limits. + test.proxy.clearBufferedMutations(first); + auto secondCursor = makeReference(Void(), true, 200); + co_await test.proxy.bufferTagCursor(test.tag, 200, secondCursor, Never(), Prefetch::True); + ASSERT_EQ(secondCursor->fetchCount(), 1); + ASSERT_EQ(second->bufferedThrough, 200); + ASSERT_EQ(second->mutations.size(), 1); + ASSERT(!second->readAhead.provesCursor(200, second->minVersion)); + ASSERT(!test.proxy.nextTagPrefetchVersion(test.tag).present()); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), test.proxy.bufferedBytes); + test.proxy.clearBufferedMutations(second); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), 0); + co_return; + } + + static Future capacity(bool queued = false) { + CDCProxyPrefetchTest test; + auto stream = test.addStream(1); + const int64_t limit = SERVER_KNOBS->CDC_PROXY_BUFFER_BYTES; + co_await test.proxy.bufferLock.take(TaskPriority::TLogPeekReply, limit); + FlowLock::Releaser held(test.proxy.bufferLock, limit); + Future waiting; + if (queued) { + waiting = test.proxy.bufferLock.take(TaskPriority::TLogPeekReply, limit); + ASSERT(!waiting.isReady()); + const auto passLimit = calculateBufferPassLimits(limit, SERVER_KNOBS->MAXIMUM_PEEK_BYTES, 1); + held.release(std::min(passLimit.get().reservationBytes, limit - 1)); + ASSERT_GT(test.proxy.bufferLock.waiters(), 0); + } + auto cursor = makeReference(Never()); + co_await test.proxy.bufferTagCursor(test.tag, 100, cursor, Never(), Prefetch::True); + ASSERT_EQ(cursor->fetchCount(), 0); + ASSERT(!stream->readAhead.armedFor(99)); + ASSERT(!stream->readAhead.claimedBy(test.tag.getPtr())); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), held.remaining); + if (queued) { + waiting.cancel(); + } + co_return; + } + + static Future extraCapacity() { + CDCProxyPrefetchTest test; + auto stream = test.addStream(1); + const int64_t limit = SERVER_KNOBS->CDC_PROXY_BUFFER_BYTES; + const auto passLimit = calculateBufferPassLimits(limit, SERVER_KNOBS->MAXIMUM_PEEK_BYTES, 1).get(); + if (passLimit.preferredBufferedBytes == passLimit.hardBufferedBytes) { + // Small randomized budgets have no valid batch requiring an additional reservation. + ASSERT_EQ(passLimit.reservationBytes, limit); + co_return; + } + co_await test.proxy.bufferLock.take(TaskPriority::TLogPeekReply, limit - passLimit.reservationBytes); + FlowLock::Releaser held(test.proxy.bufferLock, limit - passLimit.reservationBytes); + co_await test.proxy.bufferLock.take(TaskPriority::TLogPeekReply, passLimit.reservationBytes); + FlowLock::Releaser reservation(test.proxy.bufferLock, passLimit.reservationBytes); + CDCReadAheadPass pass(test.tag); + pass.claim(stream, 100); + CDCBufferSelection selection; + selection.selectedStreamIds.insert(1); + selection.selectedBytes = passLimit.preferredBufferedBytes + 1; + auto cursor = makeReference(Void()); + const auto result = test.proxy.materializeBufferSelection( + test.tag, cursor, 100, selection, passLimit.rawReplyBytes, reservation, limit, Never()); + // Waiting for an incremental reservation would deadlock against the other reader's held capacity. + ASSERT(result == CDCBufferTagPassResult::RETRY); + ASSERT_EQ(test.tag->nextPassReservation, passLimit.rawReplyBytes + selection.selectedBytes); + ASSERT_EQ(test.proxy.bufferLock.waiters(), 0); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), limit); + ASSERT(stream->mutations.empty()); + ASSERT_EQ(stream->bufferedThrough, 99); + co_return; + } + + static Future competingExpandedReservations() { + CDCProxyPrefetchTest test; + const int64_t limit = SERVER_KNOBS->CDC_PROXY_BUFFER_BYTES; + const int peekBytes = std::min(SERVER_KNOBS->MAXIMUM_PEEK_BYTES, limit / 2); + const auto passLimit = calculateBufferPassLimits(limit, peekBytes, 1).get(); + if (2 * passLimit.reservationBytes > limit) { + co_return; // This topology permits only one initial reader reservation. + } + const MutationRef mutation(MutationRef::SetValue, "k"_sr, "value"_sr); + const int64_t mutationBytes = mutation.expectedSize() + sizeof(MutationRef); + const int mutationCount = + std::max(1, (peekBytes - int64_t(sizeof(VersionedMutationsRef))) / mutationBytes + 1); + const int64_t batchBytes = sizeof(VersionedMutationsRef) + mutationCount * mutationBytes; + BinaryWriter writer(AssumeVersion(g_network->protocolVersion())); + writer << mutation; + if (int64_t(writer.toValue().size()) * mutationCount >= peekBytes || + 2 * batchBytes + passLimit.rawReplyBytes >= 2 * passLimit.reservationBytes) { + co_return; // Tiny budgets cannot fit both expanded batches and the second reader's raw reply. + } + ASSERT_GT(batchBytes, passLimit.preferredBufferedBytes); + // Reserve unrelated capacity so the two initial readers exhaust the effective budget under any knob size. + const int64_t ballastBytes = limit - 2 * passLimit.reservationBytes; + co_await test.proxy.bufferLock.take(TaskPriority::TLogPeekReply, ballastBytes); + FlowLock::Releaser ballast(test.proxy.bufferLock, ballastBytes); + + auto first = test.addStream(1); + auto second = test.addStream(2); + auto secondTag = makeReference(Tag(tagLocalityCDC, 1)); + test.tag->streamIds.erase(2); + second->tagIntervals.front().tag = secondTag->tag; + secondTag->streamIds.insert(2); + test.proxy.tags[secondTag->tag] = secondTag; + first->readAhead.cancel(); + second->readAhead.cancel(); + auto firstDemand = test.proxy.waitForBufferedVersion(first, 100); + auto secondDemand = test.proxy.waitForBufferedVersion(second, 100); + Promise firstReady; + Promise secondReady; + auto firstCursor = makeReference(firstReady.getFuture(), true, 100, mutationCount); + auto secondCursor = makeReference(secondReady.getFuture(), true, 100, mutationCount); + auto firstFetch = firstCursor->onFetchStarted(); + auto secondFetch = secondCursor->onFetchStarted(); + auto firstPass = test.proxy.bufferTagCursor(test.tag, 100, firstCursor, Never(), Prefetch::False); + auto secondPass = test.proxy.bufferTagCursor(secondTag, 100, secondCursor, Never(), Prefetch::False); + co_await timeoutError(firstFetch && secondFetch, 5.0); + ASSERT(!firstPass.isReady() && !secondPass.isReady()); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), limit); + ASSERT_EQ(test.proxy.bufferedBytes, 0); + + firstReady.send(Void()); + ASSERT(co_await timeoutError(firstPass, 5.0) == CDCBufferTagPassResult::RETRY); + ASSERT_EQ(test.tag->nextPassReservation, passLimit.rawReplyBytes + batchBytes); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), ballastBytes + passLimit.reservationBytes); + auto firstRetryCursor = makeReference(Void(), true, 100, mutationCount); + auto firstRetry = test.proxy.bufferTagCursor(test.tag, 100, firstRetryCursor, Never(), Prefetch::False); + ASSERT(!firstRetry.isReady()); + ASSERT_EQ(firstRetryCursor->fetchCount(), 0); + ASSERT_GT(test.proxy.bufferLock.waiters(), 0); + secondReady.send(Void()); + ASSERT(co_await timeoutError(secondPass, 5.0) == CDCBufferTagPassResult::RETRY); + ASSERT_EQ(secondTag->nextPassReservation, passLimit.rawReplyBytes + batchBytes); + ASSERT(co_await timeoutError(firstRetry, 5.0) == CDCBufferTagPassResult::RETRY); + auto secondRetryCursor = makeReference(Void(), true, 100, mutationCount); + ASSERT(co_await timeoutError( + test.proxy.bufferTagCursor(secondTag, 100, secondRetryCursor, Never(), Prefetch::False), 5.0) == + CDCBufferTagPassResult::RETRY); + co_await timeoutError(firstDemand && secondDemand, 5.0); + ASSERT_EQ(firstRetryCursor->fetchCount(), 1); + ASSERT_EQ(secondRetryCursor->fetchCount(), 1); + for (const auto& stream : { first, second }) { + ASSERT_EQ(stream->bufferedThrough, 100); + ASSERT_EQ(stream->minVersion, 1); + ASSERT_EQ(stream->mutations.size(), 1); + ASSERT_EQ(stream->mutations.front().mutations.size(), mutationCount); + ASSERT_EQ(stream->bufferedBytes, batchBytes); + } + ASSERT_EQ(test.proxy.bufferedBytes, 2 * batchBytes); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), ballastBytes + test.proxy.bufferedBytes); + ASSERT_LE(test.proxy.peakActivePermits, limit); + test.proxy.clearBufferedMutations(first); + test.proxy.clearBufferedMutations(second); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), ballastBytes); + ballast.release(ballast.remaining); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), 0); + co_return; + } + + static Future interrupted(int action, Prefetch prefetch = Prefetch::True) { + CDCProxyPrefetchTest test; + auto stream = test.addStream(1); + Future demand; + if (!prefetch) { + stream->readAhead.cancel(); + demand = test.proxy.waitForBufferedVersion(stream, 100); + ASSERT_EQ(stream->readDemand, 1); + } + Promise ready; + Promise generationChanged; + auto cursor = makeReference(ready.getFuture()); + auto fetchStarted = cursor->onFetchStarted(); + auto work = test.proxy.bufferTagCursor(test.tag, 100, cursor, generationChanged.getFuture(), prefetch); + auto start = co_await race(fetchStarted, work); + ASSERT_EQ(start.index(), 0); + co_await delay(0); + ASSERT_EQ(cursor->fetchCount(), 1); + ASSERT_EQ(stream->readAhead.claimedBy(test.tag.getPtr()), bool(prefetch)); + if (action == 0) { + test.tag->refresh.trigger(); + } else if (action == 1) { + generationChanged.send(Void()); + } else if (action == 2) { + test.proxy.deactivateStream(stream); + test.proxy.streams[1] = makeReference(1); + } else if (action == 3) { + work.cancel(); + } else { + // No consumer comes back: the one speculative peek expires without granting another credit. + ASSERT_EQ(action, 4); + ASSERT(prefetch); + } + if (action != 3) { + co_await timeoutError(work, SERVER_KNOBS->BLOCKING_PEEK_TIMEOUT + 5.0); + } + ASSERT_EQ(stream->bufferedThrough, 99); + ASSERT(stream->mutations.empty()); + ASSERT(!stream->readAhead.claimedBy(test.tag.getPtr())); + ASSERT(!stream->readAhead.armedFor(99)); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), 0); + if (!prefetch) { + demand.cancel(); + ASSERT_EQ(stream->readDemand, 0); + } + co_return; + } + + static Future empty() { + CDCProxyPrefetchTest test; + auto stream = test.addStream(1); + auto cursor = makeReference(Void(), false); + co_await test.proxy.bufferTagCursor(test.tag, 100, cursor, Never(), Prefetch::True); + ASSERT_EQ(cursor->fetchCount(), 1); + ASSERT_EQ(stream->bufferedThrough, 100); + ASSERT(stream->mutations.empty()); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), 0); + ASSERT(!test.proxy.nextTagPrefetchVersion(test.tag).present()); + co_return; + } +}; + +} // namespace + +TEST_CASE("/NativeCDC/PrefetchCreditLifecycle") { + CDCStreamReadAhead credit; + auto tag = makeReference(Tag(tagLocalityCDC, 0)); + ASSERT(!credit.armedFor(99)); + ASSERT(!credit.issueReply(99, 99, 100, HasMutations::True)); // Already covered by the durable floor. + ASSERT( + !credit.issueReply(100, 100, 100, HasMutations::False)); // Empty progress is proved, but grants no lookahead. + ASSERT(credit.provesCursor(100, 100)); + ASSERT(!credit.issueReply(101, 102, 100, HasMutations::True)); // A capped reply has not drained the buffered tail. + ASSERT(credit.issueReply(102, 102, 100, HasMutations::True)); + ASSERT(credit.claim(tag.getPtr(), 102)); + ASSERT(!credit.issueReply(103, 103, 100, HasMutations::True)); // No credit banking while a pass is active. + credit.finish(tag.getPtr()); + ASSERT(!credit.armedFor(103)); + ASSERT(!credit.issueReply(103, 103, 100, HasMutations::True)); // Replayed cursor. + ASSERT(credit.issueReply(104, 104, 100, HasMutations::True)); + credit.cancel(); + ASSERT(!credit.claim(tag.getPtr(), 104)); + ASSERT(!credit.provesCursor(105, 100)); + ASSERT(credit.provesCursor(105, 106)); + return Void(); +} + +TEST_CASE("/NativeCDC/PrefetchCreditTailAndTags") { + auto stream = makeReference(1); + auto first = makeReference(Tag(tagLocalityCDC, 0)); + auto second = makeReference(Tag(tagLocalityCDC, 1)); + stream->initialized = true; + stream->minVersion = 1; + stream->bufferedThrough = 99; + stream->metadataReadVersion = 1000; + stream->tagIntervals.emplace_back(first->tag, 1, 101); + stream->tagIntervals.emplace_back(second->tag, 101, 200); + stream->tagIntervals[0].bufferedThrough = 99; + ASSERT(stream->readAhead.issueReply(99, 99, 1, HasMutations::True)); + ASSERT_EQ(nextCDCPrefetchVersion(stream, first).get(), 100); + stream->metadataReadVersion = 99; + ASSERT(!nextCDCPrefetchVersion(stream, first).present()); + stream->metadataReadVersion = 1000; + ASSERT(!nextCDCPrefetchVersion(stream, second).present()); + { + CDCReadAheadPass pass(first); + pass.claim(stream, 100); + ASSERT(hasCDCReadInterest(stream, first)); + ASSERT(!hasCDCReadInterest(stream, second)); + ASSERT(!nextCDCPrefetchVersion(stream, second).present()); + } + ASSERT(!stream->readAhead.claimedBy(first.getPtr())); + ASSERT(stream->readAhead.issueReply(100, 100, 1, HasMutations::True)); + stream->bufferedThrough = 102; // Real demand filled more data before the credit could start. + stream->tagIntervals[0].bufferedThrough = 100; + stream->tagIntervals[1].bufferedThrough = 102; + ASSERT(!nextCDCPrefetchVersion(stream, second).present()); + ASSERT( + !stream->readAhead.issueReply(101, 102, 1, HasMutations::True)); // Capped reply cannot revive the old credit. + ASSERT(!nextCDCPrefetchVersion(stream, second).present()); + ASSERT(stream->readAhead.issueReply(102, 102, 1, HasMutations::True)); + ASSERT_EQ(nextCDCPrefetchVersion(stream, second).get(), 103); + return Void(); +} + +TEST_CASE("/NativeCDC/PrefetchMaterializesSharedTag") { + return CDCProxyPrefetchTest::publish(); +} +TEST_CASE("/NativeCDC/PrefetchPreservesLaterSharedTagCredit") { + return CDCProxyPrefetchTest::staggeredSharedTag(); +} +TEST_CASE("/NativeCDC/PrefetchDeclinesUnavailableCapacity") { + return CDCProxyPrefetchTest::capacity(); +} +TEST_CASE("/NativeCDC/PrefetchDeclinesQueuedCapacity") { + return CDCProxyPrefetchTest::capacity(true); +} +TEST_CASE("/NativeCDC/DemandRetriesExpandedReservation") { + return CDCProxyPrefetchTest::extraCapacity(); +} +TEST_CASE("/NativeCDC/CompetingExpandedReservations") { + return CDCProxyPrefetchTest::competingExpandedReservations(); +} +TEST_CASE("/NativeCDC/PrefetchRefreshCancels") { + return CDCProxyPrefetchTest::interrupted(0); +} +TEST_CASE("/NativeCDC/PrefetchGenerationChangeCancels") { + return CDCProxyPrefetchTest::interrupted(1); +} +TEST_CASE("/NativeCDC/PrefetchReplacementCancels") { + return CDCProxyPrefetchTest::interrupted(2); +} +TEST_CASE("/NativeCDC/PrefetchActorCancellationReleases") { + return CDCProxyPrefetchTest::interrupted(3); +} +TEST_CASE("/NativeCDC/PrefetchIdleDeadline") { + return CDCProxyPrefetchTest::interrupted(4); +} +TEST_CASE("/NativeCDC/DemandRefreshCancels") { + return CDCProxyPrefetchTest::interrupted(0, Prefetch::False); +} +TEST_CASE("/NativeCDC/DemandGenerationChangeCancels") { + return CDCProxyPrefetchTest::interrupted(1, Prefetch::False); +} +TEST_CASE("/NativeCDC/DemandReplacementCancels") { + return CDCProxyPrefetchTest::interrupted(2, Prefetch::False); +} +TEST_CASE("/NativeCDC/DemandActorCancellationReleases") { + return CDCProxyPrefetchTest::interrupted(3, Prefetch::False); +} +TEST_CASE("/NativeCDC/PrefetchEmptyDoesNotRetry") { + return CDCProxyPrefetchTest::empty(); +} - Optional outOfRange = clipCDCMutation(MutationRef(MutationRef::SetValue, "z"_sr, "value"_sr), keys); - ASSERT(!outOfRange.present()); +TEST_CASE("/NativeCDC/DemandRefresh/PrefetchAcquirePreservesPeek") { + return CDCProxyPrefetchTest::sameFrontierDemand(false); +} +TEST_CASE("/NativeCDC/DemandRefresh/PrefetchReleasePreservesPeek") { + return CDCProxyPrefetchTest::sameFrontierDemand(true); +} +TEST_CASE("/NativeCDC/DemandRefresh/PrefetchReleasePreservesDeadline") { + return CDCProxyPrefetchTest::sameFrontierDemand(true, true); +} +TEST_CASE("/NativeCDC/DemandRefresh/SharedEarlierRestartsPeek") { + return CDCProxyPrefetchTest::sharedDemand(50); +} +TEST_CASE("/NativeCDC/DemandRefresh/SharedSamePreservesPeek") { + return CDCProxyPrefetchTest::sharedDemand(100); +} +TEST_CASE("/NativeCDC/DemandRefresh/SharedLaterPreservesPeek") { + return CDCProxyPrefetchTest::sharedDemand(101); +} +TEST_CASE("/NativeCDC/DemandRefresh/LastDemandCancelsPeek") { + return CDCProxyPrefetchTest::lastDemandLeaves(); +} +TEST_CASE("/NativeCDC/DemandRefresh/AbsentFrontiersNotify") { + return CDCProxyPrefetchTest::absentFrontierDemand(); +} +TEST_CASE("/NativeCDC/DemandRefresh/TagReplacementNotifiesCurrentObject") { + return CDCProxyPrefetchTest::demandRefreshReplacement(); +} - Optional clippedClear = clipCDCMutation(MutationRef(MutationRef::ClearRange, "a"_sr, "f"_sr), keys); - ASSERT(clippedClear.present()); - ASSERT_EQ(clippedClear.get().param1, "c"_sr); - ASSERT_EQ(clippedClear.get().param2, "f"_sr); +TEST_CASE("/NativeCDC/ProxyMutationFiltering") { + const std::vector ranges{ KeyRangeRef("c"_sr, "m"_sr), KeyRangeRef("q"_sr, "t"_sr) }; + std::vector filtered; + auto collect = [&filtered](MutationRef const& mutation) { filtered.push_back(mutation); }; + + for (const KeyRef key : { "c"_sr, "d"_sr, "q"_sr, "s"_sr }) { + filtered.clear(); + visitClippedCDCMutations(MutationRef(MutationRef::SetValue, key, "value"_sr), ranges, collect); + ASSERT_EQ(filtered.size(), 1); + ASSERT_EQ(filtered.front().param1, key); + } + for (const KeyRef key : { "a"_sr, "m"_sr, "n"_sr, "t"_sr, "z"_sr }) { + filtered.clear(); + visitClippedCDCMutations(MutationRef(MutationRef::SetValue, key, "value"_sr), ranges, collect); + ASSERT(filtered.empty()); + } + + visitClippedCDCMutations(MutationRef(MutationRef::ClearRange, "a"_sr, "r"_sr), ranges, collect); + ASSERT_EQ(filtered.size(), 2); + ASSERT_EQ(filtered[0].param1, "c"_sr); + ASSERT_EQ(filtered[0].param2, "m"_sr); + ASSERT_EQ(filtered[1].param1, "q"_sr); + ASSERT_EQ(filtered[1].param2, "r"_sr); + + filtered.clear(); + visitClippedCDCMutations(MutationRef(MutationRef::ClearRange, "l"_sr, "z"_sr), ranges, collect); + ASSERT_EQ(filtered.size(), 2); + ASSERT_EQ(filtered[0].param1, "l"_sr); + ASSERT_EQ(filtered[0].param2, "m"_sr); + ASSERT_EQ(filtered[1].param1, "q"_sr); + ASSERT_EQ(filtered[1].param2, "t"_sr); + + filtered.clear(); + visitClippedCDCMutations(MutationRef(MutationRef::ClearRange, "m"_sr, "q"_sr), ranges, collect); + visitClippedCDCMutations(MutationRef(MutationRef::ClearRange, "d"_sr, "d"_sr), ranges, collect); + visitClippedCDCMutations(MutationRef(MutationRef::ClearRange, "t"_sr, "z"_sr), ranges, collect); + ASSERT(filtered.empty()); - Optional excludedClear = clipCDCMutation(MutationRef(MutationRef::ClearRange, "n"_sr, "z"_sr), keys); - ASSERT(!excludedClear.present()); + return Void(); +} +TEST_CASE("/NativeCDC/ProxyMutationFiltering/MultiRangeBatch") { + const std::vector ranges{ KeyRangeRef("a"_sr, "c"_sr), KeyRangeRef("x"_sr, "z"_sr) }; + auto stream = makeReference(1); + CDCBufferedBatch batch; + const std::vector input{ MutationRef(MutationRef::SetValue, "b"_sr, "before"_sr), + MutationRef(MutationRef::ClearRange, "b"_sr, "y"_sr), + MutationRef(MutationRef::SetValue, "x"_sr, "after"_sr), + MutationRef(MutationRef::SetValue, "m"_sr, "gap"_sr) }; + for (const auto& mutation : input) { + visitClippedCDCMutations( + mutation, ranges, [&](MutationRef const& clipped) { addMutationToBatch(stream, &batch, 100, clipped); }); + } + ASSERT_EQ(batch.mutations.size(), 1); + const auto& versioned = batch.mutations.front(); + ASSERT_EQ(versioned.version, 100); + ASSERT_EQ(versioned.mutations.size(), 4); + const std::vector expected{ input[0], + MutationRef(MutationRef::ClearRange, "b"_sr, "c"_sr), + MutationRef(MutationRef::ClearRange, "x"_sr, "y"_sr), + input[2] }; + int64_t expectedBytes = sizeof(VersionedMutationsRef); + for (int i = 0; i < versioned.mutations.size(); ++i) { + ASSERT_EQ(versioned.mutations[i].type, expected[i].type); + ASSERT_EQ(versioned.mutations[i].param1, expected[i].param1); + ASSERT_EQ(versioned.mutations[i].param2, expected[i].param2); + expectedBytes += expected[i].expectedSize() + sizeof(MutationRef); + } + ASSERT_EQ(batch.bufferedBytes, expectedBytes); + + const int64_t versionBytes = estimatedCDCConsumeVersionBytes(versioned); + CDCConsumeReplySelection tooSmall; + ASSERT(!selectCDCConsumeReplyVersion(&tooSmall, 100, 100, versionBytes, versionBytes - 1)); + ASSERT(tooSmall.firstVersionTooLarge); + ASSERT_EQ(selectedCDCConsumeReplyThrough(tooSmall, 100), 99); + CDCConsumeReplySelection exactFit; + ASSERT(selectCDCConsumeReplyVersion(&exactFit, 100, 100, versionBytes, versionBytes)); + ASSERT_EQ(selectedCDCConsumeReplyThrough(exactFit, 100), 100); return Void(); } @@ -2085,6 +3319,17 @@ TEST_CASE("/NativeCDC/ProxyBufferCandidateSelection") { return Void(); } +TEST_CASE("/NativeCDC/CommitWaitDeadlineBudget") { + // Immediate legacy replies retain the fallback; time already spent in the pass is not charged twice. + ASSERT_EQ(remainingCommitWait(0.0, 0.4, 0.0), 0.4); + ASSERT_EQ(remainingCommitWait(0.0, 0.4, 0.2), 0.2); + ASSERT_EQ(remainingCommitWait(0.0, 0.4, 0.4), 0.0); + // Capacity waiting is part of the same pass, even when it exceeds the timeout before a peek is issued. + ASSERT_EQ(remainingCommitWait(0.0, 0.4, 2.0), 0.0); + ASSERT_EQ(remainingCommitWait(8.0, 0.5, 8.125), 0.375); + return Void(); +} + TEST_CASE("/NativeCDC/ProxyBufferPassLimits") { Optional replicated = calculateBufferPassLimits(1000, 100, 3); ASSERT(replicated.present()); @@ -2279,6 +3524,177 @@ TEST_CASE("/NativeCDC/CommittedDeliveryFrontier") { ASSERT_EQ(committedPeekThrough(149, 120), 120); ASSERT_EQ(committedPeekThrough(149, 200), 149); ASSERT_EQ(committedPeekThrough(103, 102), 102); + ASSERT_EQ(boundedCDCConsumeReplyThrough(100, 150, 200), 150); + ASSERT_EQ(boundedCDCConsumeReplyThrough(175, 150, 200), 175); + ASSERT_EQ(boundedCDCConsumeReplyThrough(invalidVersion, 150, 140), 140); + return Void(); +} + +TEST_CASE("/NativeCDC/ProxyHistoryReconciliation") { + const Tag oldTag(tagLocalityCDC, 1); + const Tag targetTag(tagLocalityCDC, 2); + auto stream = makeReference(1); + CDCStreamReadState initial; + initial.ranges = std::vector{ KeyRangeRef("a"_sr, "z"_sr) }; + initial.minVersion = 100; + initial.readVersion = 150; + initial.tagAssignments = { { 90, oldTag } }; + ASSERT(reconcileBufferedStreamMetadata(stream, initial).historyChanged); + stream->tagIntervals.front().bufferedThrough = 150; + stream->bufferedThrough = 150; + const auto addBufferedVersion = [&](Version version) { + auto& buffered = stream->mutations.emplace_back(); + buffered.version = version; + buffered.mutations.push_back_deep(buffered.arena(), MutationRef(MutationRef::SetValue, "key"_sr, "value"_sr)); + stream->bufferedBytes += estimatedCDCConsumeVersionBytes(buffered); + }; + addBufferedVersion(149); + const int64_t preservedBytes = estimatedCDCConsumeVersionBytes(stream->mutations.front()); + + CDCStreamReadState migrating = initial; + migrating.readVersion = 250; + migrating.tagAssignments.emplace_back(200, targetTag); + const CDCStreamMetadataUpdate migrated = reconcileBufferedStreamMetadata(stream, migrating); + ASSERT(migrated.historyChanged); + ASSERT_EQ(migrated.releasedBytes, 0); + ASSERT_EQ(stream->bufferedBytes, preservedBytes); + ASSERT_EQ(stream->mutations.size(), 1); + ASSERT_EQ(stream->mutations.front().version, 149); + ASSERT_EQ(stream->tagIntervals.size(), 2); + ASSERT_EQ(stream->tagIntervals[0].end, 200); + ASSERT_EQ(stream->tagIntervals[0].bufferedThrough, 150); + ASSERT_EQ(stream->tagIntervals[1].begin, 200); + ASSERT_EQ(stream->tagIntervals[1].bufferedThrough, 199); + ASSERT_EQ(stream->bufferedThrough, 150); + + ASSERT(advanceStreamTagBufferedThrough(stream, oldTag, 199)); + ASSERT(!advanceStreamTagBufferedThrough(stream, targetTag, 250)); + addBufferedVersion(230); + addBufferedVersion(250); + CDCStreamReadState finalized = migrating; + finalized.minVersion = 220; + finalized.readVersion = 350; + finalized.tagAssignments = { { 200, targetTag } }; + const CDCStreamMetadataUpdate completed = reconcileBufferedStreamMetadata(stream, finalized); + ASSERT(completed.historyChanged); + ASSERT_EQ(completed.releasedBytes, preservedBytes); + ASSERT_EQ(stream->tagIntervals.size(), 1); + ASSERT_EQ(stream->tagIntervals.front().tag, targetTag); + ASSERT_EQ(stream->tagIntervals.front().begin, 220); + ASSERT_EQ(stream->bufferedThrough, 250); + ASSERT_EQ(stream->mutations.size(), 2); + ASSERT_EQ(stream->mutations.front().version, 230); + + const CDCStreamMetadataUpdate stale = reconcileBufferedStreamMetadata(stream, migrating); + ASSERT(!stale.historyChanged); + ASSERT(!stale.readVersionAdvanced); + ASSERT_EQ(stale.releasedBytes, 0); + ASSERT_EQ(stream->metadataReadVersion, 350); + ASSERT_EQ(stream->minVersion, 220); + ASSERT_EQ(stream->tagIntervals.size(), 1); + ASSERT_EQ(stream->bufferedThrough, 250); + + finalized.minVersion = 251; + finalized.readVersion = 400; + const CDCStreamMetadataUpdate acknowledged = reconcileBufferedStreamMetadata(stream, finalized); + ASSERT(!acknowledged.historyChanged); + ASSERT_GT(acknowledged.releasedBytes, 0); + ASSERT(stream->mutations.empty()); + ASSERT_EQ(stream->bufferedBytes, 0); + ASSERT_EQ(stream->bufferedThrough, 250); + return Void(); +} + +TEST_CASE("/NativeCDC/ProxyHistoryReconciliationAfterAcknowledgement") { + const Tag oldTag(tagLocalityCDC, 1); + const Tag targetTag(tagLocalityCDC, 2); + auto stream = makeReference(1); + CDCStreamReadState initial; + initial.minVersion = 100; + initial.readVersion = 150; + initial.tagAssignments = { { 90, oldTag } }; + reconcileBufferedStreamMetadata(stream, initial); + stream->tagIntervals.front().bufferedThrough = 180; + stream->bufferedThrough = 180; + + // The durable acknowledgement can be observed by the pop scanner before either migration history snapshot. + advanceStreamMinVersion(stream, 225); + CDCStreamReadState finalized = initial; + finalized.minVersion = 225; + finalized.readVersion = 250; + finalized.tagAssignments = { { 200, targetTag } }; + ASSERT(reconcileBufferedStreamMetadata(stream, finalized).historyChanged); + ASSERT_EQ(stream->tagIntervals.size(), 1); + ASSERT_EQ(stream->tagIntervals.front().tag, targetTag); + ASSERT_EQ(stream->tagIntervals.front().begin, 225); + ASSERT_EQ(stream->bufferedThrough, 224); + + CDCStreamReadState olderRead = finalized; + olderRead.minVersion = 200; + olderRead.readVersion = 210; + olderRead.tagAssignments = { { 90, oldTag }, { 200, targetTag } }; + ASSERT(!reconcileBufferedStreamMetadata(stream, olderRead).historyChanged); + ASSERT_EQ(stream->bufferedThrough, 224); + ASSERT_EQ(boundedCDCConsumeReplyThrough(224, olderRead.readVersion, stream->bufferedThrough), 224); + return Void(); +} + +TEST_CASE("/NativeCDC/ProxyRetagReadAndAcknowledgementEligibility") { + const Tag firstTag(tagLocalityCDC, 1); + const Tag secondTag(tagLocalityCDC, 2); + auto stream = makeReference(1); + CDCStreamReadState metadata; + metadata.minVersion = 100; + metadata.readVersion = 400; + metadata.tagAssignments = { { 90, firstTag }, { 200, secondTag }, { 300, firstTag } }; + reconcileBufferedStreamMetadata(stream, metadata); + stream->readDemand = 1; + + ASSERT_EQ(eligibleTagReadInterval(*stream, firstTag).get(), 0); + ASSERT(!eligibleTagReadInterval(*stream, secondTag).present()); + ASSERT(canBufferTagVersion(*stream, firstTag, 150)); + ASSERT(!canBufferTagVersion(*stream, secondTag, 200)); + ASSERT(!canBufferTagVersion(*stream, firstTag, 350)); + ASSERT(!advanceStreamTagBufferedThrough(stream, firstTag, 150)); + stream->metadataReadVersion = 150; + ASSERT(!eligibleTagReadInterval(*stream, firstTag).present()); + ASSERT(!eligibleTagReadInterval(*stream, secondTag).present()); + + stream->metadataReadVersion = metadata.readVersion; + ASSERT(advanceStreamTagBufferedThrough(stream, firstTag, 400)); + ASSERT_EQ(stream->bufferedThrough, 199); + ASSERT_EQ(stream->tagIntervals[2].bufferedThrough, 299); + ASSERT_EQ(eligibleTagReadInterval(*stream, secondTag).get(), 1); + ASSERT(canBufferTagVersion(*stream, secondTag, 200)); + ASSERT(!canBufferTagVersion(*stream, firstTag, 350)); + ASSERT(advanceStreamTagBufferedThrough(stream, secondTag, 400)); + ASSERT_EQ(stream->bufferedThrough, 299); + ASSERT(canBufferTagVersion(*stream, firstTag, 350)); + ASSERT(!advanceStreamTagBufferedThrough(stream, firstTag, 450)); + ASSERT_EQ(stream->bufferedThrough, metadata.readVersion); + ASSERT_EQ(stream->minVersion, 100); + + // Acknowledgements can unlock an interval before its prefix is read. + stream = makeReference(1); + reconcileBufferedStreamMetadata(stream, metadata); + stream->readDemand = 1; + + const Optional initialReadInterval = firstIncompleteTagInterval(*stream); + advanceStreamMinVersion(stream, 225); + ASSERT(initialReadInterval != firstIncompleteTagInterval(*stream)); + ASSERT_EQ(eligibleTagReadInterval(*stream, secondTag).get(), 1); + ASSERT(canBufferTagVersion(*stream, secondTag, 225)); + ASSERT(!canBufferTagVersion(*stream, secondTag, 224)); + + const Optional acknowledgedReadInterval = firstIncompleteTagInterval(*stream); + metadata.minVersion = 325; + metadata.readVersion = 350; + const CDCStreamMetadataUpdate update = reconcileBufferedStreamMetadata(stream, metadata); + ASSERT(!update.historyChanged); + ASSERT(!update.readVersionAdvanced); + ASSERT(acknowledgedReadInterval != firstIncompleteTagInterval(*stream)); + ASSERT_EQ(eligibleTagReadInterval(*stream, firstTag).get(), 2); + ASSERT(canBufferTagVersion(*stream, firstTag, 325)); return Void(); } diff --git a/fdbserver/checkpoint/BulkSstFiles.cpp b/fdbserver/checkpoint/BulkSstFiles.cpp new file mode 100644 index 00000000000..7444d343f78 --- /dev/null +++ b/fdbserver/checkpoint/BulkSstFiles.cpp @@ -0,0 +1,180 @@ +/* + * BulkSstFiles.cpp + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "fdbserver/checkpoint/BulkSstFiles.h" +#include "fdbserver/checkpoint/RocksDBCheckpointUtils.h" +#include "fdbserver/core/BulkLoadUtil.h" +#include "fdbserver/core/Knobs.h" +#include "fdbserver/core/StorageMetrics.h" +#include "flow/genericactors.h" + +// Generate SST file given the input sortedKVS to the input filePath. +// TODO(BulkDump): This copy of sortedKVS can be a slow task if data is large. +void writeKVSToSSTFile(std::string filePath, std::map& sortedKVS, UID logId) { + const std::string absFilePath = abspath(filePath); + // Check file + if (fileExists(absFilePath)) { + TraceEvent(SevWarn, "SSBulkDumpRetriableError", logId) + .detail("Reason", "exist old File when writeKVSToSSTFile") + .detail("DataFilePathLocal", absFilePath); + ASSERT_WE_THINK(false); + throw retry(); + } + // Dump data to file + std::unique_ptr sstWriter = newRocksDBSstFileWriter(); + sstWriter->open(absFilePath); + for (const auto& [key, value] : sortedKVS) { + sstWriter->write(key, value); // assuming sorted + } + if (!sstWriter->finish()) { + // Unexpected: having data but failed to finish + TraceEvent(SevWarn, "SSBulkDumpRetriableError", logId) + .detail("Reason", "failed to finish data sst writer when writeKVSToSSTFile") + .detail("DataFilePath", absFilePath); + ASSERT_WE_THINK(false); + throw retry(); + } + return; +} + +Future dumpDataFileToLocalDirectory(UID logId, + std::shared_ptr rangeDumpRawData, + BulkLoadFileSet localFileSet, + BulkLoadFileSet remoteFileSet, + BulkLoadByteSampleSetting byteSampleSetting, + Version dumpVersion, + KeyRange dumpRange, + BulkLoadType dumpType, + BulkLoadTransportMethod transportMethod) { + // Step 1: Clean up local folder + resetFileFolder((abspath(localFileSet.getFolder()))); + + // Step 2: Dump data to file + bool containDataFile = false; + if (!rangeDumpRawData->kvs.empty()) { + writeKVSToSSTFile(abspath(localFileSet.getDataFileFullPath()), rangeDumpRawData->kvs, logId); + containDataFile = true; + } else { + ASSERT(rangeDumpRawData->sampled.empty()); + containDataFile = false; + } + + // Step 3: Dump sample to file + bool containByteSampleFile = false; + if (!rangeDumpRawData->sampled.empty()) { + writeKVSToSSTFile(abspath(localFileSet.getBytesSampleFileFullPath()), rangeDumpRawData->sampled, logId); + containByteSampleFile = true; + } else { + containByteSampleFile = false; + } + + // Step 4: Generate manifest file + if (fileExists(abspath(localFileSet.getManifestFileFullPath()))) { + TraceEvent(SevWarn, "SSBulkDumpRetriableError", logId) + .detail("Reason", "exist old manifestFile") + .detail("ManifestFilePathLocal", abspath(localFileSet.getManifestFileFullPath())); + ASSERT_WE_THINK(false); + throw retry(); + } + BulkLoadFileSet fileSetRemote(remoteFileSet.getRootPath(), + remoteFileSet.getRelativePath(), + remoteFileSet.getManifestFileName(), + containDataFile ? remoteFileSet.getDataFileName() : std::string(), + containByteSampleFile ? remoteFileSet.getByteSampleFileName() : std::string(), + BulkLoadChecksum()); + BulkLoadManifest manifestMetadata(fileSetRemote, + dumpRange.begin, + dumpRange.end, + dumpVersion, + rangeDumpRawData->kvsBytes, + rangeDumpRawData->kvs.size(), + byteSampleSetting, + dumpType, + transportMethod); + std::string manifestStr = manifestMetadata.toString(); + std::shared_ptr manifest = std::make_shared(std::move(manifestStr)); + co_await writeBulkFileBytes(abspath(localFileSet.getManifestFileFullPath()), manifest); + co_return manifestMetadata; +} + +// Return true if generated the byte sampling file. Otherwise, return false. +// TODO(BulkDump): directly read from special key space. +Future doBytesSamplingOnDataFile(std::string dataFileFullPath, // input file + std::string byteSampleFileFullPath, // output file + UID logId) { + int counter = 0; + bool res = false; + int retryCount = 0; + double startTime = now(); + while (true) { + Error err; + try { + std::unique_ptr sstWriter = newRocksDBSstFileWriter(); + sstWriter->open(abspath(byteSampleFileFullPath)); + bool anySampled = false; + std::unique_ptr reader = newRocksDBSstFileReader(); + reader->open(abspath(dataFileFullPath)); + while (reader->hasNext()) { + KeyValue kv = reader->next(); + ByteSampleInfo sampleInfo = isKeyValueInSample(kv); + if (sampleInfo.inSample) { + sstWriter->write(kv.key, BinaryWriter::toValue(sampleInfo.sampledSize, Unversioned())); + anySampled = true; + counter++; + if (counter > SERVER_KNOBS->BULKLOAD_BYTE_SAMPLE_BATCH_KEY_COUNT) { + co_await yield(); + counter = 0; + } + } + } + // It is possible that no key is sampled + // This can happen when the data to sample is small + // In this case, no SST sample byte file is generated + if (anySampled) { + ASSERT(sstWriter->finish()); + res = true; + } else { + ASSERT(!sstWriter->finish()); + deleteFile(abspath(byteSampleFileFullPath)); + } + break; + } catch (Error& e) { + err = e; + } + if (err.code() == error_code_actor_cancelled) { + throw err; + } + TraceEvent(SevWarn, "SSBulkLoadTaskSamplingError", logId) + .errorUnsuppressed(err) + .detail("DataFileFullPath", dataFileFullPath) + .detail("ByteSampleFileFullPath", byteSampleFileFullPath) + .detail("Duration", now() - startTime) + .detail("RetryCount", retryCount); + co_await delay(5.0); + deleteFile(abspath(byteSampleFileFullPath)); + retryCount++; + } + TraceEvent(bulkLoadVerboseEventSev(), "SSBulkLoadTaskSamplingComplete", logId) + .detail("DataFileFullPath", dataFileFullPath) + .detail("ByteSampleFileFullPath", byteSampleFileFullPath) + .detail("Duration", now() - startTime) + .detail("RetryCount", retryCount); + co_return res; +} diff --git a/fdbserver/checkpoint/CMakeLists.txt b/fdbserver/checkpoint/CMakeLists.txt new file mode 100644 index 00000000000..5ee1e12189f --- /dev/null +++ b/fdbserver/checkpoint/CMakeLists.txt @@ -0,0 +1,26 @@ +fdb_find_sources(FDBSERVER_CHECKPOINT_SRCS) + +add_flow_target(STATIC_LIBRARY NAME fdbserver_checkpoint SRCS ${FDBSERVER_CHECKPOINT_SRCS}) +add_fdbserver_link_test(fdbserver_checkpointlinktest + fdbserver_checkpoint + fdbserver_core) + +configure_fdbserver_common_includes(fdbserver_checkpoint) +target_include_directories(fdbserver_checkpoint + PUBLIC + ${CMAKE_CURRENT_SOURCE_DIR}/include + PRIVATE + ${CMAKE_SOURCE_DIR}/fdbserver/include) +target_link_libraries(fdbserver_checkpoint PUBLIC fdbserver_core) + +if(WITH_ROCKSDB) + add_dependencies(fdbserver_checkpoint rocksdb) + if(WITH_LIBURING) + target_include_directories(fdbserver_checkpoint PRIVATE ${ROCKSDB_INCLUDE_DIR} ${uring_INCLUDE_DIR}) + target_link_libraries(fdbserver_checkpoint PRIVATE ${ROCKSDB_LIBRARIES} ${uring_LIBRARIES} ${LZ4_LIBRARY}) + else() + target_include_directories(fdbserver_checkpoint PRIVATE ${ROCKSDB_INCLUDE_DIR}) + target_link_libraries(fdbserver_checkpoint PRIVATE ${ROCKSDB_LIBRARIES} ${LZ4_LIBRARY}) + endif() + target_compile_definitions(fdbserver_checkpoint PRIVATE WITH_ROCKSDB) +endif() diff --git a/fdbserver/checkpoint/Checkpoint.cpp b/fdbserver/checkpoint/Checkpoint.cpp new file mode 100644 index 00000000000..d84ac534f96 --- /dev/null +++ b/fdbserver/checkpoint/Checkpoint.cpp @@ -0,0 +1,99 @@ +/* + * Checkpoint.cpp + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "fdbserver/checkpoint/Checkpoint.h" +#include "fdbserver/checkpoint/RocksDBCheckpointUtils.h" + +ICheckpointReader* newCheckpointReader(const CheckpointMetaData& checkpoint, + const CheckpointAsKeyValues checkpointAsKeyValues, + UID logID) { + const CheckpointFormat format = checkpoint.getFormat(); + if (format == DataMoveRocksCF || format == RocksDB) { + return newRocksDBCheckpointReader(checkpoint, checkpointAsKeyValues, logID); + } else { + throw not_implemented(); + } + + return nullptr; +} + +Future deleteCheckpoint(CheckpointMetaData checkpoint) { + co_await delay(0, TaskPriority::FetchKeys); + const CheckpointFormat format = checkpoint.getFormat(); + if (format == DataMoveRocksCF || format == RocksDB || format == RocksDBKeyValues) { + if (!checkpoint.dir.empty()) { + platform::eraseDirectoryRecursive(checkpoint.dir); + } else { + TraceEvent(SevWarn, "CheckpointDirNotFound").detail("Checkpoint", checkpoint.toString()); + } + } else { + throw not_implemented(); + } +} + +Future fetchCheckpoint(Database cx, + CheckpointMetaData initialState, + std::string dir, + std::function(const CheckpointMetaData&)> cFun) { + TraceEvent("FetchCheckpointBegin", initialState.checkpointID).detail("CheckpointMetaData", initialState.toString()); + + CheckpointMetaData result; + const CheckpointFormat format = initialState.getFormat(); + ASSERT(format != RocksDBKeyValues); + if (format == DataMoveRocksCF || format == RocksDB) { + result = co_await fetchRocksDBCheckpoint(cx, initialState, dir, cFun); + } else { + throw not_implemented(); + } + + TraceEvent("FetchCheckpointEnd", initialState.checkpointID).detail("CheckpointMetaData", result.toString()); + co_return result; +} + +Future fetchCheckpointRanges(Database cx, + CheckpointMetaData initialState, + std::string dir, + std::vector ranges, + std::function(const CheckpointMetaData&)> cFun) { + TraceEvent(SevDebug, "FetchCheckpointRangesBegin", initialState.checkpointID) + .detail("CheckpointMetaData", initialState.toString()) + .detail("Ranges", describe(ranges)); + ASSERT(!ranges.empty()); + + CheckpointMetaData result; + const CheckpointFormat format = initialState.getFormat(); + if (format != RocksDBKeyValues) { + if (format != DataMoveRocksCF) { + throw not_implemented(); + } + initialState.setFormat(RocksDBKeyValues); + initialState.ranges = ranges; + initialState.dir = dir; + initialState.setSerializedCheckpoint( + ObjectWriter::toValue(RocksDBCheckpointKeyValues(ranges), IncludeVersion())); + } + + result = co_await fetchRocksDBCheckpoint(cx, initialState, dir, cFun); + + TraceEvent(SevDebug, "FetchCheckpointRangesEnd", initialState.checkpointID) + .detail("CheckpointMetaData", result.toString()) + .detail("Ranges", describe(ranges)); + co_return result; +} diff --git a/fdbserver/core/RocksDBCheckpointUtils.cpp b/fdbserver/checkpoint/RocksDBCheckpointUtils.cpp similarity index 99% rename from fdbserver/core/RocksDBCheckpointUtils.cpp rename to fdbserver/checkpoint/RocksDBCheckpointUtils.cpp index 118936d5fca..d0b87f2ef47 100644 --- a/fdbserver/core/RocksDBCheckpointUtils.cpp +++ b/fdbserver/checkpoint/RocksDBCheckpointUtils.cpp @@ -1,5 +1,5 @@ /* - *RocksDBCheckpointUtils.cpp + * RocksDBCheckpointUtils.cpp * * This source file is part of the FoundationDB open source project * @@ -18,7 +18,7 @@ * limitations under the License. */ -#include "fdbserver/core/RocksDBCheckpointUtils.h" +#include "fdbserver/checkpoint/RocksDBCheckpointUtils.h" #ifdef WITH_ROCKSDB #include @@ -508,10 +508,11 @@ void RocksDBColumnFamilyReader::Reader::action(RocksDBColumnFamilyReader::Reader return; } - a.done.send(Void()); TraceEvent(SevDebug, "RocksDBCheckpointReaderInitEnd", logId) .detail("Path", path) .detail("ColumnFamily", cf->GetName()); + // Readiness lets another thread close the column family, so finish accessing it before publishing. + a.done.send(Void()); } void RocksDBColumnFamilyReader::Reader::action(RocksDBColumnFamilyReader::Reader::CloseAction& a) { @@ -891,7 +892,7 @@ RangeResult RocksDBSstFileReader::getRange(const KeyRange& range) { class RocksDBCheckpointByteSampleReader : public ICheckpointByteSampleReader { public: - explicit(false) RocksDBCheckpointByteSampleReader(const CheckpointMetaData& checkpoint); + explicit RocksDBCheckpointByteSampleReader(const CheckpointMetaData& checkpoint); ~RocksDBCheckpointByteSampleReader() override = default; KeyValue next() override; diff --git a/fdbserver/checkpoint/include/fdbserver/checkpoint/BulkSstFiles.h b/fdbserver/checkpoint/include/fdbserver/checkpoint/BulkSstFiles.h new file mode 100644 index 00000000000..76806bf0b36 --- /dev/null +++ b/fdbserver/checkpoint/include/fdbserver/checkpoint/BulkSstFiles.h @@ -0,0 +1,40 @@ +/* + * BulkSstFiles.h + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#pragma once + +#include "fdbserver/core/BulkDumpUtil.h" + +// Generate key-value data, byte sampling data, and manifest file. +// Return BulkLoadManifest metadata (equivalent to content of the manifest file). +// TODO(BulkDump): can cause slow tasks, do the task in a separate thread in the future. +// The size of sortedData is defined at the place of generating the data (getRangeDataToDump). +// The size is configured by MOVE_SHARD_KRM_ROW_LIMIT. +Future dumpDataFileToLocalDirectory(UID logId, + std::shared_ptr rangeDumpRawData, + BulkLoadFileSet localFileSet, + BulkLoadFileSet remoteFileSet, + BulkLoadByteSampleSetting byteSampleSetting, + Version dumpVersion, + KeyRange dumpRange, + BulkLoadType dumpType, + BulkLoadTransportMethod transportMethod); + +Future doBytesSamplingOnDataFile(std::string dataFileFullPath, std::string byteSampleFileFullPath, UID logId); diff --git a/fdbserver/checkpoint/include/fdbserver/checkpoint/Checkpoint.h b/fdbserver/checkpoint/include/fdbserver/checkpoint/Checkpoint.h new file mode 100644 index 00000000000..984fec7d359 --- /dev/null +++ b/fdbserver/checkpoint/include/fdbserver/checkpoint/Checkpoint.h @@ -0,0 +1,45 @@ +/* + * Checkpoint.h + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#pragma once + +#include "fdbserver/core/ServerCheckpoint.h" + +ICheckpointReader* newCheckpointReader(const CheckpointMetaData& checkpoint, + const CheckpointAsKeyValues checkpointAsKeyValues, + UID logID); + +// Delete a checkpoint. +Future deleteCheckpoint(CheckpointMetaData checkpoint); + +// Fetches checkpoint to a local `dir`, `initialState` provides the checkpoint formats, location, restart point, etc. +// If cFun is provided, the progress can be checkpointed. +// Returns a CheckpointMetaData, which could contain KVS-specific results, e.g., the list of fetched checkpoint files. +Future fetchCheckpoint(Database cx, + CheckpointMetaData initialState, + std::string dir, + std::function(const CheckpointMetaData&)> cFun = nullptr); + +// Same as above, except that the checkpoint is fetched as key-value pairs. +Future fetchCheckpointRanges(Database cx, + CheckpointMetaData initialState, + std::string dir, + std::vector ranges, + std::function(const CheckpointMetaData&)> cFun = nullptr); diff --git a/fdbserver/core/include/fdbserver/core/RocksDBCheckpointUtils.h b/fdbserver/checkpoint/include/fdbserver/checkpoint/RocksDBCheckpointUtils.h similarity index 99% rename from fdbserver/core/include/fdbserver/core/RocksDBCheckpointUtils.h rename to fdbserver/checkpoint/include/fdbserver/checkpoint/RocksDBCheckpointUtils.h index 9813caea9bf..6d966dff4bf 100644 --- a/fdbserver/core/include/fdbserver/core/RocksDBCheckpointUtils.h +++ b/fdbserver/checkpoint/include/fdbserver/checkpoint/RocksDBCheckpointUtils.h @@ -1,5 +1,5 @@ /* - *RocksDBCheckpointUtils.h + * RocksDBCheckpointUtils.h * * This source file is part of the FoundationDB open source project * @@ -21,7 +21,7 @@ #pragma once #include "fdbclient/NativeAPI.h" -#include "fdbserver/core/ServerCheckpoint.h" +#include "fdbserver/checkpoint/Checkpoint.h" #include "flow/flow.h" class ICheckpointByteSampleReader { diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index ea93c5a6c90..3e848c0a720 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -29,7 +29,7 @@ #include "fdbclient/ClientBooleanParams.h" #include "fdbclient/FDBTypes.h" -#include "NativeCdcInternal.h" +#include "fdbserver/core/NativeCdcMetadata.h" #include "fdbclient/SystemData.h" #include "fdbclient/DatabaseContext.h" #include "fdbrpc/FailureMonitor.h" @@ -45,6 +45,7 @@ #include "fdbserver/core/CoordinatedState.h" #include "fdbserver/core/CoordinationInterface.h" // copy constructors for ServerCoordinators class #include "fdbserver/clustercontroller/ClusterController.h" +#include "fdbserver/clustercontroller/NativeCdcProxyBalancer.h" #include "ClusterController.h" #include "ClusterRecovery.h" #include "fdbserver/core/DataDistributorInterface.h" @@ -1479,6 +1480,14 @@ void clusterRegisterMaster(ClusterControllerData* self, RegisterMasterRequest co } if (req.recoveryState == RecoveryState::FULLY_RECOVERED) { + // Retaining old role advertisements must not interrupt an otherwise completed recovery. + if (!req.logSystemConfig.oldTLogs.empty()) { + TraceEvent(SevError, "FullyRecoveredWithOldTLogs", self->id) + .detail("MasterId", req.id) + .detail("RecoveryCount", req.recoveryCount) + .detail("OldLogGenerations", req.logSystemConfig.oldTLogs.size()); + } + ASSERT_WE_THINK(req.logSystemConfig.oldTLogs.empty()); self->db.unfinishedRecoveries = 0; } @@ -2480,6 +2489,58 @@ Future monitorCDCProxyAssignments(ClusterControllerData* self) { } } +Future rebalanceCDCProxyAssignments(ClusterControllerData* self) { + while (true) { + co_await delay(std::max(1.0, SERVER_KNOBS->CDC_PROXY_REBALANCE_INTERVAL)); + if (!SERVER_KNOBS->CDC_PROXY_REBALANCE_ENABLED) { + TraceEvent("CDCProxyRebalanceDisabled", self->id); + co_return; + } + if (!self->db.recoveryData.isValid() || + self->db.serverInfo->get().recoveryState != RecoveryState::FULLY_RECOVERED || + !self->db.clientInfo->get().nativeCdcEnabled) { + continue; + } + const uint64_t expectedRecoveryCount = self->db.recoveryData->cstate.myDBState.recoveryCount; + const std::vector& published = self->db.clientInfo->get().cdcProxies; + if (published.size() < 2 || published.size() != self->db.cdcProxies.size()) { + continue; + } + std::vector available; + available.reserve(published.size()); + for (const auto& proxy : published) { + if (!containsCDCProxy(self->db.cdcProxies, proxy.id())) { + available.clear(); + break; + } + available.push_back(proxy.id()); + } + if (available.size() < 2) { + continue; + } + try { + const std::vector expectedProxies = available; + auto stillEligible = [self, expectedRecoveryCount, expectedProxies] { + return SERVER_KNOBS->CDC_PROXY_REBALANCE_ENABLED && self->db.recoveryData.isValid() && + self->db.recoveryData->cstate.myDBState.recoveryCount == expectedRecoveryCount && + self->db.serverInfo->get().recoveryState == RecoveryState::FULLY_RECOVERED && + self->db.clientInfo->get().nativeCdcEnabled && + self->db.cdcProxies.size() == expectedProxies.size() && + std::all_of(expectedProxies.begin(), expectedProxies.end(), [self](UID proxyId) { + return containsCDCProxy(self->db.cdcProxies, proxyId); + }); + }; + co_await rebalanceNativeCdcProxyAssignments(self->db.db, std::move(available), std::move(stillEligible)); + } catch (Error& e) { + if (e.code() == error_code_actor_cancelled || e.code() == error_code_broken_promise) { + throw; + } + // An ambiguous commit is reconciled by the assignment monitor; the next scheduled pass may try again. + TraceEvent(SevWarn, "CDCProxyRebalanceError", self->id).error(e); + } + } +} + Future updatedChangingDatacenters(ClusterControllerData* self) { // do not change the cluster controller until all the processes have had a chance to register co_await delay(SERVER_KNOBS->WAIT_FOR_GOOD_RECRUITMENT_DELAY); @@ -2532,6 +2593,13 @@ Future updatedChangingDatacenters(ClusterControllerData* self) { } co_await onChange; + // React on the next event loop turn instead of in the caller's stack. The body above updates + // worker priorities and completes pending worker registrations, whose continuations can draw + // from the deterministic generator; doing that inside whatever called desiredDcIds.set() puts + // those draws inside unrelated work. The recruitment determinism check is one such caller: a + // draw that only happens on its first pass makes the replay pick different (equally fit) + // workers and fail the check. + co_await delay(0); } } @@ -3063,7 +3131,8 @@ Future monitorRatekeeper(ClusterControllerData* self) { const UID monitoredRatekeeperID = self->db.serverInfo->get().ratekeeper.get().id(); auto res = co_await race(waitFailureClient(self->db.serverInfo->get().ratekeeper.get().waitFailure, SERVER_KNOBS->RATEKEEPER_FAILURE_TIME), - self->recruitRatekeeper.onChange()); + self->recruitRatekeeper.onChange(), + self->db.serverInfo->onChange()); if (res.index() == 0) { const auto& ratekeeper = self->db.serverInfo->get().ratekeeper; if (!ratekeeper.present() || ratekeeper.get().id() != monitoredRatekeeperID) { @@ -3515,6 +3584,7 @@ Future clusterControllerCore(ClusterControllerFullInterface interf, self.addActor.send(monitorGlobalConfig(&self.db)); // These actors also drain durable CDC state when new stream registration is disabled. self.addActor.send(monitorCDCProxyAssignments(&self)); + self.addActor.send(rebalanceCDCProxyAssignments(&self)); self.addActor.send(monitorAndRecruitCDCProxies(&self)); self.addActor.send(updatedChangingDatacenters(&self)); self.addActor.send(updatedChangedDatacenters(&self)); @@ -3893,6 +3963,72 @@ TEST_CASE("/fdbserver/clustercontroller/replacedRatekeeperSurvivesPreviousFailur monitor.cancel(); } +TEST_CASE("/fdbserver/clustercontroller/ratekeeperReplacementRefreshesMonitors") { + LocalityData controllerLocality; + controllerLocality.set(LocalityData::keyDcId, "primary"_sr); + ClusterControllerData data(ClusterControllerFullInterface(), + controllerLocality, + ServerCoordinators(Reference( + new ClusterConnectionMemoryRecord(ClusterConnectionString()))), + makeReference>>()); + WorkerInterface oldWorker = addSingletonTestWorker(data, "old-ratekeeper"_sr, "primary"_sr); + WorkerInterface newWorker = addSingletonTestWorker(data, "new-ratekeeper"_sr, "primary"_sr); + RatekeeperInterface oldRatekeeper(oldWorker.locality, UID(1, 1)); + RatekeeperInterface newRatekeeper(newWorker.locality, UID(1, 2)); + FutureStream> oldFailures = oldRatekeeper.waitFailure.getFuture(); + FutureStream> newFailures = newRatekeeper.waitFailure.getFuture(); + FutureStream oldHalts = oldRatekeeper.haltRatekeeper.getFuture(); + FutureStream newHalts = newRatekeeper.haltRatekeeper.getFuture(); + FutureStream oldEvents = oldWorker.eventLogRequest.getFuture(); + FutureStream newEvents = newWorker.eventLogRequest.getFuture(); + + auto serverInfo = data.db.serverInfo->get(); + serverInfo.recoveryState = RecoveryState::ACCEPTING_COMMITS; + serverInfo.id = UID(3, 1); + data.db.serverInfo->set(serverInfo); + data.db.setRatekeeper(oldRatekeeper); + Future monitor = monitorRatekeeper(&data); + auto oldWaitOrTimeout = co_await race(oldFailures, delay(2.0)); + ASSERT_EQ(oldWaitOrTimeout.index(), 0); + ReplyPromise oldFailure = std::get<0>(std::move(oldWaitOrTimeout)); + + processRegisteredSingletons(&data, newWorker, {}, newRatekeeper, {}); + ASSERT(data.db.serverInfo->get().ratekeeper.get().id() == newRatekeeper.id()); + auto oldHaltOrTimeout = co_await race(oldHalts, delay(2.0)); + ASSERT_EQ(oldHaltOrTimeout.index(), 0); + HaltRatekeeperRequest oldHalt = std::get<0>(std::move(oldHaltOrTimeout)); + auto newWaitOrTimeout = co_await race(newFailures, delay(2.0)); + ASSERT_EQ(newWaitOrTimeout.index(), 0); + ReplyPromise newFailure = std::get<0>(std::move(newWaitOrTimeout)); + + auto latestEvents = data.clusterHealthWorkerEventProvider->getLatestRatekeeperEvents("RkUpdate"); + auto eventOrTimeout = co_await race(oldEvents, newEvents, delay(2.0)); + ASSERT_EQ(eventOrTimeout.index(), 1); + EventLogRequest request = std::get<1>(std::move(eventOrTimeout)); + ASSERT(request.eventName == "RkUpdate"_sr); + TraceEventFields fields; + fields.addField("ReleasedTPS", "100"); + fields.addField("TPSLimit", "125"); + request.reply.send(fields); + auto events = co_await latestEvents; + ASSERT(events.present()); + ASSERT_EQ(events.get().first.size(), 1); + ASSERT(events.get().second.empty()); + ASSERT_EQ(events.get().first.begin()->second.getDouble("TPSLimit"), 125.0); + + newFailure.sendError(connection_failed()); + auto newHaltOrTimeout = co_await race(newHalts, delay(2.0)); + ASSERT_EQ(newHaltOrTimeout.index(), 0); + HaltRatekeeperRequest newHalt = std::get<0>(std::move(newHaltOrTimeout)); + newHalt.reply.send(Void()); + ASSERT(!data.db.serverInfo->get().ratekeeper.present()); + auto clearedEvents = co_await data.clusterHealthWorkerEventProvider->getLatestRatekeeperEvents("RkUpdate"); + ASSERT(!clearedEvents.present()); + oldHalt.reply.send(Void()); + oldFailure.sendError(connection_failed()); + monitor.cancel(); +} + TEST_CASE("/fdbserver/clustercontroller/deferCrossDatacenterSingletonHaltsUntilRecovery") { LocalityData controllerLocality; controllerLocality.set(LocalityData::keyDcId, "new-primary"_sr); @@ -4229,7 +4365,7 @@ TEST_CASE("/fdbserver/clustercontroller/deferBetterMasterRecoveryUntilInitialSto return Void(); } -TEST_CASE("/fdbserver/clustercontroller/recoverForExcludedOldTLogLocality") { +TEST_CASE("/fdbserver/clustercontroller/ignoreExcludedOldTLogLocality") { ClusterControllerData data(ClusterControllerFullInterface(), LocalityData(), ServerCoordinators(Reference( @@ -4261,12 +4397,24 @@ TEST_CASE("/fdbserver/clustercontroller/recoverForExcludedOldTLogLocality") { oldTLogSet.tLogs.push_back(OptionalInterface(oldTLog)); OldTLogConf oldTLogConf; oldTLogConf.tLogs.push_back(oldTLogSet); + LocalityData currentLocality; + currentLocality.set(LocalityData::keyProcessId, Standalone(std::string{ "current-tlog" })); + TLogInterface currentTLog(currentLocality); + TLogSet currentTLogSet; + currentTLogSet.tLogs.push_back(OptionalInterface(currentTLog)); ServerDBInfo dbInfo; dbInfo.master.locality = masterLocality; + dbInfo.logSystemConfig.tLogs.push_back(currentTLogSet); dbInfo.logSystemConfig.oldTLogs.push_back(oldTLogConf); dbInfo.recoveryState = RecoveryState::FULLY_RECOVERED; data.db.serverInfo->set(dbInfo); + // An unregistered current log stops unrelated placement comparisons after checking the old roles. + ASSERT(!data.betterMasterExists()); + + auto& currentWorker = data.id_worker[currentLocality.processId()]; + currentWorker.details.interf = WorkerInterface(currentLocality); + currentWorker.priorityInfo.isExcluded = true; ASSERT(data.betterMasterExists()); return Void(); } @@ -4644,8 +4792,7 @@ TEST_CASE("/fdbserver/clustercontroller/getDegradationInfo") { data.workerHealth[badPeer4].disconnectedPeers[worker] = { now() - SERVER_KNOBS->CC_MIN_DEGRADATION_INTERVAL - 1, now() }; ASSERT(data.getDegradationInfo().disconnectedServers.size() == 1); - ASSERT(data.getDegradationInfo().disconnectedServers.find(worker) != - data.getDegradationInfo().disconnectedServers.end()); + ASSERT(data.getDegradationInfo().disconnectedServers.contains(worker)); data.workerHealth.clear(); } @@ -5067,4 +5214,275 @@ TEST_CASE("/fdbserver/clustercontroller/invalidateExcludedProcessComplaints") { return Void(); } +// Adds `count` verified workers of the given process class to `data.id_worker`, each in its +// own zone within datacenter `dcId`, and returns their interfaces. `classType` is honored so +// that per-role recruitment fitness differs across the workers. +static std::vector addRecruitmentTestWorkers(ClusterControllerData& data, + Key const& dcId, + ProcessClass::ClassType classType, + StringRef prefix, + int count) { + std::vector workers; + for (int i = 0; i < count; i++) { + std::string pid = prefix.toString() + std::to_string(i); + LocalityData locality; + locality.set(LocalityData::keyProcessId, Standalone(pid)); + locality.set(LocalityData::keyZoneId, Standalone(pid + "_zone")); + locality.set(LocalityData::keyDcId, dcId); + WorkerInterface worker(locality); + worker.initEndpoints(); + auto& info = data.id_worker[locality.processId()]; + info.verified = true; + info.details.interf = worker; + info.details.processClass = ProcessClass(classType, ProcessClass::CommandLineSource); + info.details.recoveredDiskFiles = true; + workers.push_back(worker); + } + return workers; +} + +static ClusterControllerData makeRecruitmentTestData(Key const& dcId) { + LocalityData locality; + locality.set(LocalityData::keyDcId, dcId); + return ClusterControllerData(ClusterControllerFullInterface(), + locality, + ServerCoordinators(Reference( + new ClusterConnectionMemoryRecord(ClusterConnectionString()))), + makeReference>>()); +} + +// Regression test for the original bug: in a small cluster where the desired commit-proxy +// count equals the number of stateless processes, only one proxy was recruited. The candidate +// filter compared each candidate's usage against the first-selected worker's usage and so +// rejected every equally-fit process that already hosted the master or cluster controller. +// All equally-fit processes must now remain eligible. +TEST_CASE("/fdbserver/clustercontroller/proxyRecruitmentFillsEqualFitnessProcesses") { + const Key dcId = "dc1"_sr; + ClusterControllerData data = makeRecruitmentTestData(dcId); + + constexpr int kCount = 3; + auto stateless = addRecruitmentTestWorkers(data, dcId, ProcessClass::StatelessClass, "sl"_sr, kCount); + + // Two of the three stateless processes already host the master and cluster controller, + // so they start with higher usage than the third. + data.masterProcessId = stateless[0].locality.processId(); + data.clusterControllerProcessId = stateless[1].locality.processId(); + + DatabaseConfiguration config; + config.initialized = true; + + std::map>, int> id_used; + data.updateKnownIds(&id_used); + + auto first = + data.getWorkerForRoleInDatacenter(dcId, recruitment::CommitProxy, recruitment::ExcludeFit, config, id_used); + auto proxies = + data.getWorkersForRoleInDatacenter(dcId, recruitment::CommitProxy, kCount, config, id_used, {}, first); + + ASSERT_EQ(proxies.size(), kCount); + std::set>> pids; + for (const auto& w : proxies) { + pids.insert(w.interf.locality.processId()); + } + ASSERT_EQ(pids.size(), kCount); // every proxy on a distinct process + return Void(); +} + +// Recruitment must still span fitness levels: dedicated commit-proxy processes (BestFit) are +// preferred, but stateless processes (GoodFit) fill the remainder so the desired count is +// reached instead of stopping at the best-fit class. +TEST_CASE("/fdbserver/clustercontroller/proxyRecruitmentSpansFitnessLevels") { + const Key dcId = "dc1"_sr; + ClusterControllerData data = makeRecruitmentTestData(dcId); + + auto dedicated = addRecruitmentTestWorkers(data, dcId, ProcessClass::CommitProxyClass, "cp"_sr, 2); + auto stateless = addRecruitmentTestWorkers(data, dcId, ProcessClass::StatelessClass, "sl"_sr, 3); + + DatabaseConfiguration config; + config.initialized = true; + + std::map>, int> id_used; + auto first = + data.getWorkerForRoleInDatacenter(dcId, recruitment::CommitProxy, recruitment::ExcludeFit, config, id_used); + auto proxies = data.getWorkersForRoleInDatacenter(dcId, recruitment::CommitProxy, 3, config, id_used, {}, first); + + ASSERT_EQ(proxies.size(), 3); + std::set>> dedicatedPids, statelessPids; + for (const auto& w : dedicated) { + dedicatedPids.insert(w.locality.processId()); + } + for (const auto& w : stateless) { + statelessPids.insert(w.locality.processId()); + } + int dedicatedCount = 0, statelessCount = 0; + std::set>> usedPids; + for (const auto& w : proxies) { + auto pid = w.interf.locality.processId(); + usedPids.insert(pid); + if (dedicatedPids.count(pid) == 1) { + dedicatedCount++; + } else if (statelessPids.count(pid) == 1) { + statelessCount++; + } + } + ASSERT_EQ(usedPids.size(), 3); // all on distinct processes + ASSERT_EQ(dedicatedCount, 2); // both dedicated processes used first + ASSERT_EQ(statelessCount, 1); // remainder filled from stateless + return Void(); +} + +// Regression test for NonDeterministicRecruitment: findWorkersForConfiguration() recruits the +// configuration twice in simulation and requires both recruitments to produce equal RoleFitness +// (which includes the worst usage of the recruited workers). The determinism check replays the +// random sequence of the first pass, so equal fitness here additionally means the candidate +// filter did not starve the pool: with the master and cluster controller occupying two of the +// three stateless processes, every proxy must still land on a distinct process, and both passes +// must agree on the worst usage that results. +TEST_CASE("/fdbserver/clustercontroller/proxyRecruitmentDeterministicUsage") { + const Key dcId = "dc1"_sr; + ClusterControllerData data = makeRecruitmentTestData(dcId); + + constexpr int kCount = 3; + auto stateless = addRecruitmentTestWorkers(data, dcId, ProcessClass::StatelessClass, "sl"_sr, kCount); + + // Two of the three stateless processes already host the master and cluster controller. + data.masterProcessId = stateless[0].locality.processId(); + data.clusterControllerProcessId = stateless[1].locality.processId(); + + DatabaseConfiguration config; + config.initialized = true; + + ClusterControllerData::RoleFitness firstFit; + ClusterControllerData::RoleFitness secondFit; + for (int pass = 0; pass < 2; pass++) { + // Each pass re-recruits from scratch with the same initial id_used, mimicking the + // two-pass determinism check in findWorkersForConfiguration. + std::map>, int> id_used; + data.updateKnownIds(&id_used); + + auto first = + data.getWorkerForRoleInDatacenter(dcId, recruitment::GrvProxy, recruitment::ExcludeFit, config, id_used); + auto proxies = + data.getWorkersForRoleInDatacenter(dcId, recruitment::GrvProxy, kCount, config, id_used, {}, first); + + // The pool is not artificially cut: every proxy lands on a distinct stateless process. + ASSERT_EQ(proxies.size(), kCount); + std::set>> pids; + for (const auto& w : proxies) { + pids.insert(w.interf.locality.processId()); + } + ASSERT_EQ(pids.size(), kCount); + + // Mimic findWorkersForConfiguration's comparison accounting: usage of the recruited + // workers is added on top of the initial id_used before computing the fitness. + std::map>, int> compareUsed; + data.updateKnownIds(&compareUsed); + for (const auto& w : proxies) { + compareUsed[w.interf.locality.processId()]++; + } + ClusterControllerData::RoleFitness fit(proxies, recruitment::GrvProxy, compareUsed); + if (pass == 0) { + firstFit = fit; + } else { + secondFit = fit; + } + } + + ASSERT(firstFit == secondFit); + return Void(); +} + +// Without a minWorker there is no fitness ceiling, so recruitment fills across fitness levels +// from the best available. This locks in that the fix (which gates on the minWorker's fitness) +// does not change the no-minWorker path used by log-router recruitment. +TEST_CASE("/fdbserver/clustercontroller/logRouterRecruitmentWithoutMinWorker") { + const Key dcId = "dc1"_sr; + ClusterControllerData data = makeRecruitmentTestData(dcId); + + auto stateless = addRecruitmentTestWorkers(data, dcId, ProcessClass::StatelessClass, "sl"_sr, 2); + auto transaction = addRecruitmentTestWorkers(data, dcId, ProcessClass::TransactionClass, "tx"_sr, 2); + + DatabaseConfiguration config; + config.initialized = true; + + std::map>, int> id_used; + auto routers = data.getWorkersForRoleInDatacenter(dcId, recruitment::LogRouter, 4, config, id_used); + + ASSERT_EQ(routers.size(), 4); + std::set>> pids; + for (const auto& w : routers) { + pids.insert(w.interf.locality.processId()); + } + ASSERT_EQ(pids.size(), 4); // all distinct, spanning both fitness levels + return Void(); +} + +// End-to-end regression test for the original bug through the production recruitment path +// (findWorkersForConfigurationDispatch). In a small cluster whose desired proxy counts equal +// the number of stateless processes, every commit proxy, GRV proxy and resolver must be +// recruited (previously only one commit proxy was), each on a distinct stateless process. +// Process classes are honored: TLogs land on transaction processes, proxies on stateless ones. +TEST_CASE("/fdbserver/clustercontroller/proxyColocationOnStateless") { + const Key dcId = "dc1"_sr; + ClusterControllerData data = makeRecruitmentTestData(dcId); + // Make the good-recruitment gate pass so we assert on the recruitment result itself. + data.goodRecruitmentTime = Void(); + + constexpr int kCount = 3; + auto stateless = addRecruitmentTestWorkers(data, dcId, ProcessClass::StatelessClass, "sl"_sr, kCount); + auto transaction = addRecruitmentTestWorkers(data, dcId, ProcessClass::TransactionClass, "tx"_sr, kCount); + auto storage = addRecruitmentTestWorkers(data, dcId, ProcessClass::StorageClass, "ss"_sr, kCount); + + // Two of the stateless processes already host the master and cluster controller. + data.masterProcessId = stateless[0].locality.processId(); + data.clusterControllerProcessId = stateless[1].locality.processId(); + + DatabaseConfiguration config; + config.initialized = true; + config.usableRegions = 1; + RegionInfo region; + region.dcId = dcId; + config.regions.push_back(region); + config.tLogReplicationFactor = kCount; + config.desiredTLogCount = kCount; + config.commitProxyCount = kCount; + config.grvProxyCount = kCount; + config.resolverCount = kCount; + config.tLogPolicy = makeReference(); + data.db.config = config; + data.db.fullyRecoveredConfig = config; + + RecruitFromConfigurationRequest req(config, /*recruitSeedServers=*/false, /*maxOldLogRouters=*/0); + auto result = data.findWorkersForConfigurationDispatch(req, /*checkGoodRecruitment=*/true); + + std::set>> statelessPids, transactionPids; + for (const auto& w : stateless) { + statelessPids.insert(w.locality.processId()); + } + for (const auto& w : transaction) { + transactionPids.insert(w.locality.processId()); + } + + // TLogs are recruited on the transaction processes. + ASSERT_EQ(result.tLogs.size(), kCount); + for (const auto& interf : result.tLogs) { + ASSERT(transactionPids.count(interf.locality.processId()) == 1); + } + + // Each proxy/resolver set is fully recruited onto distinct stateless processes. + auto assertDistinctStateless = [&statelessPids, kCount](const std::vector& interfaces) { + ASSERT_EQ(interfaces.size(), kCount); + std::set>> pids; + for (const auto& interf : interfaces) { + pids.insert(interf.locality.processId()); + ASSERT(statelessPids.count(interf.locality.processId()) == 1); + } + ASSERT_EQ(pids.size(), kCount); + }; + assertDistinctStateless(result.commitProxies); + assertDistinctStateless(result.grvProxies); + assertDistinctStateless(result.resolvers); + return Void(); +} + } // namespace diff --git a/fdbserver/clustercontroller/ClusterController.h b/fdbserver/clustercontroller/ClusterController.h index a1950b7166d..db3d95f4524 100644 --- a/fdbserver/clustercontroller/ClusterController.h +++ b/fdbserver/clustercontroller/ClusterController.h @@ -22,6 +22,7 @@ #include #include +#include #include #include "fdbclient/DatabaseContext.h" @@ -1283,9 +1284,15 @@ class ClusterControllerData { exclusionWorkerIds); if (g_network->isSimulated()) { + // The comparison below validates the TLog method, not recruitment, so its draws must + // not leak into the shared stream: the replay pass of the determinism check would + // otherwise continue from a different stream and report a divergence the + // recruitment did not cause. + uint64_t entryFingerprint = deterministicRandom()->peek(); try { auto testWorkers = getWorkersForTlogsBackup( conf, required, desired, policy, testUsed, checkStable, dcIds, exclusionWorkerIds); + deterministicRandom()->resetSeed(entryFingerprint); RoleFitness testFitness(testWorkers, recruitment::TLog, testUsed); RoleFitness fitness(workers, recruitment::TLog, id_used); @@ -1320,6 +1327,7 @@ class ClusterControllerData { ASSERT(false); } } catch (Error& e) { + deterministicRandom()->resetSeed(entryFingerprint); ASSERT(false); // Simulation only validation should not throw errors } } @@ -1340,9 +1348,14 @@ class ClusterControllerData { getWorkersForTlogsSimple(conf, required, desired, id_used, checkStable, dcIds, exclusionWorkerIds); if (g_network->isSimulated()) { + // The comparison below validates the TLog method, not recruitment, so its draws must not + // leak into the shared stream: the replay pass of the determinism check would otherwise + // continue from a different stream and report a divergence the recruitment did not cause. + uint64_t entryFingerprint = deterministicRandom()->peek(); try { auto testWorkers = getWorkersForTlogsBackup( conf, required, desired, policy, testUsed, checkStable, dcIds, exclusionWorkerIds); + deterministicRandom()->resetSeed(entryFingerprint); RoleFitness testFitness(testWorkers, recruitment::TLog, testUsed); RoleFitness fitness(workers, recruitment::TLog, id_used); // backup recruitment is not required to use degraded processes that have better fitness @@ -1362,6 +1375,7 @@ class ClusterControllerData { ASSERT(false); } } catch (Error& e) { + deterministicRandom()->resetSeed(entryFingerprint); ASSERT(false); // Simulation only validation should not throw errors } } @@ -1546,14 +1560,18 @@ class ClusterControllerData { for (auto& it : id_worker) { auto fitness = recruitment::machineClassFitness(it.second.details.processClass, role); + // Candidates must not be worse than the already-accepted minWorker. Usage is + // deliberately not part of this check: gating on it empties the pool whenever the + // desired count exceeds the number of least-used equal-fitness processes (e.g. + // every stateless process, when some of them already host the master or cluster + // controller). Spreading across processes is instead provided by the bucket + // ordering on `used` in the fill loop below. if (workerAvailable(it.second, checkStable) && !conf.isExcludedServer(it.second.details.interf.addresses(), it.second.details.interf.locality) && !isExcludedDegradedServer(it.second.details.interf.addresses()) && it.second.details.interf.locality.dcId() == dcId && - (!minWorker.present() || - (it.second.details.interf.id() != minWorker.get().worker.interf.id() && - (fitness < minWorker.get().fitness || - (fitness == minWorker.get().fitness && id_used[it.first] <= minWorker.get().used))))) { + (!minWorker.present() || (it.second.details.interf.id() != minWorker.get().worker.interf.id() && + fitness <= minWorker.get().fitness))) { auto sharing = preferredSharing.find(it.first); fitness_workers[{ fitness, id_used[it.first], @@ -2234,38 +2252,79 @@ class ClusterControllerData { recruitment::ClusterRole role, std::string description) { std::vector firstDetails; + std::set>> firstPids; for (auto& worker : first) { auto w = id_worker.find(worker.locality.processId()); ASSERT(w != id_worker.end()); auto const& [_, workerInfo] = *w; ASSERT(!conf.isExcludedServer(workerInfo.details.interf.addresses(), workerInfo.details.interf.locality)); firstDetails.push_back(workerInfo.details); + firstPids.insert(worker.locality.processId()); //TraceEvent("CompareAddressesFirst").detail(description.c_str(), w->second.details.interf.address()); } RoleFitness firstFitness(firstDetails, role, firstUsed); std::vector secondDetails; + std::set>> secondPids; for (auto& worker : second) { auto w = id_worker.find(worker.locality.processId()); ASSERT(w != id_worker.end()); auto const& [_, workerInfo] = *w; ASSERT(!conf.isExcludedServer(workerInfo.details.interf.addresses(), workerInfo.details.interf.locality)); secondDetails.push_back(workerInfo.details); + secondPids.insert(worker.locality.processId()); //TraceEvent("CompareAddressesSecond").detail(description.c_str(), w->second.details.interf.address()); } RoleFitness secondFitness(secondDetails, role, secondUsed); - if (!(firstFitness == secondFitness)) { + // Compare the recruited process sets as well as the fitness summary: the summary alone can + // coincide for different sets, which would hide the divergence here and instead surface + // against a later role whose fitness is computed from the usage this role left behind. + bool sameWorkerSet = firstPids == secondPids; + if (!sameWorkerSet || !(firstFitness == secondFitness)) { + auto describe = [&](const std::vector& details, + const std::map>, int>& used) { + std::string s; + // Cap the dump so the trace event stays well under the size limit. + const int n = std::min(details.size(), 8); + for (int i = 0; i < n; i++) { + auto pid = details[i].interf.locality.processId(); + auto u = used.find(pid); + s += "(" + (pid.present() ? pid.get().toString() : std::string("[not set]")) + ",fit=" + + std::to_string((int)recruitment::machineClassFitness(details[i].processClass, role)) + + ",used=" + (u != used.end() ? std::to_string(u->second) : std::string("?")) + ") "; + } + if ((int)details.size() > n) { + s += "...(" + std::to_string(details.size()) + " total)"; + } + return s; + }; TraceEvent(SevError, "NonDeterministicRecruitment") + .detail("Kind", sameWorkerSet ? "Fitness" : "WorkerSet") + .detail("Description", description) .detail("FirstFitness", firstFitness.toString()) .detail("SecondFitness", secondFitness.toString()) - .detail("ClusterRole", role); + .detail("ClusterRole", role) + .detail("FirstWorkers", describe(firstDetails, firstUsed)) + .detail("SecondWorkers", describe(secondDetails, secondUsed)); } } RecruitFromConfigurationReply findWorkersForConfiguration(RecruitFromConfigurationRequest const& req) { + // The determinism check below re-runs recruitment and compares the result against the first + // pass. Recruitment deliberately randomizes (randomShuffle/randomChoice among equal candidates), + // so both passes must start from the same RNG state or they would trivially disagree. Seed the + // generator before the first pass, then reseed it from the same value before the replay, so both + // passes draw an identical random sequence. This is simulation-only; production runs are unaffected + // because the generator there is not seeded deterministically. + uint64_t seed = 0; + if (g_network->isSimulated()) { + seed = deterministicRandom()->randomUInt64(); + deterministicRandom()->resetSeed(seed); + } RecruitFromConfigurationReply rep = findWorkersForConfigurationDispatch(req, true); if (g_network->isSimulated()) { + deterministicRandom()->resetSeed(seed); try { // FIXME: The logic to pick a satellite in a remote region is not // deterministic and can therefore break this nondeterminism check. @@ -2302,12 +2361,12 @@ class ClusterControllerData { secondUsed, recruitment::TLog, "Satellite"); + // Each role is compared against the usage map as it stood when that role was + // recruited: roles recruited later (grv proxies, resolvers) must not leak their + // usage into the comparison of an earlier role, which would compare it against a + // placement it never produced and blame it for a later divergence. updateIdUsed(rep.commitProxies, firstUsed); updateIdUsed(compare.commitProxies, secondUsed); - updateIdUsed(rep.grvProxies, firstUsed); - updateIdUsed(compare.grvProxies, secondUsed); - updateIdUsed(rep.resolvers, firstUsed); - updateIdUsed(compare.resolvers, secondUsed); compareWorkers(req.configuration, rep.commitProxies, firstUsed, @@ -2315,6 +2374,8 @@ class ClusterControllerData { secondUsed, recruitment::CommitProxy, "CommitProxy"); + updateIdUsed(rep.grvProxies, firstUsed); + updateIdUsed(compare.grvProxies, secondUsed); compareWorkers(req.configuration, rep.grvProxies, firstUsed, @@ -2322,6 +2383,8 @@ class ClusterControllerData { secondUsed, recruitment::GrvProxy, "GrvProxy"); + updateIdUsed(rep.resolvers, firstUsed); + updateIdUsed(compare.resolvers, secondUsed); compareWorkers(req.configuration, rep.resolvers, firstUsed, @@ -2458,31 +2521,8 @@ class ClusterControllerData { std::vector backup_workers; std::set backup_addresses; - if (dbi.recoveryState == RecoveryState::FULLY_RECOVERED) { - for (const auto& oldLog : dbi.logSystemConfig.oldTLogs) { - for (const auto& logSet : oldLog.tLogs) { - for (const auto& tlog : logSet.tLogs) { - if (!tlog.present()) { - continue; - } - - auto tlogWorker = std::find_if(id_worker.begin(), id_worker.end(), [&tlog](const auto& worker) { - return worker.second.details.interf.address() == tlog.interf().address(); - }); - const auto& locality = tlogWorker == id_worker.end() - ? tlog.interf().filteredLocality - : tlogWorker->second.details.interf.locality; - if (db.config.isExcludedServer(tlog.interf().addresses(), locality)) { - TraceEvent("BetterMasterExists", id) - .detail("Reason", "OldTLogExcluded") - .detail("ProcessID", locality.processId()); - return true; - } - } - } - } - } - + // Old TLog roles retire after terminal recovery; their exclusion does not require another recovery. + // Excluded current TLogs still need a replacement transaction system. for (auto& logSet : dbi.logSystemConfig.tLogs) { for (auto& it : logSet.tLogs) { auto tlogWorker = id_worker.find(it.interf().filteredLocality.processId()); diff --git a/fdbserver/clustercontroller/ClusterHealthIFactor.cpp b/fdbserver/clustercontroller/ClusterHealthIFactor.cpp index de593105438..8609e472b73 100644 --- a/fdbserver/clustercontroller/ClusterHealthIFactor.cpp +++ b/fdbserver/clustercontroller/ClusterHealthIFactor.cpp @@ -21,6 +21,7 @@ #include #include +#include "fdbserver/core/Knobs.h" #include "fdbserver/core/RecoveryState.h" #include "flow/Trace.h" @@ -134,6 +135,15 @@ std::string_view StorageReplicationFactor::getName() const { Future StorageReplicationFactor::fetchLevel(Reference workerEventProvider, TrackCodeProbes trackCodeProbes) { + auto teamEventsAndErrors = co_await workerEventProvider->getLatestDataDistributorEvents("TotalDataInFlight"); + if (!teamEventsAndErrors.present()) { + co_return Level::METRICS_MISSING; + } + WorkerEvents teamEvents = filterEmptyEvents(teamEventsAndErrors.get().first); + if (teamEvents.empty()) { + co_return Level::METRICS_MISSING; + } + auto eventsAndErrors = co_await workerEventProvider->getLatestDataDistributorEvents("MovingData"); if (!eventsAndErrors.present()) { co_return Level::METRICS_MISSING; @@ -146,9 +156,18 @@ Future StorageReplicationFactor::fetchLevel(Reference StorageReplicationFactor::fetchLevel(Reference 0 || inFlight > 0) { queuedOrInFlightRepairMoves += priorityTeamUnhealthy + priorityTeam2Left + priorityTeam1Left + priorityTeam0Left; } } - if (zeroReplicaTeams > 0) { + if (highestTeamPriority >= SERVER_KNOBS->PRIORITY_TEAM_0_LEFT) { CODE_PROBE(trackCodeProbes, "ClusterHealth StorageReplicationFactor returns OUTAGE"); co_return Level::OUTAGE; } - if (oneReplicaTeams > 0 && workerEventProvider->shouldTreatStorageTeamOneReplicaLeftAsCritical()) { + if (highestTeamPriority >= SERVER_KNOBS->PRIORITY_TEAM_1_LEFT && + workerEventProvider->shouldTreatStorageTeamOneReplicaLeftAsCritical()) { CODE_PROBE(trackCodeProbes, "ClusterHealth StorageReplicationFactor returns CRITICAL_INTERVENTION_REQUIRED"); co_return Level::CRITICAL_INTERVENTION_REQUIRED; } - if (queuedOrInFlightRepairMoves > 0) { + if (highestTeamPriority >= SERVER_KNOBS->PRIORITY_TEAM_UNHEALTHY || queuedOrInFlightRepairMoves > 0) { CODE_PROBE(trackCodeProbes, "ClusterHealth StorageReplicationFactor returns SELF_HEALING"); co_return Level::SELF_HEALING; } diff --git a/fdbserver/clustercontroller/ClusterHealthMonitorTesting.cpp b/fdbserver/clustercontroller/ClusterHealthMonitorTesting.cpp index 296de920996..2d53233dcae 100644 --- a/fdbserver/clustercontroller/ClusterHealthMonitorTesting.cpp +++ b/fdbserver/clustercontroller/ClusterHealthMonitorTesting.cpp @@ -20,6 +20,7 @@ #include +#include "fdbserver/core/Knobs.h" #include "fdbserver/core/RecoveryState.h" #include "flow/UnitTest.h" @@ -270,43 +271,84 @@ TEST_CASE("/fdbserver/clustercontroller/ClusterHealthMonitor/TLogSpaceFactor") { TEST_CASE("/fdbserver/clustercontroller/ClusterHealthMonitor/StorageReplicationFactor") { StorageReplicationFactor factor; auto provider = makeReference(); - Level level; - + auto teamMetrics = [](int priority) { + TraceEventFields fields; + fields.addField("HighestTeamPriority", std::to_string(priority)); + return fields; + }; + provider->setLatestEvents("TotalDataInFlight", + makeLatestWorkerEvents(teamMetrics(SERVER_KNOBS->PRIORITY_TEAM_HEALTHY))); provider->setLatestEvents("MovingData", makeLatestWorkerEvents(MovingDataMetricsBuilder().build())); - level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); + Level level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); ASSERT_EQ(level, Level::HEALTHY); + // Repair activity remains SelfHealing, even after the teams themselves recover. provider->setLatestEvents( "MovingData", makeLatestWorkerEvents(MovingDataMetricsBuilder().inQueue(1).priorityTeamUnhealthy(1).build())); level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); ASSERT_EQ(level, Level::SELF_HEALING); - provider->setLatestEvents( - "MovingData", makeLatestWorkerEvents(MovingDataMetricsBuilder().inFlight(1).priorityTeam1Left(1).build())); + provider->setLatestEvents("TotalDataInFlight", + makeLatestWorkerEvents(teamMetrics(SERVER_KNOBS->PRIORITY_TEAM_1_LEFT))); + provider->setLatestEvents("MovingData", makeLatestWorkerEvents(MovingDataMetricsBuilder().build())); level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); ASSERT_EQ(level, Level::SELF_HEALING); - provider->setStorageTeamOneReplicaLeftIsCritical(true); - level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); - ASSERT_EQ(level, Level::CRITICAL_INTERVENTION_REQUIRED); + // Admission and completion can change the queue's priority counts without changing team health. + // In particular, a full relocation pipeline can temporarily contain no zero-replica relocations. + for (int priority : { SERVER_KNOBS->PRIORITY_TEAM_0_LEFT, SERVER_KNOBS->PRIORITY_TEAM_1_LEFT }) { + provider->setLatestEvents("TotalDataInFlight", makeLatestWorkerEvents(teamMetrics(priority))); + for (int zeroReplicaMoves : { 2, 0, 1 }) { + provider->setLatestEvents("MovingData", + makeLatestWorkerEvents(MovingDataMetricsBuilder() + .inQueue(900) + .inFlight(100) + .priorityTeam1Left(700) + .priorityTeam0Left(zeroReplicaMoves) + .build())); + level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); + ASSERT_EQ(level, + priority == SERVER_KNOBS->PRIORITY_TEAM_0_LEFT ? Level::OUTAGE + : Level::CRITICAL_INTERVENTION_REQUIRED); + } + provider->setLatestEvents("MovingData", makeLatestWorkerEvents(MovingDataMetricsBuilder().build())); + level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); + ASSERT_EQ(level, + priority == SERVER_KNOBS->PRIORITY_TEAM_0_LEFT ? Level::OUTAGE + : Level::CRITICAL_INTERVENTION_REQUIRED); + } + + // Old severe relocations must not keep a recovered team at Outage or Critical. + provider->setLatestEvents("TotalDataInFlight", + makeLatestWorkerEvents(teamMetrics(SERVER_KNOBS->PRIORITY_TEAM_HEALTHY))); provider->setLatestEvents( "MovingData", makeLatestWorkerEvents(MovingDataMetricsBuilder().inQueue(1).priorityTeam0Left(1).build())); level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); - ASSERT_EQ(level, Level::OUTAGE); + ASSERT_EQ(level, Level::SELF_HEALING); - WorkerEvents staleWorkerEvents; - staleWorkerEvents.emplace(NetworkAddress(IPAddress(0x01010101), 1), - MovingDataMetricsBuilder().inQueue(1).priorityTeam0Left(1).build()); - staleWorkerEvents.emplace(NetworkAddress(IPAddress(0x02020202), 2), MovingDataMetricsBuilder().build()); - provider->setLatestEvents("MovingData", makeLatestWorkerEvents(std::move(staleWorkerEvents))); - provider->setLatestDataDistributorEvents( - "MovingData", - makeLatestWorkerEvents(NetworkAddress(IPAddress(0x02020202), 2), MovingDataMetricsBuilder().build())); + // Both summaries must come from the current DD, not a former DD worker's cached events. + provider->setLatestEvents("TotalDataInFlight", + makeLatestWorkerEvents(teamMetrics(SERVER_KNOBS->PRIORITY_TEAM_0_LEFT))); + provider->setLatestDataDistributorEvents("TotalDataInFlight", + makeLatestWorkerEvents(teamMetrics(SERVER_KNOBS->PRIORITY_TEAM_HEALTHY))); + provider->setLatestDataDistributorEvents("MovingData", makeLatestWorkerEvents(MovingDataMetricsBuilder().build())); level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); ASSERT_EQ(level, Level::HEALTHY); - provider->setLatestEvents("MovingData", LatestWorkerEvents()); + // Uninitialized, old-format, empty, and unavailable team summaries must not fabricate health. + TraceEventFields oldFormat; + oldFormat.addField("HighestPriority", "0"); + for (const auto& fields : { teamMetrics(-1), oldFormat, TraceEventFields() }) { + provider->setLatestDataDistributorEvents("TotalDataInFlight", makeLatestWorkerEvents(fields)); + level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); + ASSERT_EQ(level, Level::METRICS_MISSING); + } + provider->setLatestDataDistributorEvents("TotalDataInFlight", LatestWorkerEvents()); + level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); + ASSERT_EQ(level, Level::METRICS_MISSING); + provider->setLatestDataDistributorEvents("TotalDataInFlight", + makeLatestWorkerEvents(teamMetrics(SERVER_KNOBS->PRIORITY_TEAM_HEALTHY))); provider->setLatestDataDistributorEvents("MovingData", LatestWorkerEvents()); level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); ASSERT_EQ(level, Level::METRICS_MISSING); diff --git a/fdbserver/clustercontroller/ClusterRecovery.cpp b/fdbserver/clustercontroller/ClusterRecovery.cpp index d1de04e86dd..6b5034ab198 100644 --- a/fdbserver/clustercontroller/ClusterRecovery.cpp +++ b/fdbserver/clustercontroller/ClusterRecovery.cpp @@ -531,6 +531,10 @@ Future trackTlogRecovery(Reference self, bool allLogs = newState.tLogs.size() == configuration.expectedLogSets(!self->primaryDcId.empty() ? self->primaryDcId[0] : Optional()); + // Anti-quorum permits STORAGE_RECOVERED before remote catch-up, so a lost region can be removed. + // Until catch-up or reconfiguration makes old history unnecessary, retain it in coordinator state and + // defer FULLY_RECOVERED and old-role retirement; initialization alone does not prove durable catch-up. + bool storageRecovered = newState.oldTLogData.empty() || self->logSystem->storageRecovered(); bool finalUpdate = newState.oldTLogData.empty() && allLogs; TraceEvent("TrackTLogRecovery") .detail("FinalUpdate", finalUpdate) @@ -540,8 +544,6 @@ Future trackTlogRecovery(Reference self, configuration.expectedLogSets(!self->primaryDcId.empty() ? self->primaryDcId[0] : Optional())) .detail("RecoveryCount", newState.recoveryCount); co_await self->cstate.write(newState, finalUpdate); - // Keep oldLogData in memory even after the coordinated state drops old generations. ServerDBInfo uses - // it to keep old-generation TLogs serving in case this master has to run recovery again. if (self->cstateUpdated.canBeSet()) { self->cstateUpdated.send(Void()); } @@ -554,6 +556,8 @@ Future trackTlogRecovery(Reference self, } if (finalUpdate) { + oldLogSystems->get()->stopRejoins(); + self->logSystem->retireOldLogRoles(newState); self->recoveryState = RecoveryState::FULLY_RECOVERED; TraceEvent(getRecoveryEventName(ClusterRecoveryEventType::CLUSTER_RECOVERY_STATE_EVENT_NAME).c_str(), self->dbgid) @@ -565,7 +569,7 @@ Future trackTlogRecovery(Reference self, self->dbgid) .detail("ActiveGenerations", 1) .trackLatest(self->clusterRecoveryGenerationsEventHolder->trackingKey); - } else if (newState.oldTLogData.empty() && self->recoveryState < RecoveryState::STORAGE_RECOVERED) { + } else if (storageRecovered && self->recoveryState < RecoveryState::STORAGE_RECOVERED) { self->recoveryState = RecoveryState::STORAGE_RECOVERED; TraceEvent(getRecoveryEventName(ClusterRecoveryEventType::CLUSTER_RECOVERY_STATE_EVENT_NAME).c_str(), self->dbgid) @@ -584,7 +588,6 @@ Future trackTlogRecovery(Reference self, self->registrationTrigger.trigger(); if (finalUpdate) { - oldLogSystems->get()->stopRejoins(); rejoinRequests = rejoinRequestHandler(self); co_return; } @@ -1453,7 +1456,7 @@ Future readTransactionSystemState(Reference self, RangeResult rawCdcHistoryTags = co_await self->txnStateStore->readRange(cdcTagHistoryKeys); for (auto& kv : rawCdcHistoryTags) { - const CDCTagHistoryEntry tagHistory = decodeCDCTagHistoryKey(kv.key); + const CDCTagHistoryEntry tagHistory = decodeCDCTagHistoryEntry(kv.key, kv.value); if (activeCdcStreams.contains(tagHistory.streamId)) { self->allTags.push_back(tagHistory.tag); } diff --git a/fdbserver/clustercontroller/NativeCdcProxyBalancer.cpp b/fdbserver/clustercontroller/NativeCdcProxyBalancer.cpp new file mode 100644 index 00000000000..9807ce2fe0d --- /dev/null +++ b/fdbserver/clustercontroller/NativeCdcProxyBalancer.cpp @@ -0,0 +1,204 @@ +/* + * NativeCdcProxyBalancer.cpp + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include +#include +#include +#include + +#include "fdbclient/DatabaseContext.h" +#include "fdbclient/SystemData.h" +#include "fdbserver/clustercontroller/NativeCdcProxyBalancer.h" +#include "flow/CodeProbe.h" +#include "flow/DeterministicRandom.h" +#include "flow/Trace.h" + +namespace { + +bool containsNativeCdcProxy(ClientDBInfo const& clientInfo, UID proxyId) { + return std::any_of(clientInfo.cdcProxies.begin(), + clientInfo.cdcProxies.end(), + [proxyId](CDCProxyInterface const& proxy) { return proxy.id() == proxyId; }); +} + +void signalNativeCdcProxyAssignmentChange(Transaction* tr) { + tr->set(cdcProxyAssignmentChangeKey, + BinaryWriter::toValue(deterministicRandom()->randomUniqueID(), + IncludeVersion(ProtocolVersion::withNativeCdc()))); +} + +} // namespace + +Future rebalanceNativeCdcProxyAssignments(Database cx, + std::vector availableProxies, + std::function stillEligible) { + // The metadata is read and the whole tag group is moved in one transaction. Do not split a shared tag + // across owners or make a partial move when the scan approaches the transaction size/lifetime limits. + constexpr int maxStreams = 512; + constexpr int maxHistoryRows = 2048; + constexpr int maxStreamsPerMove = 64; + constexpr int maxRangeBytes = 1 << 20; + const int64_t timeoutMs = 5000; + const std::set available(availableProxies.begin(), availableProxies.end()); + if (available.size() < 2) { + co_return false; + } + + Transaction tr(cx); + int attempts = 0; + while (true) { + if (++attempts > 3) { + co_return false; + } + Error err; + try { + tr.setOption(FDBTransactionOptions::LOCK_AWARE); + tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); + tr.setOption(FDBTransactionOptions::TIMEOUT, + StringRef(reinterpret_cast(&timeoutMs), sizeof(timeoutMs))); + const UID publishedInfoId = cx->clientInfo->get().id; + if (!cx->clientInfo->get().nativeCdcEnabled || !stillEligible()) { + co_return false; + } + + RangeResult active = co_await tr.getRange(cdcStreamKeys, GetRangeLimits(maxStreams + 1, maxRangeBytes)); + RangeResult assignments = co_await tr.getRange(cdcProxyKeys, GetRangeLimits(maxStreams + 1, maxRangeBytes)); + RangeResult histories = + co_await tr.getRange(cdcTagHistoryKeys, GetRangeLimits(maxHistoryRows + 1, maxRangeBytes)); + if (active.more || assignments.more || histories.more || active.size() > maxStreams || + assignments.size() > maxStreams || histories.size() > maxHistoryRows) { + CODE_PROBE(true, "Native CDC proxy rebalancing skips oversized metadata"); + co_return false; + } + + std::set activeIds; + for (const auto& stream : active) { + activeIds.insert(decodeCDCStreamKey(stream.key)); + } + std::map ownerByStream; + for (const auto& assignment : assignments) { + const auto [streamId, owner] = decodeCDCProxyKey(assignment.key); + if (activeIds.contains(streamId) && !ownerByStream.emplace(streamId, owner).second) { + co_return false; + } + } + std::map currentTagByStream; + std::set pendingHistories; + for (const auto& history : histories) { + const CDCTagHistoryEntry entry = decodeCDCTagHistoryKey(history.key); + if (activeIds.contains(entry.streamId)) { + currentTagByStream[entry.streamId] = entry.tag; + if (!history.value.empty()) { + pendingHistories.insert(entry.streamId); + } + } + } + + std::map ownerLoads; + for (const UID& proxyId : available) { + ownerLoads.emplace(proxyId, 0); + } + std::map> membersByTag; + for (const CDCStreamId streamId : activeIds) { + auto owner = ownerByStream.find(streamId); + auto tag = currentTagByStream.find(streamId); + if (owner == ownerByStream.end() || tag == currentTagByStream.end() || + !ownerLoads.contains(owner->second)) { + // Let the cluster controller repair missing or stale ownership before balancing. + co_return false; + } + const auto published = cx->clientInfo->get().streamToCDCProxyId.find(streamId); + if (published == cx->clientInfo->get().streamToCDCProxyId.end() || published->second != owner->second) { + co_return false; + } + ++ownerLoads[owner->second]; + membersByTag[tag->second].push_back(streamId); + } + + Optional selectedTag; + Optional selectedSource; + Optional selectedTarget; + int bestImprovement = 0; + size_t bestGroupSize = 0; + for (const auto& [tag, members] : membersByTag) { + const UID source = ownerByStream.at(members.front()); + for (const CDCStreamId streamId : members) { + if (ownerByStream.at(streamId) != source) { + // Registration relies on every stream sharing a current tag having one owner. + co_return false; + } + } + if (members.size() > maxStreamsPerMove || + std::any_of(members.begin(), members.end(), [&](CDCStreamId id) { + return pendingHistories.contains(id); + })) { + continue; + } + for (const UID& target : available) { + if (source == target) { + continue; + } + const int difference = ownerLoads.at(source) - ownerLoads.at(target); + const int moved = 2 * static_cast(members.size()); + const int after = difference >= moved ? difference - moved : moved - difference; + const int improvement = difference - after; + if (improvement > bestImprovement || + (improvement == bestImprovement && improvement > 0 && members.size() > bestGroupSize)) { + bestImprovement = improvement; + bestGroupSize = members.size(); + selectedTag = tag; + selectedSource = source; + selectedTarget = target; + } + } + } + if (!selectedTag.present()) { + co_return false; + } + if (!stillEligible() || !cx->clientInfo->get().nativeCdcEnabled || + cx->clientInfo->get().id != publishedInfoId || + !containsNativeCdcProxy(cx->clientInfo->get(), selectedSource.get()) || + !containsNativeCdcProxy(cx->clientInfo->get(), selectedTarget.get())) { + co_return false; + } + + for (const CDCStreamId streamId : membersByTag.at(selectedTag.get())) { + tr.clear(cdcProxyKeyFor(streamId, selectedSource.get())); + tr.set(cdcProxyKeyFor(streamId, selectedTarget.get()), Value()); + } + signalNativeCdcProxyAssignmentChange(&tr); + co_await tr.commit(); + CODE_PROBE(true, "Native CDC rebalances an entire shared tag across live proxies"); + TraceEvent("CDCProxyTagRebalanced") + .detail("Tag", selectedTag.get().toString()) + .detail("OldCDCProxyID", selectedSource.get()) + .detail("NewCDCProxyID", selectedTarget.get()) + .detail("StreamCount", bestGroupSize); + co_return true; + } catch (Error& e) { + // An ambiguous commit may already have moved a group. Reconcile on the next controller pass. + if (e.code() == error_code_commit_unknown_result) { + throw; + } + err = e; + } + co_await tr.onError(err); + } +} diff --git a/fdbserver/clustercontroller/Status.cpp b/fdbserver/clustercontroller/Status.cpp index 71af46f3cf8..820d60a5826 100644 --- a/fdbserver/clustercontroller/Status.cpp +++ b/fdbserver/clustercontroller/Status.cpp @@ -178,7 +178,7 @@ class StatusCounter { public: StatusCounter() : hz(0), roughness(0), counter(0) {} StatusCounter(double hz, double roughness, int64_t counter) : hz(hz), roughness(roughness), counter(counter) {} - explicit(false) StatusCounter(const std::string& parsableText) { parseText(parsableText); } + explicit StatusCounter(const std::string& parsableText) { parseText(parsableText); } StatusCounter& parseText(const std::string& parsableText) { sscanf(parsableText.c_str(), "%lf %lf %" SCNd64 "", &hz, &roughness, &counter); @@ -454,6 +454,16 @@ struct RolesInfo { return latencyStats; } + void appendLatencyStatistics(JsonBuilderObject& object, + EventMap const& metrics, + const char* eventName, + const char* jsonKey) { + TraceEventFields const& latencyMetrics = metrics.at(eventName); + if (latencyMetrics.size()) { + object[jsonKey] = addLatencyStatistics(latencyMetrics); + } + } + JsonBuilderObject addLatencyBandInfo(TraceEventFields const& metrics) { JsonBuilderObject latencyBands; std::map bands; @@ -547,10 +557,7 @@ struct RolesInfo { maxTLogVersion - version - SERVER_KNOBS->STORAGE_LOGGING_DELAY * SERVER_KNOBS->VERSIONS_PER_SECOND); } - TraceEventFields const& readLatencyMetrics = metrics.at("ReadLatencyMetrics"); - if (readLatencyMetrics.size()) { - obj["read_latency_statistics"] = addLatencyStatistics(readLatencyMetrics); - } + appendLatencyStatistics(obj, metrics, "ReadLatencyMetrics", "read_latency_statistics"); TraceEventFields const& readLatencyBands = metrics.at("ReadLatencyBands"); if (readLatencyBands.size()) { @@ -662,60 +669,22 @@ struct RolesInfo { obj["id"] = iface.id().shortString(); obj["role"] = role; try { - TraceEventFields const& commitLatencyMetrics = metrics.at("CommitLatencyMetrics"); - if (commitLatencyMetrics.size()) { - obj["commit_latency_statistics"] = addLatencyStatistics(commitLatencyMetrics); - } + appendLatencyStatistics(obj, metrics, "CommitLatencyMetrics", "commit_latency_statistics"); TraceEventFields const& commitLatencyBands = metrics.at("CommitLatencyBands"); if (commitLatencyBands.size()) { obj["commit_latency_bands"] = addLatencyBandInfo(commitLatencyBands); } - TraceEventFields const& commitBatchingWindowSize = metrics.at("CommitBatchingWindowSize"); - if (commitBatchingWindowSize.size()) { - obj["commit_batching_window_size"] = addLatencyStatistics(commitBatchingWindowSize); - } - - TraceEventFields const& commitBatchTransactions = metrics.at("CommitBatchTransactions"); - if (commitBatchTransactions.size()) { - obj["commit_batch_transactions"] = addLatencyStatistics(commitBatchTransactions); - } - - TraceEventFields const& commitBatchBytes = metrics.at("CommitBatchBytes"); - if (commitBatchBytes.size()) { - obj["commit_batch_bytes"] = addLatencyStatistics(commitBatchBytes); - } - - TraceEventFields const& commitBatchingWaiting = metrics.at("CommitBatchingWaiting"); - if (commitBatchingWaiting.size()) { - obj["commit_batching_waiting"] = addLatencyStatistics(commitBatchingWaiting); - } - - TraceEventFields const& commitPreresolutionLatency = metrics.at("CommitPreresolutionLatency"); - if (commitPreresolutionLatency.size()) { - obj["commit_preresolution_latency"] = addLatencyStatistics(commitPreresolutionLatency); - } - - TraceEventFields const& commitResolutionLatency = metrics.at("CommitResolutionLatency"); - if (commitResolutionLatency.size()) { - obj["commit_resolution_latency"] = addLatencyStatistics(commitResolutionLatency); - } - - TraceEventFields const& commitPostresolutionLatency = metrics.at("CommitPostresolutionLatency"); - if (commitPostresolutionLatency.size()) { - obj["commit_postresolution_latency"] = addLatencyStatistics(commitPostresolutionLatency); - } - - TraceEventFields const& commitTLogLoggingLatency = metrics.at("CommitTLogLoggingLatency"); - if (commitTLogLoggingLatency.size()) { - obj["commit_tlog_logging_latency"] = addLatencyStatistics(commitTLogLoggingLatency); - } - - TraceEventFields const& commitReplyLatency = metrics.at("CommitReplyLatency"); - if (commitReplyLatency.size()) { - obj["commit_reply_latency"] = addLatencyStatistics(commitReplyLatency); - } + appendLatencyStatistics(obj, metrics, "CommitBatchingWindowSize", "commit_batching_window_size"); + appendLatencyStatistics(obj, metrics, "CommitBatchTransactions", "commit_batch_transactions"); + appendLatencyStatistics(obj, metrics, "CommitBatchBytes", "commit_batch_bytes"); + appendLatencyStatistics(obj, metrics, "CommitBatchingWaiting", "commit_batching_waiting"); + appendLatencyStatistics(obj, metrics, "CommitPreresolutionLatency", "commit_preresolution_latency"); + appendLatencyStatistics(obj, metrics, "CommitResolutionLatency", "commit_resolution_latency"); + appendLatencyStatistics(obj, metrics, "CommitPostresolutionLatency", "commit_postresolution_latency"); + appendLatencyStatistics(obj, metrics, "CommitTLogLoggingLatency", "commit_tlog_logging_latency"); + appendLatencyStatistics(obj, metrics, "CommitReplyLatency", "commit_reply_latency"); } catch (Error& e) { if (e.code() != error_code_attribute_not_found) { throw e; @@ -735,15 +704,8 @@ struct RolesInfo { // GRV Latency metrics are grouped according to priority (currently batch or default). // Other priorities can be added in the future. - TraceEventFields const& grvLatencyMetrics = metrics.at("GRVLatencyMetrics"); - if (grvLatencyMetrics.size()) { - priorityStats["default"] = addLatencyStatistics(grvLatencyMetrics); - } - - TraceEventFields const& grvBatchMetrics = metrics.at("GRVBatchLatencyMetrics"); - if (grvBatchMetrics.size()) { - priorityStats["batch"] = addLatencyStatistics(grvBatchMetrics); - } + appendLatencyStatistics(priorityStats, metrics, "GRVLatencyMetrics", "default"); + appendLatencyStatistics(priorityStats, metrics, "GRVBatchLatencyMetrics", "batch"); // Add GRV Latency metrics (for all priorities) to parent node. if (!priorityStats.empty()) { @@ -1738,8 +1700,8 @@ loadConfiguration(Database cx, JsonBuilderArray* messages, std::set res.healthyZone = healthyZone.first; } else if (healthyZone.second > tr.getReadVersion().get()) { res.healthyZone = healthyZone.first; - res.healthyZoneSeconds = - (healthyZone.second - tr.getReadVersion().get()) / CLIENT_KNOBS->CORE_VERSIONSPERSECOND; + res.healthyZoneSeconds = static_cast(healthyZone.second - tr.getReadVersion().get()) / + CLIENT_KNOBS->CORE_VERSIONSPERSECOND; } } res.rebalanceDDIgnored = rebalanceDDIgnored.get().present(); diff --git a/fdbserver/clustercontroller/include/fdbserver/clustercontroller/NativeCdcProxyBalancer.h b/fdbserver/clustercontroller/include/fdbserver/clustercontroller/NativeCdcProxyBalancer.h new file mode 100644 index 00000000000..d15441f4d16 --- /dev/null +++ b/fdbserver/clustercontroller/include/fdbserver/clustercontroller/NativeCdcProxyBalancer.h @@ -0,0 +1,32 @@ +/* + * NativeCdcProxyBalancer.h + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#pragma once + +#include +#include + +#include "fdbclient/NativeCdc.h" + +// Moves at most one complete current-tag group between live proxies when doing so reduces stream-count skew. +// Returns false if the metadata is incomplete, too large to scan safely, or already balanced. +Future rebalanceNativeCdcProxyAssignments(Database cx, + std::vector availableProxies, + std::function stillEligible); diff --git a/fdbserver/commitproxy/CommitProxyServer.cpp b/fdbserver/commitproxy/CommitProxyServer.cpp index 9a3402d3303..8181459d0d8 100644 --- a/fdbserver/commitproxy/CommitProxyServer.cpp +++ b/fdbserver/commitproxy/CommitProxyServer.cpp @@ -1752,13 +1752,14 @@ Future postResolution(CommitBatchContext* self) { // @todo probably there is no need to get the (entire) version vector from the sequencer // in this case, and if so, consider adding a flag to the request to tell the sequencer // to not send the version vector information. + // Receive master replies at socket priority so incoming commits cannot starve an already-arrived reply. auto res = co_await race(pProxyCommitData->committedVersion.whenAtLeast( self->commitVersion - SERVER_KNOBS->MAX_READ_TRANSACTION_LIFE_VERSIONS), pProxyCommitData->cx->onProxiesChanged(), pProxyCommitData->master.getLiveCommittedVersion.getReply( GetRawCommittedVersionRequest(waitVersionSpan.context, debugID, invalidVersion), - TaskPriority::GetLiveCommittedVersionReply)); + TaskPriority::ReadSocket)); if (res.index() == 0) { co_await yield(); break; @@ -2208,9 +2209,7 @@ Future commitBatch(ProxyCommitData* pCommitData, if (err.code() == error_code_actor_cancelled) { throw; } - TraceEvent(SevInfo, "CommitBatchFailed", pCommitData->dbgid) - .detail("Stage", context.stage) - .detail("ErrorCode", err.code()); + TraceEvent(SevInfo, "CommitBatchFailed", pCommitData->dbgid).error(err).detail("Stage", context.stage); throw failed_to_progress(); } } @@ -2304,10 +2303,9 @@ static Future readRequestServer(CommitProxyInterface proxy, while (true) { GetKeyServerLocationsRequest req = co_await proxy.getKeyServersLocations.getFuture(); // WARNING: this code is run at a high priority, so it needs to do as little work as possible - if (req.limit != CLIENT_KNOBS->STORAGE_METRICS_SHARD_LIMIT && // Always do data distribution requests - (commitData->stats.keyServerLocationIn.getValue() - commitData->stats.keyServerLocationOut.getValue() > - SERVER_KNOBS->KEY_LOCATION_MAX_QUEUE_SIZE || - (g_network->isSimulated() && buggify(0.001)))) { + if (commitData->stats.keyServerLocationIn.getValue() - commitData->stats.keyServerLocationOut.getValue() > + SERVER_KNOBS->KEY_LOCATION_MAX_QUEUE_SIZE || + (g_network->isSimulated() && buggify(0.001))) { ++commitData->stats.keyServerLocationErrors; req.reply.sendError(commit_proxy_memory_limit_exceeded()); TraceEvent(SevWarnAlways, "ProxyLocationRequestThresholdExceeded").suppressFor(60); diff --git a/fdbserver/core/BulkDumpUtil.cpp b/fdbserver/core/BulkDumpUtil.cpp index 254689ae87b..5012582d6a9 100644 --- a/fdbserver/core/BulkDumpUtil.cpp +++ b/fdbserver/core/BulkDumpUtil.cpp @@ -22,12 +22,11 @@ #include "fdbclient/BulkLoading.h" #include "fdbclient/FDBTypes.h" #include "fdbclient/KeyRangeMap.h" +#include "fdbclient/NativeAPI.h" #include "fdbclient/S3Client.h" #include "fdbserver/core/BulkDumpUtil.h" #include "fdbserver/core/BulkLoadUtil.h" #include "fdbserver/core/Knobs.h" -#include "fdbserver/core/RocksDBCheckpointUtils.h" -#include "fdbserver/core/StorageMetrics.h" SSBulkDumpTask getSSBulkDumpTask(const std::map>& locations, const BulkDumpState& bulkDumpState) { StorageServerInterface targetServer; @@ -81,95 +80,6 @@ std::pair getLocalRemoteFileSetSetting(Version return std::make_pair(fileSetLocal, fileSetRemote); } -// Generate SST file given the input sortedKVS to the input filePath. -// TODO(BulkDump): This copy of sortedKVS can be a slow task if data is large. -void writeKVSToSSTFile(std::string filePath, std::map& sortedKVS, UID logId) { - const std::string absFilePath = abspath(filePath); - // Check file - if (fileExists(absFilePath)) { - TraceEvent(SevWarn, "SSBulkDumpRetriableError", logId) - .detail("Reason", "exist old File when writeKVSToSSTFile") - .detail("DataFilePathLocal", absFilePath); - ASSERT_WE_THINK(false); - throw retry(); - } - // Dump data to file - std::unique_ptr sstWriter = newRocksDBSstFileWriter(); - sstWriter->open(absFilePath); - for (const auto& [key, value] : sortedKVS) { - sstWriter->write(key, value); // assuming sorted - } - if (!sstWriter->finish()) { - // Unexpected: having data but failed to finish - TraceEvent(SevWarn, "SSBulkDumpRetriableError", logId) - .detail("Reason", "failed to finish data sst writer when writeKVSToSSTFile") - .detail("DataFilePath", absFilePath); - ASSERT_WE_THINK(false); - throw retry(); - } - return; -} - -Future dumpDataFileToLocalDirectory(UID logId, - std::shared_ptr rangeDumpRawData, - BulkLoadFileSet localFileSet, - BulkLoadFileSet remoteFileSet, - BulkLoadByteSampleSetting byteSampleSetting, - Version dumpVersion, - KeyRange dumpRange, - BulkLoadType dumpType, - BulkLoadTransportMethod transportMethod) { - // Step 1: Clean up local folder - resetFileFolder((abspath(localFileSet.getFolder()))); - - // Step 2: Dump data to file - bool containDataFile = false; - if (!rangeDumpRawData->kvs.empty()) { - writeKVSToSSTFile(abspath(localFileSet.getDataFileFullPath()), rangeDumpRawData->kvs, logId); - containDataFile = true; - } else { - ASSERT(rangeDumpRawData->sampled.empty()); - containDataFile = false; - } - - // Step 3: Dump sample to file - bool containByteSampleFile = false; - if (!rangeDumpRawData->sampled.empty()) { - writeKVSToSSTFile(abspath(localFileSet.getBytesSampleFileFullPath()), rangeDumpRawData->sampled, logId); - containByteSampleFile = true; - } else { - containByteSampleFile = false; - } - - // Step 4: Generate manifest file - if (fileExists(abspath(localFileSet.getManifestFileFullPath()))) { - TraceEvent(SevWarn, "SSBulkDumpRetriableError", logId) - .detail("Reason", "exist old manifestFile") - .detail("ManifestFilePathLocal", abspath(localFileSet.getManifestFileFullPath())); - ASSERT_WE_THINK(false); - throw retry(); - } - BulkLoadFileSet fileSetRemote(remoteFileSet.getRootPath(), - remoteFileSet.getRelativePath(), - remoteFileSet.getManifestFileName(), - containDataFile ? remoteFileSet.getDataFileName() : std::string(), - containByteSampleFile ? remoteFileSet.getByteSampleFileName() : std::string(), - BulkLoadChecksum()); - BulkLoadManifest manifestMetadata(fileSetRemote, - dumpRange.begin, - dumpRange.end, - dumpVersion, - rangeDumpRawData->kvsBytes, - rangeDumpRawData->kvs.size(), - byteSampleSetting, - dumpType, - transportMethod); - std::string manifestStr = manifestMetadata.toString(); - std::shared_ptr manifest = std::make_shared(std::move(manifestStr)); - co_await writeBulkFileBytes(abspath(localFileSet.getManifestFileFullPath()), manifest); - co_return manifestMetadata; -} - // Validate the invariant of filenames. Source is the file stored locally. Destination is the file going to move to. bool validateSourceDestinationFileSets(const BulkLoadFileSet& source, const BulkLoadFileSet& destination) { // Manifest file must be present diff --git a/fdbserver/core/BulkLoadUtil.cpp b/fdbserver/core/BulkLoadUtil.cpp index cc4ed52da33..21621ec72ea 100644 --- a/fdbserver/core/BulkLoadUtil.cpp +++ b/fdbserver/core/BulkLoadUtil.cpp @@ -24,8 +24,6 @@ #include "fdbclient/S3Client.h" #include "fdbserver/core/BulkLoadUtil.h" #include "fdbserver/core/Knobs.h" -#include "fdbserver/core/RocksDBCheckpointUtils.h" -#include "fdbserver/core/StorageMetrics.h" #include "flow/genericactors.h" #include "flow/UnitTest.h" @@ -182,71 +180,6 @@ Future getBulkLoadTaskStateFromDataMove(Database cx, } } -// Return true if generated the byte sampling file. Otherwise, return false. -// TODO(BulkDump): directly read from special key space. -Future doBytesSamplingOnDataFile(std::string dataFileFullPath, // input file - std::string byteSampleFileFullPath, // output file - UID logId) { - int counter = 0; - bool res = false; - int retryCount = 0; - double startTime = now(); - while (true) { - Error err; - try { - std::unique_ptr sstWriter = newRocksDBSstFileWriter(); - sstWriter->open(abspath(byteSampleFileFullPath)); - bool anySampled = false; - std::unique_ptr reader = newRocksDBSstFileReader(); - reader->open(abspath(dataFileFullPath)); - while (reader->hasNext()) { - KeyValue kv = reader->next(); - ByteSampleInfo sampleInfo = isKeyValueInSample(kv); - if (sampleInfo.inSample) { - sstWriter->write(kv.key, BinaryWriter::toValue(sampleInfo.sampledSize, Unversioned())); - anySampled = true; - counter++; - if (counter > SERVER_KNOBS->BULKLOAD_BYTE_SAMPLE_BATCH_KEY_COUNT) { - co_await yield(); - counter = 0; - } - } - } - // It is possible that no key is sampled - // This can happen when the data to sample is small - // In this case, no SST sample byte file is generated - if (anySampled) { - ASSERT(sstWriter->finish()); - res = true; - } else { - ASSERT(!sstWriter->finish()); - deleteFile(abspath(byteSampleFileFullPath)); - } - break; - } catch (Error& e) { - err = e; - } - if (err.code() == error_code_actor_cancelled) { - throw err; - } - TraceEvent(SevWarn, "SSBulkLoadTaskSamplingError", logId) - .errorUnsuppressed(err) - .detail("DataFileFullPath", dataFileFullPath) - .detail("ByteSampleFileFullPath", byteSampleFileFullPath) - .detail("Duration", now() - startTime) - .detail("RetryCount", retryCount); - co_await delay(5.0); - deleteFile(abspath(byteSampleFileFullPath)); - retryCount++; - } - TraceEvent(bulkLoadVerboseEventSev(), "SSBulkLoadTaskSamplingComplete", logId) - .detail("DataFileFullPath", dataFileFullPath) - .detail("ByteSampleFileFullPath", byteSampleFileFullPath) - .detail("Duration", now() - startTime) - .detail("RetryCount", retryCount); - co_return res; -} - // TODO(BulkLoad): slow task void clearFileFolder(const std::string& folderPath, const UID& logId, bool ignoreError) { try { diff --git a/fdbserver/core/CMakeLists.txt b/fdbserver/core/CMakeLists.txt index a94b1df9681..763c014ecad 100644 --- a/fdbserver/core/CMakeLists.txt +++ b/fdbserver/core/CMakeLists.txt @@ -19,17 +19,6 @@ target_include_directories(fdbserver_core ${CMAKE_CURRENT_SOURCE_DIR}/include ${CMAKE_CURRENT_BINARY_DIR}/include PRIVATE + ${CMAKE_SOURCE_DIR}/fdbclient ${CMAKE_SOURCE_DIR}/fdbserver/include) target_link_libraries(fdbserver_core PUBLIC fdbclient) - -if(WITH_ROCKSDB) - add_dependencies(fdbserver_core rocksdb) - if(WITH_LIBURING) - target_include_directories(fdbserver_core PRIVATE ${ROCKSDB_INCLUDE_DIR} ${uring_INCLUDE_DIR}) - target_link_libraries(fdbserver_core PRIVATE ${ROCKSDB_LIBRARIES} ${uring_LIBRARIES} ${LZ4_LIBRARY}) - else() - target_include_directories(fdbserver_core PRIVATE ${ROCKSDB_INCLUDE_DIR}) - target_link_libraries(fdbserver_core PRIVATE ${ROCKSDB_LIBRARIES} ${LZ4_LIBRARY}) - endif() - target_compile_definitions(fdbserver_core PUBLIC WITH_ROCKSDB) -endif() diff --git a/fdbserver/core/NativeCdcMetadata.cpp b/fdbserver/core/NativeCdcMetadata.cpp new file mode 100644 index 00000000000..58fdb60b2b5 --- /dev/null +++ b/fdbserver/core/NativeCdcMetadata.cpp @@ -0,0 +1,581 @@ +/* + * NativeCdcMetadata.cpp + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include +#include +#include +#include +#include +#include + +#include "fdbclient/DatabaseContext.h" +#include "fdbclient/Knobs.h" +#include "fdbclient/SystemData.h" +#include "NativeCdcInternal.h" +#include "fdbserver/core/NativeCdcMetadata.h" +#include "flow/CodeProbe.h" +#include "flow/Error.h" +#include "flow/Trace.h" +#include "flow/UnitTest.h" + +namespace { + +using CDCTagId = uint16_t; + +constexpr uint32_t maxNativeCdcTagCount = static_cast(std::numeric_limits::max()) + 1; + +bool validNativeCdcTagCount(int tagCount) { + return tagCount > 0 && static_cast(tagCount) <= maxNativeCdcTagCount; +} + +class NativeCdcIdentifierAllocator { + bool sawStream = false; + CDCStreamId maxStreamId = 0; + std::unordered_map tagStreamCounts; + +public: + void observeStreamId(CDCStreamId streamId) { + sawStream = true; + maxStreamId = std::max(maxStreamId, streamId); + } + + void observeTag(Tag tag) { + ASSERT_WE_THINK(tag.locality == tagLocalityCDC); + ++tagStreamCounts[tag.id]; + } + + bool hasStreams(Tag tag) const { return tagStreamCounts.contains(tag.id); } + + std::pair allocate(int tagCount) const { + if (sawStream && maxStreamId == std::numeric_limits::max()) { + throw operation_failed(); + } + + const CDCStreamId streamId = sawStream ? maxStreamId + 1 : 1; + if (!validNativeCdcTagCount(tagCount)) { + throw invalid_option_value(); + } + uint32_t leastStreams = std::numeric_limits::max(); + CDCTagId selectedTagId = 0; + for (uint32_t tagId = 0; tagId < static_cast(tagCount); ++tagId) { + auto count = tagStreamCounts.find(static_cast(tagId)); + const uint32_t streamCount = count == tagStreamCounts.end() ? 0 : count->second; + if (streamCount < leastStreams) { + leastStreams = streamCount; + selectedTagId = static_cast(tagId); + } + } + return { streamId, Tag(tagLocalityCDC, selectedTagId) }; + } +}; + +Future> getNativeCdcProxyAssignment(Transaction* tr, CDCStreamId streamId) { + RangeResult assignments = co_await tr->getRange(cdcProxyRangeFor(streamId), 2); + ASSERT_LE(assignments.size(), 1); + if (assignments.empty()) { + co_return Optional(); + } + const auto [assignedStreamId, proxyId] = decodeCDCProxyKey(assignments[0].key); + ASSERT_WE_THINK(assignedStreamId == streamId); + co_return proxyId; +} + +Future getNativeCdcCurrentTag(Transaction* tr, CDCStreamId streamId) { + // Tag-history keys sort by their big-endian assignment version, so the final + // key in this stream's prefix range contains its current tag. + RangeResult history = co_await tr->getRange(cdcTagHistoryRangeFor(streamId), 1, Snapshot::False, Reverse::True); + if (history.empty()) { + throw client_invalid_operation(); + } + co_return decodeCDCTagHistoryKey(history.front().key).tag; +} + +Future readNativeCdcCurrentTags(Transaction* tr, + std::unordered_map* currentTags, + NativeCdcIdentifierAllocator* allocator = nullptr) { + std::set activeStreamIds; + Key begin = cdcStreamKeys.begin; + while (begin < cdcStreamKeys.end) { + RangeResult streams = co_await tr->getRange(KeyRangeRef(begin, cdcStreamKeys.end), CLIENT_KNOBS->TOO_MANY); + for (const auto& kv : streams) { + const CDCStreamId streamId = decodeCDCStreamKey(kv.key); + activeStreamIds.insert(streamId); + if (allocator) { + allocator->observeStreamId(streamId); + } + } + if (!streams.more) { + break; + } + begin = keyAfter(streams.back().key); + } + + begin = cdcTagHistoryKeys.begin; + while (begin < cdcTagHistoryKeys.end) { + RangeResult histories = + co_await tr->getRange(KeyRangeRef(begin, cdcTagHistoryKeys.end), CLIENT_KNOBS->TOO_MANY); + for (const auto& kv : histories) { + const CDCTagHistoryEntry history = decodeCDCTagHistoryKey(kv.key); + if (allocator) { + allocator->observeStreamId(history.streamId); + } + if (activeStreamIds.contains(history.streamId)) { + (*currentTags)[history.streamId] = history.tag; + } + } + if (!histories.more) { + break; + } + begin = keyAfter(histories.back().key); + } +} + +Future> getNativeCdcProxyAssignmentForTag(Transaction* tr, Tag targetTag) { + const Key ownerKey = cdcTagOwnerKeyFor(targetTag); + Optional indexedStream = co_await tr->get(ownerKey); + if (indexedStream.present()) { + const CDCStreamId streamId = decodeCDCTagOwnerValue(indexedStream.get()); + Future> activeStream = tr->get(cdcStreamKeyFor(streamId)); + RangeResult history = co_await tr->getRange(cdcTagHistoryRangeFor(streamId), 1, Snapshot::False, Reverse::True); + // Keep the await separate so GCC 13 does not evaluate history.front() before the short-circuit guards. + const Optional activeStreamValue = co_await activeStream; + // The index is derived: removal or retagging can invalidate its representative, and the per-stream + // assignment remains authoritative across proxy replacement, including by older metadata writers. + if (activeStreamValue.present() && !history.empty() && + decodeCDCTagHistoryKey(history.front().key).tag == targetTag) { + Optional proxyId = co_await getNativeCdcProxyAssignment(tr, streamId); + if (proxyId.present()) { + CODE_PROBE(true, "Native CDC resolves a shared tag owner from its persisted index"); + co_return proxyId; + } + } + CODE_PROBE(true, "Native CDC rebuilds a stale tag owner index"); + tr->clear(ownerKey); + } + + std::unordered_map currentTags; + co_await readNativeCdcCurrentTags(tr, ¤tTags); + for (const auto& [streamId, tag] : currentTags) { + if (tag == targetTag) { + Optional proxyId = co_await getNativeCdcProxyAssignment(tr, streamId); + if (proxyId.present()) { + tr->set(ownerKey, cdcTagOwnerValue(streamId)); + CODE_PROBE(true, "Native CDC reconstructs a missing tag owner index from active streams"); + co_return proxyId; + } + } + } + co_return Optional(); +} + +void retireNativeCdcTag(Transaction* tr, Tag tag) { + // Dropping a history row must retain its final-pop obligation, including + // when another stream still protects the same tag or recovery intervenes. + tr->set(cdcRetiredTagPopKeyFor(tag), Value()); + tr->atomicOp( + cdcRetiredTagPopVersionKeyFor(tag), cdcVersionstampedMinVersionValue(), MutationRef::SetVersionstampedValue); +} + +void signalNativeCdcProxyAssignmentChange(Transaction* tr) { + // Assignment updates are low-rate control-plane operations. A single + // coalescing signal lets the cluster controller rescan all durable owners. + tr->set(cdcProxyAssignmentChangeKey, + BinaryWriter::toValue(deterministicRandom()->randomUniqueID(), + IncludeVersion(ProtocolVersion::withNativeCdc()))); +} + +Future observeNativeCdcMetadata(Transaction* tr, NativeCdcIdentifierAllocator* allocator) { + Optional maxStreamId = co_await tr->get(cdcMaxStreamIdKey); + if (maxStreamId.present()) { + allocator->observeStreamId(decodeCDCMaxStreamIdValue(maxStreamId.get())); + } + + std::unordered_map currentTags; + co_await readNativeCdcCurrentTags(tr, ¤tTags, allocator); + for (const auto& tagAssignment : currentTags) { + allocator->observeTag(tagAssignment.second); + } +} + +Future> readNativeCdcTagStateImpl(Transaction* tr, CDCStreamId streamId) { + Future> keysFuture = tr->get(cdcStreamKeyFor(streamId)); + Future> minimumFuture = tr->get(cdcMinVersionKeyFor(streamId)); + Future> ownerFuture = getNativeCdcProxyAssignment(tr, streamId); + Future historyFuture = + tr->getRange(cdcTagHistoryRangeFor(streamId), 3, Snapshot::False, Reverse::True); + const Optional keys = co_await keysFuture; + const Optional minimum = co_await minimumFuture; + const Optional owner = co_await ownerFuture; + const RangeResult history = co_await historyFuture; + if (!keys.present() || !minimum.present() || !owner.present() || history.empty() || history.more || + history.size() > 2) { + co_return Optional(); + } + NativeCdcTagState state; + state.streamId = streamId; + state.ranges = decodeCDCStreamKeysValue(keys.get()); + state.historyKey = history.front().key; + state.assignment = decodeCDCTagHistoryEntry(history.front().key, history.front().value); + state.proxyId = owner.get(); + state.minVersion = decodeCDCMinVersionValue(minimum.get()); + state.pending = history.size() > 1 || !history.front().value.empty(); + co_return state; +} + +bool sameNativeCdcTagState(NativeCdcTagState const& current, NativeCdcTagState const& expected) { + return current.streamId == expected.streamId && current.ranges == expected.ranges && + current.historyKey == expected.historyKey && current.proxyId == expected.proxyId && + current.assignment.version == expected.assignment.version && + current.assignment.tag == expected.assignment.tag; +} + +} // namespace + +Future> readNativeCdcTagState(Transaction* tr, CDCStreamId streamId) { + return readNativeCdcTagStateImpl(tr, streamId); +} + +Future>> readNativeCdcTagStates(Transaction* tr, int maxStreams) { + if (maxStreams <= 0 || maxStreams == std::numeric_limits::max()) { + throw invalid_option_value(); + } + const RangeResult streams = co_await tr->getRange(cdcStreamKeys, maxStreams + 1); + if (streams.more || streams.size() > maxStreams) { + co_return Optional>(); + } + std::vector>> reads; + reads.reserve(streams.size()); + for (const auto& stream : streams) { + reads.push_back(readNativeCdcTagState(tr, decodeCDCStreamKey(stream.key))); + } + const std::vector> states = co_await getAll(reads); + std::vector result; + result.reserve(states.size()); + for (const auto& state : states) { + if (!state.present()) { + co_return Optional>(); + } + result.push_back(state.get()); + } + co_return Optional>(std::move(result)); +} + +Future retagNativeCdcStream(Transaction* tr, NativeCdcTagState expected, Tag destination) { + Optional current = co_await readNativeCdcTagState(tr, expected.streamId); + if (!current.present() || !sameNativeCdcTagState(current.get(), expected) || current.get().pending || + destination.locality != tagLocalityCDC || destination == current.get().assignment.tag) { + co_return false; + } + const Optional destinationOwner = co_await getNativeCdcProxyAssignmentForTag(tr, destination); + if (destinationOwner.present() && destinationOwner.get() != current.get().proxyId) { + co_return false; + } + const Version readVersion = co_await tr->getReadVersion(); + const auto& clientInfo = tr->getDatabase()->clientInfo->get(); + if (!clientInfo.nativeCdcEnabled || !validNativeCdcTagCount(clientInfo.nativeCdcTagCount) || + destination.id >= clientInfo.nativeCdcTagCount) { + co_return false; + } + const Key historyKey = cdcTagHistoryKeyFor(expected.streamId, readVersion, destination); + if (historyKey <= current.get().historyKey) { + co_return false; + } + // The key orders assignments, while the value supplies the exact routing + // cutover. A read-version boundary could skip writes before this commit. + tr->atomicOp(historyKey, cdcVersionstampedMinVersionValue(), MutationRef::SetVersionstampedValue); + signalNativeCdcProxyAssignmentChange(tr); + co_return true; +} + +Future finishNativeCdcRetag(Transaction* tr, NativeCdcTagState expected) { + const Optional current = co_await readNativeCdcTagState(tr, expected.streamId); + if (!current.present() || !sameNativeCdcTagState(current.get(), expected) || !current.get().pending || + current.get().minVersion < current.get().assignment.version) { + co_return false; + } + const RangeResult history = co_await tr->getRange(cdcTagHistoryRangeFor(expected.streamId), 3); + if (history.more || history.empty() || history.size() > 2 || history.back().key != current.get().historyKey) { + co_return false; + } + std::set retiredTags; + for (const auto& row : history) { + const Tag tag = decodeCDCTagHistoryEntry(row.key, row.value).tag; + if (tag != current.get().assignment.tag) { + retiredTags.insert(tag); + } + } + tr->clear(cdcTagHistoryRangeFor(expected.streamId)); + tr->set(cdcTagHistoryKeyFor(expected.streamId, current.get().assignment.version, current.get().assignment.tag), + Value()); + for (const Tag tag : retiredTags) { + retireNativeCdcTag(tr, tag); + } + signalNativeCdcProxyAssignmentChange(tr); + co_return true; +} + +Future prepareNativeCdcStreamRegistration(Transaction* tr, + Key name, + std::vector ranges, + UID proxyId) { + normalizeNativeCdcStreamRanges(name, ranges); + + const Key nameKey = cdcStreamNameKeyFor(name); + Optional currentId = co_await tr->get(nameKey); + if (currentId.present()) { + const CDCStreamId streamId = decodeCDCStreamNameValue(currentId.get()); + Optional currentKeys = co_await tr->get(cdcStreamKeyFor(streamId)); + if (!currentKeys.present() || decodeCDCStreamKeysValue(currentKeys.get()) != ranges) { + throw client_invalid_operation(); + } + if (!(co_await getNativeCdcProxyAssignment(tr, streamId)).present()) { + CODE_PROBE(true, "Native CDC registration restores missing stream owner", probe::decoration::rare); + const Tag tag = co_await getNativeCdcCurrentTag(tr, streamId); + Optional sharedTagProxy = co_await getNativeCdcProxyAssignmentForTag(tr, tag); + CODE_PROBE( + sharedTagProxy.present(), "Native CDC shared-tag streams use one owner", probe::decoration::rare); + const UID selectedProxy = sharedTagProxy.present() ? sharedTagProxy.get() : proxyId; + tr->set(cdcProxyKeyFor(streamId, selectedProxy), Value()); + if (!sharedTagProxy.present()) { + tr->set(cdcTagOwnerKeyFor(tag), cdcTagOwnerValue(streamId)); + } + signalNativeCdcProxyAssignmentChange(tr); + co_return NativeCdcRegistrationResult{ streamId, true }; + } + co_return NativeCdcRegistrationResult{ streamId, false }; + } + + // Disabling CDC stops new admission, but existing registrations and + // owner repair must remain available so durable streams can drain. + const bool nativeCdcEnabled = tr->getDatabase()->clientInfo->get().nativeCdcEnabled; + const int nativeCdcTagCount = tr->getDatabase()->clientInfo->get().nativeCdcTagCount; + validateNativeCdcEnabled(nativeCdcEnabled); + NativeCdcIdentifierAllocator allocator; + co_await observeNativeCdcMetadata(tr, &allocator); + const auto [streamId, tag] = allocator.allocate(nativeCdcTagCount); + // The read version is a conservative lower bound for tag routing. + // The versionstamped minimum below is the commit version, and stream + // initialization takes their maximum before exposing mutations. + const Version registrationVersion = co_await tr->getReadVersion(); + + tr->set(nameKey, cdcStreamNameValue(streamId)); + tr->set(cdcMaxStreamIdKey, cdcMaxStreamIdValue(streamId)); + tr->set(cdcStreamKeyFor(streamId), cdcStreamKeysValue(ranges)); + tr->set(cdcTagHistoryKeyFor(streamId, registrationVersion, tag), Value()); + tr->atomicOp( + cdcMinVersionKeyFor(streamId), cdcVersionstampedMinVersionValue(), MutationRef::SetVersionstampedValue); + Optional sharedTagProxy; + if (allocator.hasStreams(tag)) { + sharedTagProxy = co_await getNativeCdcProxyAssignmentForTag(tr, tag); + } + const UID selectedProxy = sharedTagProxy.present() ? sharedTagProxy.get() : proxyId; + tr->set(cdcProxyKeyFor(streamId, selectedProxy), Value()); + if (!sharedTagProxy.present()) { + tr->set(cdcTagOwnerKeyFor(tag), cdcTagOwnerValue(streamId)); + } + signalNativeCdcProxyAssignmentChange(tr); + co_return NativeCdcRegistrationResult{ streamId, true }; +} + +Future registerNativeCdcStream(Database cx, Key name, std::vector ranges, UID proxyId) { + normalizeNativeCdcStreamRanges(name, ranges); + + Transaction tr(cx); + while (true) { + Error err; + try { + tr.setOption(FDBTransactionOptions::LOCK_AWARE); + tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); + + const NativeCdcRegistrationResult result = + co_await prepareNativeCdcStreamRegistration(&tr, name, ranges, proxyId); + if (result.requiresCommit) { + co_await tr.commit(); + } + co_return result.streamId; + } catch (Error& e) { + err = e; + } + co_await tr.onError(err); + } +} + +Future removeNativeCdcStream(Database cx, Key name, CDCStreamId streamId, UID proxyId) { + if (name.empty() || streamId == 0) { + throw client_invalid_operation(); + } + + Transaction tr(cx); + while (true) { + Error err; + try { + tr.setOption(FDBTransactionOptions::LOCK_AWARE); + tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); + + const Key nameKey = cdcStreamNameKeyFor(name); + Optional currentId = co_await tr.get(nameKey); + if (!nativeCdcNameMatchesStream(currentId, streamId)) { + CODE_PROBE(currentId.present(), "Native CDC preserves a replacement stream during removal retry"); + if (currentId.present()) { + TraceEvent("NativeCdcRemovalPreservesReplacement") + .detail("RemovedStreamId", streamId) + .detail("ReplacementStreamId", decodeCDCStreamNameValue(currentId.get())); + } + co_return false; + } + + Optional assignedProxy = co_await getNativeCdcProxyAssignment(&tr, streamId); + if (!assignedProxy.present() || assignedProxy.get() != proxyId) { + CODE_PROBE(true, "Native CDC rejects removal through a stale owner"); + throw wrong_shard_server(); + } + + std::set removedTags; + const KeyRange historyRange = cdcTagHistoryRangeFor(streamId); + Key begin = historyRange.begin; + while (begin < historyRange.end) { + RangeResult history = + co_await tr.getRange(KeyRangeRef(begin, historyRange.end), CLIENT_KNOBS->TOO_MANY); + for (const auto& entry : history) { + removedTags.insert(decodeCDCTagHistoryKey(entry.key).tag); + } + if (!history.more) { + break; + } + begin = keyAfter(history.back().key); + } + + tr.clear(nameKey); + tr.clear(cdcStreamKeyFor(streamId)); + for (const Tag& tag : removedTags) { + const Key ownerKey = cdcTagOwnerKeyFor(tag); + Optional indexedStream = co_await tr.get(ownerKey); + if (indexedStream.present() && decodeCDCTagOwnerValue(indexedStream.get()) == streamId) { + tr.clear(ownerKey); + } + retireNativeCdcTag(&tr, tag); + } + tr.clear(cdcTagHistoryRangeFor(streamId)); + tr.clear(cdcMinVersionKeyFor(streamId)); + tr.clear(cdcProxyRangeFor(streamId)); + if (assignedProxy.present()) { + signalNativeCdcProxyAssignmentChange(&tr); + } + co_await tr.commit(); + CODE_PROBE(!removedTags.empty(), "Native CDC removal records final tagged pop work"); + TraceEvent("NativeCdcStreamRemoved") + .detail("StreamId", streamId) + .detail("ProxyID", proxyId) + .detail("CommitVersion", tr.getCommittedVersion()) + .detail("RetiredTagCount", removedTags.size()); + co_return true; + } catch (Error& e) { + if (e.code() == error_code_wrong_shard_server) { + throw; + } + err = e; + } + co_await tr.onError(err); + } +} + +Future reassignNativeCdcStreams(Database cx, UID oldProxyId, UID newProxyId) { + if (oldProxyId == newProxyId) { + co_return; + } + + Transaction tr(cx); + while (true) { + Error err; + try { + tr.setOption(FDBTransactionOptions::LOCK_AWARE); + tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); + tr.setOption(FDBTransactionOptions::PRIORITY_SYSTEM_IMMEDIATE); + + bool changed = false; + Key begin = cdcProxyKeys.begin; + while (begin < cdcProxyKeys.end) { + RangeResult assignments = + co_await tr.getRange(KeyRangeRef(begin, cdcProxyKeys.end), CLIENT_KNOBS->TOO_MANY); + for (const auto& assignment : assignments) { + const auto [streamId, proxyId] = decodeCDCProxyKey(assignment.key); + if (proxyId == oldProxyId) { + tr.clear(assignment.key); + tr.set(cdcProxyKeyFor(streamId, newProxyId), Value()); + changed = true; + } + } + if (!assignments.more) { + break; + } + begin = keyAfter(assignments.back().key); + } + + if (changed) { + CODE_PROBE(true, "Native CDC reassigns streams after proxy replacement"); + signalNativeCdcProxyAssignmentChange(&tr); + co_await tr.commit(); + } + co_return; + } catch (Error& e) { + err = e; + } + co_await tr.onError(err); + } +} + +void forceLinkNativeCdcMetadataTests() {} + +TEST_CASE("/NativeCDC/LifecycleAllocation") { + ASSERT(!validNativeCdcTagCount(-1)); + ASSERT(!validNativeCdcTagCount(0)); + ASSERT(validNativeCdcTagCount(1)); + ASSERT(validNativeCdcTagCount(std::numeric_limits::max() + 1u)); + ASSERT(!validNativeCdcTagCount(std::numeric_limits::max() + 2u)); + + NativeCdcIdentifierAllocator allocator; + auto [initialId, initialTag] = allocator.allocate(CLIENT_KNOBS->NATIVE_CDC_TAG_COUNT); + ASSERT_EQ(initialId, 1); + ASSERT_EQ(initialTag, Tag(tagLocalityCDC, 0)); + + allocator.observeStreamId(9); + allocator.observeTag(initialTag); + allocator.observeTag(Tag(tagLocalityCDC, 2)); + auto [nextId, nextTag] = allocator.allocate(CLIENT_KNOBS->NATIVE_CDC_TAG_COUNT); + ASSERT_EQ(nextId, 10); + ASSERT_EQ(nextTag, Tag(tagLocalityCDC, 1)); + + NativeCdcIdentifierAllocator publishedPoolAllocator; + publishedPoolAllocator.observeTag(Tag(tagLocalityCDC, 0)); + // The cluster-controller-published pool is authoritative even when it differs from this process's knob. + auto [publishedPoolId, publishedPoolTag] = publishedPoolAllocator.allocate(1); + ASSERT_EQ(publishedPoolId, 1); + ASSERT_EQ(publishedPoolTag, Tag(tagLocalityCDC, 0)); + + NativeCdcIdentifierAllocator fullPoolAllocator; + for (uint32_t tagId = 0; tagId < static_cast(CLIENT_KNOBS->NATIVE_CDC_TAG_COUNT); ++tagId) { + fullPoolAllocator.observeTag(Tag(tagLocalityCDC, static_cast(tagId))); + } + auto [sharedId, sharedTag] = fullPoolAllocator.allocate(CLIENT_KNOBS->NATIVE_CDC_TAG_COUNT); + ASSERT_EQ(sharedId, 1); + ASSERT_EQ(sharedTag, Tag(tagLocalityCDC, 0)); + + return Void(); +} diff --git a/fdbserver/core/ServerCheckpoint.cpp b/fdbserver/core/ServerCheckpoint.cpp index 39bf743aa58..0869802bd0f 100644 --- a/fdbserver/core/ServerCheckpoint.cpp +++ b/fdbserver/core/ServerCheckpoint.cpp @@ -1,5 +1,5 @@ /* - *ServerCheckpoint.cpp + * ServerCheckpoint.cpp * * This source file is part of the FoundationDB open source project * @@ -19,84 +19,6 @@ */ #include "fdbserver/core/ServerCheckpoint.h" -#include "fdbserver/core/RocksDBCheckpointUtils.h" - -ICheckpointReader* newCheckpointReader(const CheckpointMetaData& checkpoint, - const CheckpointAsKeyValues checkpointAsKeyValues, - UID logID) { - const CheckpointFormat format = checkpoint.getFormat(); - if (format == DataMoveRocksCF || format == RocksDB) { - return newRocksDBCheckpointReader(checkpoint, checkpointAsKeyValues, logID); - } else { - throw not_implemented(); - } - - return nullptr; -} - -Future deleteCheckpoint(CheckpointMetaData checkpoint) { - co_await delay(0, TaskPriority::FetchKeys); - const CheckpointFormat format = checkpoint.getFormat(); - if (format == DataMoveRocksCF || format == RocksDB || format == RocksDBKeyValues) { - if (!checkpoint.dir.empty()) { - platform::eraseDirectoryRecursive(checkpoint.dir); - } else { - TraceEvent(SevWarn, "CheckpointDirNotFound").detail("Checkpoint", checkpoint.toString()); - } - } else { - throw not_implemented(); - } -} - -Future fetchCheckpoint(Database cx, - CheckpointMetaData initialState, - std::string dir, - std::function(const CheckpointMetaData&)> cFun) { - TraceEvent("FetchCheckpointBegin", initialState.checkpointID).detail("CheckpointMetaData", initialState.toString()); - - CheckpointMetaData result; - const CheckpointFormat format = initialState.getFormat(); - ASSERT(format != RocksDBKeyValues); - if (format == DataMoveRocksCF || format == RocksDB) { - result = co_await fetchRocksDBCheckpoint(cx, initialState, dir, cFun); - } else { - throw not_implemented(); - } - - TraceEvent("FetchCheckpointEnd", initialState.checkpointID).detail("CheckpointMetaData", result.toString()); - co_return result; -} - -Future fetchCheckpointRanges(Database cx, - CheckpointMetaData initialState, - std::string dir, - std::vector ranges, - std::function(const CheckpointMetaData&)> cFun) { - TraceEvent(SevDebug, "FetchCheckpointRangesBegin", initialState.checkpointID) - .detail("CheckpointMetaData", initialState.toString()) - .detail("Ranges", describe(ranges)); - ASSERT(!ranges.empty()); - - CheckpointMetaData result; - const CheckpointFormat format = initialState.getFormat(); - if (format != RocksDBKeyValues) { - if (format != DataMoveRocksCF) { - throw not_implemented(); - } - initialState.setFormat(RocksDBKeyValues); - initialState.ranges = ranges; - initialState.dir = dir; - initialState.setSerializedCheckpoint( - ObjectWriter::toValue(RocksDBCheckpointKeyValues(ranges), IncludeVersion())); - } - - result = co_await fetchRocksDBCheckpoint(cx, initialState, dir, cFun); - - TraceEvent(SevDebug, "FetchCheckpointRangesEnd", initialState.checkpointID) - .detail("CheckpointMetaData", result.toString()) - .detail("Ranges", describe(ranges)); - co_return result; -} std::string serverCheckpointDir(const std::string& baseDir, const UID& checkpointId) { return joinPath(baseDir, checkpointId.toString()); diff --git a/fdbserver/core/ServerKnobs.cpp b/fdbserver/core/ServerKnobs.cpp index 0bd7a4e8d55..5ab5d42e811 100644 --- a/fdbserver/core/ServerKnobs.cpp +++ b/fdbserver/core/ServerKnobs.cpp @@ -189,6 +189,9 @@ void ServerKnobs::initialize(Randomize randomize, ClientKnobs* clientKnobs, IsSi init( CDC_PROXY_FAILURE_COALESCE_DELAY, 0.0 ); init( CDC_PROXY_POP_MIN_INTERVAL, 0.1 ); if( randomize && buggify() ) CDC_PROXY_POP_MIN_INTERVAL = 0.01; init( CDC_PROXY_POP_SCAN_INTERVAL, 5.0 ); if( randomize && buggify() ) CDC_PROXY_POP_SCAN_INTERVAL = 0.1; + init( CDC_PROXY_REBALANCE_ENABLED, false ); if( randomize && buggify() ) CDC_PROXY_REBALANCE_ENABLED = true; + init( CDC_PROXY_REBALANCE_INTERVAL, 60.0 ); if( randomize && buggify() ) CDC_PROXY_REBALANCE_INTERVAL = deterministicRandom()->randomInt(10, 121); + init( NATIVE_CDC_RETAG_CLEANUP_INTERVAL, 30.0 ); if( randomize && buggify() ) NATIVE_CDC_RETAG_CLEANUP_INTERVAL = 0.1; init( APPLY_MUTATION_BYTES, 1e6 ); init( BUGGIFY_RECOVER_MEMORY_LIMIT, 1e6 ); init( BUGGIFY_WORKER_REMOVED_MAX_LAG, 30 ); diff --git a/fdbserver/core/include/fdbserver/core/BulkDumpUtil.h b/fdbserver/core/include/fdbserver/core/BulkDumpUtil.h index 88f798f3061..b5c767a7bd8 100644 --- a/fdbserver/core/include/fdbserver/core/BulkDumpUtil.h +++ b/fdbserver/core/include/fdbserver/core/BulkDumpUtil.h @@ -86,21 +86,6 @@ std::string generateBulkDumpJobFolder(const UID& jobId); // Define task folder name. std::string getBulkDumpJobTaskFolder(const UID& jobId, const UID& taskId); -// Generate key-value data, byte sampling data, and manifest file. -// Return BulkLoadManifest metadata (equivalent to content of the manifest file). -// TODO(BulkDump): can cause slow tasks, do the task in a separate thread in the future. -// The size of sortedData is defined at the place of generating the data (getRangeDataToDump). -// The size is configured by MOVE_SHARD_KRM_ROW_LIMIT. -Future dumpDataFileToLocalDirectory(UID logId, - std::shared_ptr rangeDumpRawData, - BulkLoadFileSet localFileSet, - BulkLoadFileSet remoteFileSet, - BulkLoadByteSampleSetting byteSampleSetting, - Version dumpVersion, - KeyRange dumpRange, - BulkLoadType dumpType, - BulkLoadTransportMethod transportMethod); - // Upload manifest file for bulkdump job // Each job has one manifest file including manifest paths of all tasks. // The local file path: /-manifest.txt @@ -119,7 +104,7 @@ Future uploadBulkDumpFileSet(BulkLoadTransportMethod transportMethod, class ParallelismLimitor { public: - explicit(false) ParallelismLimitor(int maxParallelism) : maxParallelism(maxParallelism) {} + explicit ParallelismLimitor(int maxParallelism) : maxParallelism(maxParallelism) {} inline void decrementTaskCounter() { ASSERT(numRunningTasks.get() <= maxParallelism); diff --git a/fdbserver/core/include/fdbserver/core/BulkLoadUtil.h b/fdbserver/core/include/fdbserver/core/BulkLoadUtil.h index 3805ab09e71..4ad7887a79c 100644 --- a/fdbserver/core/include/fdbserver/core/BulkLoadUtil.h +++ b/fdbserver/core/include/fdbserver/core/BulkLoadUtil.h @@ -56,8 +56,6 @@ Future bulkLoadDownloadTaskFileSets(BulkLoadTransportMethod transportMetho std::string toLocalRoot, UID logId); -Future doBytesSamplingOnDataFile(std::string dataFileFullPath, std::string byteSampleFileFullPath, UID logId); - // Download job manifest file which is generated when dumping the data Future downloadBulkLoadJobManifestFile(BulkLoadTransportMethod transportMethod, std::string localJobManifestFilePath, diff --git a/fdbserver/core/include/fdbserver/core/Knobs.h b/fdbserver/core/include/fdbserver/core/Knobs.h index 2f65564bde0..7482d191e72 100644 --- a/fdbserver/core/include/fdbserver/core/Knobs.h +++ b/fdbserver/core/include/fdbserver/core/Knobs.h @@ -80,6 +80,9 @@ class SWIFT_CXX_IMMORTAL_SINGLETON_TYPE ServerKnobs : public KnobsImpl ranges; + Key historyKey; + CDCTagHistoryEntry assignment; + UID proxyId; + Version minVersion = invalidVersion; + bool pending = false; +}; + +// An absent result means the bounded snapshot is incomplete; it must not be used +// as an empty or zero-load configuration. +Future> readNativeCdcTagState(Transaction* tr, CDCStreamId streamId); +Future>> readNativeCdcTagStates(Transaction* tr, int maxStreams); +// These helpers revalidate the durable identity and prepare mutations without +// committing. The caller must fence its controller ownership in this transaction. +Future retagNativeCdcStream(Transaction* tr, NativeCdcTagState expected, Tag destination); +Future finishNativeCdcRetag(Transaction* tr, NativeCdcTagState expected); + +struct NativeCdcRegistrationResult { + CDCStreamId streamId; + // Describes only mutations prepared by this helper, not unrelated caller mutations. + bool requiresCommit; +}; + +// Prepares one registration without committing or retrying. The caller sets LOCK_AWARE and ACCESS_SYSTEM_KEYS +// and owns commit and retry handling. Transaction does not read its own writes: prepare at most one registration +// per transaction, without earlier mutations to the CDC metadata this operation reads. +Future prepareNativeCdcStreamRegistration(Transaction* tr, + Key name, + std::vector ranges, + UID proxyId); + +// Durable metadata operations used by CDC server roles. Registration is +// feature gated; drain and cleanup operations remain available for streams +// persisted before native CDC is disabled. +Future registerNativeCdcStream(Database cx, Key name, std::vector ranges, UID proxyId); +// Persists per-tag final-pop watermarks before removing stream metadata. +Future removeNativeCdcStream(Database cx, Key name, CDCStreamId streamId, UID proxyId); +// Atomically moves any streams assigned to a failed proxy to its replacement. +Future reassignNativeCdcStreams(Database cx, UID oldProxyId, UID newProxyId); + +#endif // FDBSERVER_CORE_NATIVECDCMETADATA_H diff --git a/fdbserver/core/include/fdbserver/core/ServerCheckpoint.h b/fdbserver/core/include/fdbserver/core/ServerCheckpoint.h index 1f0458e8121..3362e7d1c3b 100644 --- a/fdbserver/core/include/fdbserver/core/ServerCheckpoint.h +++ b/fdbserver/core/include/fdbserver/core/ServerCheckpoint.h @@ -57,27 +57,5 @@ class ICheckpointReader { virtual ~ICheckpointReader() = default; }; -ICheckpointReader* newCheckpointReader(const CheckpointMetaData& checkpoint, - const CheckpointAsKeyValues checkpointAsKeyValues, - UID logID); - -// Delete a checkpoint. -Future deleteCheckpoint(CheckpointMetaData checkpoint); - -// Fetches checkpoint to a local `dir`, `initialState` provides the checkpoint formats, location, restart point, etc. -// If cFun is provided, the progress can be checkpointed. -// Returns a CheckpointMetaData, which could contain KVS-specific results, e.g., the list of fetched checkpoint files. -Future fetchCheckpoint(Database cx, - CheckpointMetaData initialState, - std::string dir, - std::function(const CheckpointMetaData&)> cFun = nullptr); - -// Same as above, except that the checkpoint is fetched as key-value pairs. -Future fetchCheckpointRanges(Database cx, - CheckpointMetaData initialState, - std::string dir, - std::vector ranges, - std::function(const CheckpointMetaData&)> cFun = nullptr); - std::string serverCheckpointDir(const std::string& baseDir, const UID& checkpointId); std::string fetchedCheckpointDir(const std::string& baseDir, const UID& checkpointId); diff --git a/fdbserver/core/include/fdbserver/core/TLogInterface.h b/fdbserver/core/include/fdbserver/core/TLogInterface.h index d1ff558636a..10ea6a3a716 100644 --- a/fdbserver/core/include/fdbserver/core/TLogInterface.h +++ b/fdbserver/core/include/fdbserver/core/TLogInterface.h @@ -399,7 +399,8 @@ struct TLogQueuingMetricsReply { int64_t instanceID; // changes if bytesDurable and bytesInput reset int64_t bytesDurable{ 0 }, bytesInput{ 0 }; StorageBytes storageBytes; - Version v; // committed version + // Durable known-committed version. Recovery uses this to decide when old log history is safe to discard. + Version v; template void serialize(Ar& ar) { diff --git a/fdbserver/core/include/fdbserver/core/WorkerInterface.h b/fdbserver/core/include/fdbserver/core/WorkerInterface.h index 428ad321268..989e33d32d3 100644 --- a/fdbserver/core/include/fdbserver/core/WorkerInterface.h +++ b/fdbserver/core/include/fdbserver/core/WorkerInterface.h @@ -79,7 +79,7 @@ struct WorkerInterface { Optional grpcAddress() const { return clientInterface.grpcAddress; } WorkerInterface() = default; - explicit(false) WorkerInterface(const LocalityData& locality) : locality(locality) {} + explicit WorkerInterface(const LocalityData& locality) : locality(locality) {} void initEndpoints() { clientInterface.initEndpoints(); @@ -573,7 +573,7 @@ struct GetEncryptionAtRestModeResponse { uint32_t mode; GetEncryptionAtRestModeResponse() : mode(EncryptionAtRestModeDeprecated::Mode::DISABLED) {} - explicit(false) GetEncryptionAtRestModeResponse(uint32_t m) : mode(m) {} + explicit GetEncryptionAtRestModeResponse(uint32_t m) : mode(m) {} template void serialize(Ar& ar) { @@ -587,7 +587,7 @@ struct GetEncryptionAtRestModeRequest { ReplyPromise reply; GetEncryptionAtRestModeRequest() = default; - explicit(false) GetEncryptionAtRestModeRequest(UID tId) : tlogId(tId) {} + explicit GetEncryptionAtRestModeRequest(UID tId) : tlogId(tId) {} template void serialize(Ar& ar) { @@ -938,7 +938,7 @@ struct ExecuteRequest { Arena arena; StringRef execPayload; - explicit(false) ExecuteRequest(StringRef execPayload) : execPayload(execPayload) {} + explicit ExecuteRequest(StringRef execPayload) : execPayload(execPayload) {} ExecuteRequest() : execPayload() {} @@ -1059,7 +1059,7 @@ struct DiskStoreRequest { bool includePartialStores; ReplyPromise>> reply; - explicit(false) DiskStoreRequest(bool includePartialStores = false) : includePartialStores(includePartialStores) {} + explicit DiskStoreRequest(bool includePartialStores = false) : includePartialStores(includePartialStores) {} template void serialize(Ar& ar) { diff --git a/fdbserver/datadistributor/CMakeLists.txt b/fdbserver/datadistributor/CMakeLists.txt index e888f7d8718..69f2895c4a5 100644 --- a/fdbserver/datadistributor/CMakeLists.txt +++ b/fdbserver/datadistributor/CMakeLists.txt @@ -13,5 +13,6 @@ target_include_directories(fdbserver_datadistributor PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/include PRIVATE + ${CMAKE_SOURCE_DIR}/fdbclient ${CMAKE_CURRENT_SOURCE_DIR}) target_link_libraries(fdbserver_datadistributor PRIVATE fdbserver_core) diff --git a/fdbserver/datadistributor/DDTeamCollection.cpp b/fdbserver/datadistributor/DDTeamCollection.cpp index dbcec8a9e48..916417872ee 100644 --- a/fdbserver/datadistributor/DDTeamCollection.cpp +++ b/fdbserver/datadistributor/DDTeamCollection.cpp @@ -153,7 +153,7 @@ class DDTeamCollectionImpl { start = now(); } } catch (Error& e) { - TraceEvent("CheckAndRemoveInvalidLocalityAddrRetry", self->distributorId).detail("Error", e.what()); + TraceEvent("CheckAndRemoveInvalidLocalityAddrRetry", self->distributorId).error(e); } } } @@ -3202,7 +3202,7 @@ class DDTeamCollectionImpl { .detail("NumExistingSS", numExistingSS); } - if (hasHealthyTeam && !tssState->active && tssToRecruit > 0) { + if (hasHealthyTeam && !tssState->isActive() && tssToRecruit > 0) { TraceEvent("TSS_Recruit", self->distributorId) .detail("Stage", "HoldTSS") .detail("Addr", candidateSSAddr.toString()) @@ -3218,7 +3218,7 @@ class DDTeamCollectionImpl { initializeStorage(self, candidateWorker, ddEnabledState, true, tssState)); checkTss = self->initialFailureReactionDelay; } else { - if (tssState->active && tssState->inDataZone(candidateWorker.worker.locality)) { + if (tssState->isActive() && tssState->inDataZone(candidateWorker.worker.locality)) { CODE_PROBE(true, "TSS recruits pair in same dc/datahall"); self->isTssRecruiting = false; TraceEvent("TSS_Recruit", self->distributorId) @@ -3232,7 +3232,7 @@ class DDTeamCollectionImpl { tssState = makeReference(); } else { CODE_PROBE( - tssState->active, + tssState->isActive(), "TSS recruitment skipped potential pair because it's in a different dc/datahall"); self->addActor.send(initializeStorage( self, candidateWorker, ddEnabledState, false, makeReference())); @@ -3690,6 +3690,7 @@ class DDTeamCollectionImpl { .detail("StorageTeamSize", self->configuration.storageTeamSize) .detail("ZeroHealthy", self->zeroOptimalTeams.get()) .detail("HighestPriority", highestPriority) + .detail("HighestTeamPriority", self->getHighestTeamPriority()) .trackLatest(self->primary ? "TotalDataInFlight" : "TotalDataInFlightRemote"); // This trace event's trackLatest // lifetime is controlled by @@ -4638,6 +4639,31 @@ void DDTeamCollection::resetLocalitySet() { } } +int DDTeamCollection::getHighestTeamPriority() const { + if (teamCollections.empty()) { + return -1; + } + int highestPriority = 0; + for (const auto* collection : teamCollections) { + if (collection == nullptr || !collection->initialFailureReactionDelay.isReady()) { + return -1; + } + // Team health is updated independently of relocation admission and completion. Include both regions + // and conservatively keep counting degraded teams until their trackers are retired. + int collectionPriority = -1; + for (const auto& [priority, count] : collection->priority_teams) { + if (count > 0) { + collectionPriority = std::max(collectionPriority, priority); + } + } + if (collectionPriority < 0) { + return -1; + } + highestPriority = std::max(highestPriority, collectionPriority); + } + return highestPriority; +} + bool DDTeamCollection::satisfiesPolicy(const std::vector>& team, int amount) const { std::vector forcedEntries, resultEntries; if (amount == -1) { @@ -5337,7 +5363,7 @@ void DDTeamCollection::rebuildMachineLocalityMap() { for (auto& [_, machine] : machine_info) { if (machine->serversOnMachine.empty()) { TraceEvent(SevWarn, "RebuildMachineLocalityMapError") - .detail("Machine", machine->machineID.toString()) + .detail("MachineID", machine->machineID.toString()) .detail("NumServersOnMachine", 0); continue; } @@ -5348,7 +5374,7 @@ void DDTeamCollection::rebuildMachineLocalityMap() { auto& locality = representativeServer->getLastKnownInterface().locality; if (!isValidLocality(configuration.storagePolicy, locality)) { TraceEvent(SevWarn, "RebuildMachineLocalityMapError") - .detail("Machine", machine->machineID.toString()) + .detail("MachineID", machine->machineID.toString()) .detail("InvalidLocality", locality.toString()); continue; } @@ -6430,7 +6456,7 @@ bool DDTeamCollection::exclusionSafetyCheck(std::vector& excludeServerIDs) std::pair DDTeamCollection::getStorageWigglerState() const { if (storageWiggler) { - return { storageWiggler->getWiggleState(), storageWiggler->lastStateChangeTs }; + return storageWiggler->getWiggleStateSnapshot(); } return { StorageWiggler::INVALID, 0.0 }; } diff --git a/fdbserver/datadistributor/DDTeamCollection.h b/fdbserver/datadistributor/DDTeamCollection.h index 2627041e4de..6a1e933065d 100644 --- a/fdbserver/datadistributor/DDTeamCollection.h +++ b/fdbserver/datadistributor/DDTeamCollection.h @@ -54,7 +54,7 @@ class TCMachineInfo; class TCMachineTeamInfo; // All state that represents an ongoing tss pair recruitment -struct TSSPairState : ReferenceCounted, NonCopyable { +class TSSPairState : public ReferenceCounted, NonCopyable { Promise>> ssPairInfo; // if set, for ss to pass its id to tss pair once it is successfully recruited Promise tssPairDone; // if set, for tss to pass ss that it was successfully recruited @@ -65,11 +65,14 @@ struct TSSPairState : ReferenceCounted, NonCopyable { bool active; +public: TSSPairState() : active(false) {} explicit TSSPairState(const LocalityData& locality) : dcId(locality.dcId()), dataHallId(locality.dataHallId()), active(true) {} + bool isActive() const { return active; } + bool inDataZone(const LocalityData& locality) const { return locality.dcId() == dcId && locality.dataHallId() == dataHallId; } @@ -231,6 +234,8 @@ class DDTeamCollection : public ReferenceCounted { std::vector allServers; int64_t unhealthyServers; std::map priority_teams; + // Across all team collections; -1 until each collection has initialized its team health. + int getHighestTeamPriority() const; std::map> tss_info_by_pair; std::map> server_and_tss_info; // TODO could replace this with an efficient way to do a // read-only concatenation of 2 data structures? @@ -691,13 +696,12 @@ class DDTeamCollection : public ReferenceCounted { std::map, Reference> machine_info; std::vector> machineTeams; // all machine teams - // IMPORTANT: teams and teamsByServerIDs MUST be consistent, so any time we - // mutate teams, we must also mutate teamsByServerIDs +private: + // These must be updated together when adding or removing a team. std::vector> teams; - // O(1) hash map from server ID string to team information - // Currently used by getTeamByServers std::unordered_map> teamsByServerIDs; +public: std::vector teamCollections; AsyncTrigger printDetailedTeamsInfo; Reference storageServerSet; @@ -705,6 +709,7 @@ class DDTeamCollection : public ReferenceCounted { explicit DDTeamCollection(DDTeamCollectionInitParams const& params); ~DDTeamCollection(); + size_t teamCount() const { return teams.size(); } void addLaggingStorageServer(Key zoneId); diff --git a/fdbserver/datadistributor/DataDistribution.cpp b/fdbserver/datadistributor/DataDistribution.cpp index 65382d05335..7d3a2ae3234 100644 --- a/fdbserver/datadistributor/DataDistribution.cpp +++ b/fdbserver/datadistributor/DataDistribution.cpp @@ -43,6 +43,7 @@ #include "DDTeamCollection.h" #include "DataDistribution.h" #include "DDRelocationQueue.h" +#include "NativeCdcRetagCleanup.h" #include "fdbserver/core/Knobs.h" #include "fdbserver/core/MoveKeys.h" #include "fdbserver/core/QuietDatabase.h" @@ -137,7 +138,7 @@ enum class DDAuditContext : uint8_t { }; struct DDAudit { - explicit(false) DDAudit(AuditStorageState coreState) + explicit DDAudit(AuditStorageState coreState) : coreState(coreState), actors(true), foundError(false), auditStorageAnyChildFailed(false), retryCount(0), cancelled(false), overallCompleteDoAuditCount(0), overallIssuedDoAuditCount(0), overallSkippedDoAuditCount(0), remainingBudgetForAuditTasks(SERVER_KNOBS->CONCURRENT_AUDIT_TASK_COUNT_MAX), context(DDAuditContext::INVALID) {} @@ -714,6 +715,7 @@ struct DataDistributor : NonCopyable, ReferenceCounted { .detail("TotalBytes", 0) .detail("UnhealthyServers", 0) .detail("HighestPriority", 0) + .detail("HighestTeamPriority", -1) .trackLatest(self->totalDataInFlightEventHolder->trackingKey); TraceEvent("TotalDataInFlight", self->ddId) .detail("Primary", false) @@ -1111,34 +1113,31 @@ Future> triggerBulkLoadTask(Reference splitBulkLoadTask(Reference self, BulkLoadTaskState parent) { +// The children's ranges tile the parent's, so the caller must write both in a single transaction: then no +// version exists in which the parent's range is unowned, or owned by anything but tasks whose union is the +// parent -- that is the difference from erasing the task and relying on something to rebuild it, which drops +// the range's data if nothing does. +Optional> deriveSplitBulkLoadTasks(const BulkLoadTaskState& parent, UID logId) { std::vector manifests = parent.getManifests(); if (manifests.size() < 2) { // A single manifest is as narrow as a task gets, and a manifest can span an arbitrarily wide range, // so this is reachable with a range covering the whole key space. Nothing here can place it: the // cluster needs servers outside src, or the manifest needs to have been dumped more finely. - TraceEvent(SevWarnAlways, "DDBulkLoadTaskSplitDeclined", self->ddId) + TraceEvent(SevWarnAlways, "DDBulkLoadTaskSplitDeclined", logId) .detail("Reason", "Task holds a single manifest and cannot be narrowed") .detail("TaskRange", parent.getRange()) .detail("TaskID", parent.getTaskId()); - co_return false; + return {}; } std::sort(manifests.begin(), manifests.end(), [](BulkLoadManifest const& a, BulkLoadManifest const& b) { return a.getBeginKey() < b.getBeginKey(); @@ -1169,12 +1168,12 @@ Future splitBulkLoadTask(Reference self, BulkLoadTaskStat } if (insideBoundaries.empty()) { // Every manifest boundary is outside the parent's clipped range, so the range cannot be cut at one. - TraceEvent(SevWarnAlways, "DDBulkLoadTaskSplitDeclined", self->ddId) + TraceEvent(SevWarnAlways, "DDBulkLoadTaskSplitDeclined", logId) .detail("Reason", "No manifest split point lies inside the task's range") .detail("TaskRange", parent.getRange()) .detail("TaskID", parent.getTaskId()) .detail("ManifestCount", manifests.size()); - co_return false; + return {}; } int const half = insideBoundaries[insideBoundaries.size() / 2]; // The parent's range starts at or after its first manifest's begin key, so that key can never be @@ -1201,7 +1200,7 @@ Future splitBulkLoadTask(Reference self, BulkLoadTaskStat ASSERT(children[0].getRange().begin == parent.getRange().begin); ASSERT(children[0].getRange().end == children[1].getRange().begin); ASSERT(children[1].getRange().end == parent.getRange().end); - // The writes below must be issued in ascending key order, so keep the guard next to the reason. + // The caller must issue the writes in ascending key order, so keep the guard next to the reason. // krmSetRange reads oldValue at Snapshot::True on a plain Transaction, which has no read-your-writes, // so each call is blind to the previous one's mutations and only their order makes the result correct. // Each call emits clear(range); set(begin, value); set(end, oldValue). Ascending, the second call's @@ -1210,6 +1209,18 @@ Future splitBulkLoadTask(Reference self, BulkLoadTaskStat // lands last and republishes the parent over the second child's range -- precisely the state the // tiling comment above says cannot exist. ASSERT(children[0].getRange().begin < children[1].getRange().begin); + return children; +} + +// Install the derived children in place of the parent, in one transaction. Returns false without writing +// anything if the task cannot be narrowed, and also if the parent turns out to be no longer ours, which is +// not a statement about the range. +Future splitBulkLoadTask(Reference self, BulkLoadTaskState parent) { + Optional> derived = deriveSplitBulkLoadTasks(parent, self->ddId); + if (!derived.present()) { + co_return false; + } + std::vector const& children = derived.get(); Database cx = self->txnProcessor->context(); Transaction tr(cx); @@ -1230,7 +1241,7 @@ Future splitBulkLoadTask(Reference self, BulkLoadTaskStat .detail("CommitVersion", tr.getCommittedVersion()) .detail("TaskRange", parent.getRange()) .detail("TaskID", parent.getTaskId()) - .detail("ManifestCount", manifests.size()) + .detail("ManifestCount", parent.getManifests().size()) .detail("FirstRange", children[0].getRange()) .detail("FirstTaskID", children[0].getTaskId()) .detail("SecondRange", children[1].getRange()) @@ -3041,6 +3052,8 @@ Future bulkDumpCore(Reference self, Future readyToS } void addDataDistributionActors(Reference self, std::vector>& actors) { + actors.push_back(nativeCdcRetagCleanup( + self->txnProcessor->context(), self->lock, self->context->ddEnabledState.get(), self->initialized.getFuture())); if (bulkLoadIsEnabled(self->initData->bulkLoadMode)) { TraceEvent(SevInfo, "DDBulkLoadModeEnabled", self->ddId) .detail("UsableRegions", self->configuration.usableRegions); @@ -3790,7 +3803,7 @@ Future ddExclusionSafetyCheck(DistributorExclusionSafetyCheckRequest req, co_return; } // If there is only 1 team, unsafe to mark failed: team building can get stuck due to lack of servers left - if (self->teamCollection->teams.size() <= 1) { + if (self->teamCollection->teamCount() <= 1) { TraceEvent("DDExclusionSafetyCheckNotEnoughTeams", self->ddId).log(); reply.safe = false; req.reply.send(reply); @@ -5962,3 +5975,107 @@ TEST_CASE("/DataDistribution/Initialization/ResumeFromShard") { self->shardsAffectedByTeamFailure->check(); co_return; } + +namespace { + +// Only the key range matters to deriveSplitBulkLoadTasks(). The remaining fields are whatever satisfies +// BulkLoadManifest::isValid(), which the constructor asserts. +BulkLoadManifest splitTestManifest(KeyRef begin, KeyRef end) { + return BulkLoadManifest(BulkLoadFileSet("root", "relative", "0-manifest.txt", "0-data.sst", "", {}), + begin, + end, + /*version=*/1, + /*bytes=*/1, + /*keyCount=*/1, + BulkLoadByteSampleSetting(0, "hashlittle2", 250, 100, 0.5), + BulkLoadType::SST, + BulkLoadTransportMethod::CP); +} + +BulkLoadTaskState splitTestTask(const std::vector& manifestRanges, const KeyRange& taskRange) { + BulkLoadManifestSet set(manifestRanges.size()); + for (const auto& range : manifestRanges) { + ASSERT(set.addManifest(splitTestManifest(range.begin, range.end))); + } + return BulkLoadTaskState(deterministicRandom()->randomUniqueID(), set, taskRange); +} + +} // namespace + +TEST_CASE("/DataDistribution/BulkLoad/DeriveSplitTasks") { + // The children tile the parent and partition its manifests, and inherit its job. + { + auto parent = splitTestTask({ KeyRangeRef("a"_sr, "c"_sr), + KeyRangeRef("c"_sr, "e"_sr), + KeyRangeRef("e"_sr, "g"_sr), + KeyRangeRef("g"_sr, "i"_sr) }, + KeyRangeRef("a"_sr, "i"_sr)); + auto children = deriveSplitBulkLoadTasks(parent, UID()).get(); + ASSERT_EQ(children.size(), 2); + ASSERT(children[0].getRange().begin == parent.getRange().begin); + ASSERT(children[0].getRange().end == children[1].getRange().begin); + ASSERT(children[1].getRange().end == parent.getRange().end); + ASSERT_EQ(children[0].getManifests().size() + children[1].getManifests().size(), parent.getManifests().size()); + ASSERT(children[0].getJobId() == parent.getJobId()); + ASSERT(children[1].getJobId() == parent.getJobId()); + } + + // The cut falls on a manifest boundary, so neither child is handed a range whose data lives in the + // other's manifests. + { + auto parent = + splitTestTask({ KeyRangeRef("a"_sr, "c"_sr), KeyRangeRef("c"_sr, "e"_sr) }, KeyRangeRef("a"_sr, "e"_sr)); + auto children = deriveSplitBulkLoadTasks(parent, UID()).get(); + ASSERT(children[0].getRange() == KeyRangeRef("a"_sr, "c"_sr)); + ASSERT(children[1].getRange() == KeyRangeRef("c"_sr, "e"_sr)); + } + + // A single manifest is as narrow as a task gets. Declining is terminal for the task -- the caller marks + // it Error rather than re-dispatching it. + { + auto parent = splitTestTask({ KeyRangeRef("a"_sr, "z"_sr) }, KeyRangeRef("a"_sr, "z"_sr)); + ASSERT(!deriveSplitBulkLoadTasks(parent, UID()).present()); + } + + // REGRESSION: a task at a job-range edge holds manifests whose boundaries lie outside its clipped + // range. The cut must come from the boundaries strictly inside that range: here the manifest midpoint + // is "c", below the parent's own begin key. + { + auto parent = + splitTestTask({ KeyRangeRef("a"_sr, "c"_sr), KeyRangeRef("c"_sr, "e"_sr), KeyRangeRef("e"_sr, "g"_sr) }, + KeyRangeRef("d"_sr, "f"_sr)); + auto children = deriveSplitBulkLoadTasks(parent, UID()).get(); + ASSERT(children[0].getRange() == KeyRangeRef("d"_sr, "e"_sr)); + ASSERT(children[1].getRange() == KeyRangeRef("e"_sr, "f"_sr)); + } + + // Every boundary outside the clipped range leaves nothing to cut at, so the task declines rather than + // producing an empty child. + { + auto parent = + splitTestTask({ KeyRangeRef("a"_sr, "e"_sr), KeyRangeRef("e"_sr, "i"_sr) }, KeyRangeRef("f"_sr, "h"_sr)); + ASSERT(!deriveSplitBulkLoadTasks(parent, UID()).present()); + } + + // Halving terminates: each child holds strictly fewer manifests than its parent, so repeated splitting + // of the lower child reaches a single manifest and declines. + { + StringRef const boundaries = "abcdefghi"_sr; + std::vector ranges; + for (int i = 0; i + 1 < boundaries.size(); i++) { + ranges.push_back(KeyRangeRef(boundaries.substr(i, 1), boundaries.substr(i + 1, 1))); + } + BulkLoadTaskState task = splitTestTask(ranges, KeyRangeRef(ranges.front().begin, ranges.back().end)); + while (true) { + Optional> children = deriveSplitBulkLoadTasks(task, UID()); + if (!children.present()) { + break; + } + ASSERT_LT(children.get()[0].getManifests().size(), task.getManifests().size()); + task = children.get()[0]; + } + ASSERT_EQ(task.getManifests().size(), 1); + } + + return Void(); +} diff --git a/fdbserver/datadistributor/DataDistribution.h b/fdbserver/datadistributor/DataDistribution.h index 268e90d92c3..40345b12056 100644 --- a/fdbserver/datadistributor/DataDistribution.h +++ b/fdbserver/datadistributor/DataDistribution.h @@ -162,7 +162,7 @@ struct GetMetricsRequest { KeyRange keys; Promise reply; GetMetricsRequest() = default; - explicit(false) GetMetricsRequest(KeyRange const& keys) : keys(keys) {} + explicit GetMetricsRequest(KeyRange const& keys) : keys(keys) {} }; struct GetTopKMetricsReply { @@ -188,10 +188,10 @@ struct GetTopKMetricsRequest { double maxReadLoadPerKSecond = 0, minReadLoadPerKSecond = 0; // all returned shards won't exceed this read load GetTopKMetricsRequest() = default; - explicit(false) GetTopKMetricsRequest(std::vector const& keys, - int topK = 1, - double maxReadLoadPerKSecond = std::numeric_limits::max(), - double minReadLoadPerKSecond = 0) + explicit GetTopKMetricsRequest(std::vector const& keys, + int topK = 1, + double maxReadLoadPerKSecond = std::numeric_limits::max(), + double minReadLoadPerKSecond = 0) : topK(topK), keys(keys), maxReadLoadPerKSecond(maxReadLoadPerKSecond), minReadLoadPerKSecond(minReadLoadPerKSecond) { ASSERT_GE(topK, 1); @@ -252,7 +252,7 @@ FDB_BOOLEAN_PARAM(MoveKeyRangeOutPhysicalShard); class PhysicalShardCollection : public ReferenceCounted { public: PhysicalShardCollection() : lastTransitionStartTime(now()), requireTransition(false) {} - explicit(false) PhysicalShardCollection(Reference db) + explicit PhysicalShardCollection(Reference db) : txnProcessor(db), lastTransitionStartTime(now()), requireTransition(false) {} enum class PhysicalShardCreationTime { DDInit, DDRelocator }; @@ -604,7 +604,7 @@ inline bool bulkDumpIsEnabled(int bulkDumpModeValue) { class BulkLoadTaskCollection : public ReferenceCounted { public: - explicit(false) BulkLoadTaskCollection(UID ddId) : ddId(ddId) { + explicit BulkLoadTaskCollection(UID ddId) : ddId(ddId) { bulkLoadTaskMap.insert(allKeys, Optional()); } diff --git a/fdbserver/datadistributor/NativeCdcRetagCleanup.cpp b/fdbserver/datadistributor/NativeCdcRetagCleanup.cpp new file mode 100644 index 00000000000..3446ef207d4 --- /dev/null +++ b/fdbserver/datadistributor/NativeCdcRetagCleanup.cpp @@ -0,0 +1,207 @@ +/* + * NativeCdcRetagCleanup.cpp + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include +#include + +#include "NativeCdcRetagCleanup.h" +#include "fdbserver/core/NativeCdcMetadata.h" +#include "fdbclient/DatabaseContext.h" +#include "fdbclient/SystemData.h" +#include "fdbserver/core/Knobs.h" +#include "flow/CodeProbe.h" +#include "flow/Trace.h" +#include "flow/UnitTest.h" + +namespace { + +class NativeCdcCleanupProgress { + Optional assignmentChange; + Key next = cdcStreamKeys.begin; + bool cycleComplete = false; + bool sawPending = false; + bool rescan = false; + + bool sameGeneration(ValueRef generation) const { + return assignmentChange.present() && assignmentChange.get() == generation; + } + +public: + bool needsScan(ValueRef generation) const { + return !sameGeneration(generation) || !cycleComplete || sawPending || rescan; + } + + Key begin() const { return cycleComplete ? Key(cdcStreamKeys.begin) : next; } + + void scanned(Value generation, Key nextBegin, bool lastPage, bool pagePending) { + const bool continuing = assignmentChange.present() && !cycleComplete; + sawPending = (continuing && sawPending) || pagePending; + rescan = continuing && (rescan || !sameGeneration(generation)); + assignmentChange = std::move(generation); + next = std::move(nextBegin); + cycleComplete = lastPage; + } + + // Churn requires another traversal, not a restart that can starve later pages. + void changed() { rescan = true; } +}; + +class NativeCdcRetagCleanup { + Database cx; + MoveKeysLock lock; + const DDEnabledState* ddEnabledState; + NativeCdcCleanupProgress cleanupProgress; + + Future finishPendingPage() { + constexpr int cleanupPageSize = 100; + Transaction tr(cx); + while (true) { + Error err; + try { + tr.setOption(FDBTransactionOptions::LOCK_AWARE); + tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); + const Optional change = co_await tr.get(cdcProxyAssignmentChangeKey); + const Value generation = change.present() ? change.get() : Value(); + if (!cleanupProgress.needsScan(generation)) { + co_return false; + } + const Key begin = cleanupProgress.begin(); + const RangeResult page = co_await tr.getRange(KeyRangeRef(begin, cdcStreamKeys.end), cleanupPageSize); + std::vector>> reads; + for (const auto& row : page) { + reads.push_back(readNativeCdcTagState(&tr, decodeCDCStreamKey(row.key))); + } + const std::vector> states = co_await getAll(reads); + int finished = 0; + bool pagePending = false; + for (const auto& state : states) { + if (!state.present()) { + // Incomplete ownership/metadata is not evidence that all transitions have drained. + pagePending = true; + continue; + } + pagePending = pagePending || state.get().pending; + if (state.get().pending && (co_await finishNativeCdcRetag(&tr, state.get()))) { + ++finished; + } + } + if (finished == 0) { + cleanupProgress.scanned(generation, + page.more ? keyAfter(page.back().key) : Key(cdcStreamKeys.end), + !page.more, + pagePending); + co_return false; + } + co_await checkMoveKeysLock(&tr, lock, ddEnabledState); + co_await tr.commit(); + cleanupProgress.scanned(generation, + page.more ? keyAfter(page.back().key) : Key(cdcStreamKeys.end), + !page.more, + pagePending); + cleanupProgress.changed(); + CODE_PROBE(true, "Native CDC DD finishes acknowledged tag transitions"); + TraceEvent("NativeCdcTagTransitionsFinished", lock.myOwner).detail("Streams", finished); + co_return true; + } catch (Error& e) { + err = e; + } + co_await tr.onError(err); + } + } + +public: + NativeCdcRetagCleanup(Database cx, MoveKeysLock lock, const DDEnabledState* ddEnabledState) + : cx(cx), lock(lock), ddEnabledState(ddEnabledState) {} + + Future run(Future initialized) { + co_await initialized; + while (true) { + try { + co_await finishPendingPage(); + } catch (Error& e) { + if (e.code() == error_code_actor_cancelled || e.code() == error_code_broken_promise || + e.code() == error_code_movekeys_conflict) { + throw; + } + TraceEvent(SevWarn, "NativeCdcRetagCleanupError", lock.myOwner).error(e); + } + const double interval = SERVER_KNOBS->NATIVE_CDC_RETAG_CLEANUP_INTERVAL; + co_await delay(std::isfinite(interval) && interval > 0 ? interval : 30.0, TaskPriority::DataDistribution); + } + } +}; + +TEST_CASE("/NativeCDC/RetagCleanup/Paging") { + NativeCdcCleanupProgress progress; + const Value firstGeneration = "first"_sr; + const Value secondGeneration = "second"_sr; + const Key nextPage = keyAfter(cdcStreamKeyFor(100)); + ASSERT(progress.needsScan(firstGeneration)); + ASSERT_EQ(progress.begin(), cdcStreamKeys.begin); + progress.scanned(firstGeneration, nextPage, false, false); + ASSERT(progress.needsScan(firstGeneration)); + ASSERT_EQ(progress.begin(), nextPage); + progress.scanned(firstGeneration, cdcStreamKeys.end, true, true); + // Acknowledgements do not change the assignment generation, so pending cycles must repeat. + ASSERT(progress.needsScan(firstGeneration)); + ASSERT_EQ(progress.begin(), cdcStreamKeys.begin); + progress.scanned(firstGeneration, cdcStreamKeys.end, true, false); + ASSERT(!progress.needsScan(firstGeneration)); + ASSERT(progress.needsScan(secondGeneration)); + ASSERT_EQ(progress.begin(), cdcStreamKeys.begin); + progress.changed(); + ASSERT(progress.needsScan(firstGeneration)); + return Void(); +} + +TEST_CASE("/NativeCDC/RetagCleanup/ProgressAcrossChurn") { + NativeCdcCleanupProgress progress; + const Value firstGeneration = "first"_sr; + const Value secondGeneration = "second"_sr; + const Value thirdGeneration = "third"_sr; + const Key secondPage = keyAfter(cdcStreamKeyFor(100)); + const Key thirdPage = keyAfter(cdcStreamKeyFor(200)); + progress.scanned(firstGeneration, secondPage, false, true); + progress.changed(); // A successful cleanup on the first page changes the generation. + ASSERT(progress.needsScan(secondGeneration)); + ASSERT_EQ(progress.begin(), secondPage); + progress.scanned(secondGeneration, thirdPage, false, true); + progress.changed(); + ASSERT(progress.needsScan(thirdGeneration)); + ASSERT_EQ(progress.begin(), thirdPage); + progress.scanned(thirdGeneration, cdcStreamKeys.end, true, false); + ASSERT(progress.needsScan(thirdGeneration)); + ASSERT_EQ(progress.begin(), cdcStreamKeys.begin); + progress.scanned(thirdGeneration, cdcStreamKeys.end, true, false); + ASSERT(!progress.needsScan(thirdGeneration)); + return Void(); +} + +} // namespace + +Future nativeCdcRetagCleanup(Database cx, + MoveKeysLock lock, + const DDEnabledState* ddEnabledState, + Future initialized) { + NativeCdcRetagCleanup cleanup(cx, lock, ddEnabledState); + co_await cleanup.run(initialized); +} + +void forceLinkNativeCdcRetagCleanupTests() {} diff --git a/fdbserver/datadistributor/NativeCdcRetagCleanup.h b/fdbserver/datadistributor/NativeCdcRetagCleanup.h new file mode 100644 index 00000000000..16978ce7e1b --- /dev/null +++ b/fdbserver/datadistributor/NativeCdcRetagCleanup.h @@ -0,0 +1,33 @@ +/* + * NativeCdcRetagCleanup.h + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#pragma once + +#include "fdbclient/NativeAPI.h" +#include "fdbserver/core/MoveKeys.h" + +// The DD epoch fences finalization of acknowledged transitions. Cleanup remains +// active while CDC admission and production of new transitions are disabled. +Future nativeCdcRetagCleanup(Database cx, + MoveKeysLock lock, + const DDEnabledState* ddEnabledState, + Future initialized); + +void forceLinkNativeCdcRetagCleanupTests(); diff --git a/fdbserver/datadistributor/StorageWiggler.h b/fdbserver/datadistributor/StorageWiggler.h index df3a905a093..f1dbc7b1fe9 100644 --- a/fdbserver/datadistributor/StorageWiggler.h +++ b/fdbserver/datadistributor/StorageWiggler.h @@ -34,7 +34,8 @@ class DDTeamCollection; -struct StorageWiggler : ReferenceCounted { +class StorageWiggler : public ReferenceCounted { +public: static constexpr double MIN_ON_CHECK_DELAY_SEC = 5.0; using State = StorageWigglerState::Value; static constexpr State INVALID = StorageWigglerState::INVALID; @@ -46,6 +47,8 @@ struct StorageWiggler : ReferenceCounted { StorageWiggleMetrics metrics; AsyncVar stopWiggleSignal; + +private: // data structures using MetadataUIDP = std::pair; // min-heap @@ -56,6 +59,7 @@ struct StorageWiggler : ReferenceCounted { State wiggleState = INVALID; double lastStateChangeTs = 0.0; // timestamp describes when did the state change +public: explicit StorageWiggler(DDTeamCollection* collection) : teamCollection(collection), stopWiggleSignal(true) {}; // wiggle related actors will quit when this signal is set to true void setStopSignal(bool value) { stopWiggleSignal.set(value); } @@ -76,7 +80,7 @@ struct StorageWiggler : ReferenceCounted { Optional getNextServerId(bool necessaryOnly = true); // next check time to avoid busy loop Future onCheck() const; - State getWiggleState() const { return wiggleState; } + std::pair getWiggleStateSnapshot() const { return { wiggleState, lastStateChangeTs }; } void setWiggleState(State s) { if (wiggleState != s) { wiggleState = s; diff --git a/fdbserver/fdbserver.cpp b/fdbserver/fdbserver.cpp index 8e4f8f37de8..40c2f3995a1 100644 --- a/fdbserver/fdbserver.cpp +++ b/fdbserver/fdbserver.cpp @@ -59,7 +59,6 @@ #include "fdbserver/CoroFlow.h" #include "fdbserver/core/MoveKeys.h" #include "fdbserver/core/Knobs.h" -#include "fdbserver/NetworkTest.h" #include "fdbserver/kvstore/KVFileUtils.h" #include "fdbserver/core/ServerDBInfo.h" #include "fdbserver/core/FDBSimulationPolicy.h" @@ -130,7 +129,7 @@ enum { OPT_CONNFILE, OPT_SEEDCONNFILE, OPT_SEEDCONNSTRING, OPT_ROLE, OPT_LISTEN, OPT_PUBLICADDR, OPT_DATAFOLDER, OPT_TLOG_SPILL_DATAFOLDER, OPT_LOGFOLDER, OPT_PARENTPID, OPT_TRACER, OPT_NEWCONSOLE, OPT_NOBOX, OPT_TESTFILE, OPT_RESTARTING, OPT_RESTORING, OPT_RANDOMSEED, OPT_RESEED_TIME, OPT_KEY, OPT_MEMLIMIT, OPT_VMEMLIMIT, OPT_STORAGEMEMLIMIT, OPT_CACHEMEMLIMIT, OPT_MACHINEID, OPT_DCID, OPT_MACHINE_CLASS, OPT_BUGGIFY, OPT_VERSION, OPT_BUILD_FLAGS, OPT_CRASHONERROR, OPT_HELP, OPT_NETWORKIMPL, OPT_NOBUFSTDOUT, OPT_BUFSTDOUTERR, - OPT_TRACECLOCK, OPT_NUMTESTERS, OPT_DEVHELP, OPT_PRINT_CODE_PROBES, OPT_ROLLSIZE, OPT_MAXLOGS, OPT_MAXLOGSSIZE, OPT_KNOB, OPT_UNITTESTPARAM, OPT_TESTSERVERS, OPT_TEST_ON_SERVERS, OPT_METRICSCONNFILE, + OPT_TRACECLOCK, OPT_NUMTESTERS, OPT_DEVHELP, OPT_PRINT_CODE_PROBES, OPT_ROLLSIZE, OPT_MAXLOGS, OPT_MAXLOGSSIZE, OPT_KNOB, OPT_UNITTESTPARAM, OPT_TEST_ON_SERVERS, OPT_METRICSCONNFILE, OPT_METRICSPREFIX, OPT_LOGGROUP, OPT_LOCALITY, OPT_IO_TRUST_SECONDS, OPT_IO_TRUST_WARN_ONLY, OPT_FILESYSTEM, OPT_TLOG_SPILL_FILESYSTEM, OPT_PROFILER_RSS_SIZE, OPT_KVFILE, OPT_TRACE_FORMAT, OPT_WHITELIST_BINPATH, OPT_BLOB_CREDENTIALS, OPT_PROXY, OPT_DEPRECATED_CONFIG_PATH, OPT_DEPRECATED_USE_TEST_CONFIG_DB, OPT_DEPRECATED_NO_CONFIG_DB, OPT_FAULT_INJECTION, OPT_PROFILER, OPT_PRINT_SIMTIME, OPT_IP_TRUSTED_MASK, @@ -214,7 +213,6 @@ CSimpleOpt::SOption g_rgOptions[] = { { OPT_KNOB, "--knob-", SO_REQ_SEP }, { OPT_UNITTESTPARAM, "--test-", SO_REQ_SEP }, { OPT_LOCALITY, "--locality-", SO_REQ_SEP }, - { OPT_TESTSERVERS, "--testservers", SO_REQ_SEP }, { OPT_TEST_ON_SERVERS, "--testonservers", SO_NONE }, { OPT_METRICSCONNFILE, "--metrics-cluster", SO_REQ_SEP }, { OPT_METRICSPREFIX, "--metrics-prefix", SO_REQ_SEP }, @@ -608,7 +606,7 @@ static void printUsage(const char* name, bool devhelp) { printf(" --build-flags Print build information and exit.\n"); printOptionUsage("-r ROLE, --role ROLE", " Server role (valid options are fdbd, test, multitest," - " simulation, networktestclient, networktestserver, restore" + " simulation, restore" " consistencycheck, consistencycheckurgent, kvfileintegritycheck, kvfilegeneratesums, " "kvfiledump, mocks3server, unittests)." " The default is `fdbd'."); @@ -648,9 +646,6 @@ static void printUsage(const char* name, bool devhelp) { " the given threshold. fdbserver needs to be compiled with" " USE_GPERFTOOLS flag in order to use this feature."); #endif - printOptionUsage("--testservers ADDRESSES", - " The addresses of networktestservers" - " specified as ADDRESS:PORT,ADDRESS:PORT..."); printOptionUsage("--testonservers", " Testers are recruited on servers."); printOptionUsage("--metrics-cluster CONNFILE", " The cluster file designating where this process will" @@ -944,8 +939,6 @@ enum class ServerRole { KVFileDump, MockS3Server, MultiTester, - NetworkTestClient, - NetworkTestServer, Restore, SearchMutations, Simulation, @@ -972,7 +965,6 @@ struct CLIOptions { const char* testFile = "tests/default.txt"; std::string kvFile; - std::string testServersStr; std::string whitelistBinPaths; std::vector publicAddressStrs, listenAddressStrs, grpcAddressStrs; @@ -1227,10 +1219,6 @@ struct CLIOptions { role = ServerRole::VersionedMapTest; else if (!strcmp(sRole, "createtemplatedb")) role = ServerRole::CreateTemplateDatabase; - else if (!strcmp(sRole, "networktestclient")) - role = ServerRole::NetworkTestClient; - else if (!strcmp(sRole, "networktestserver")) - role = ServerRole::NetworkTestServer; else if (!strcmp(sRole, "kvfileintegritycheck")) role = ServerRole::KVFileIntegrityCheck; else if (!strcmp(sRole, "kvfilegeneratesums")) @@ -1547,9 +1535,6 @@ struct CLIOptions { case OPT_CRASHONERROR: g_crashOnError = true; break; - case OPT_TESTSERVERS: - testServersStr = args.OptionArg(); - break; case OPT_TEST_ON_SERVERS: testOnServers = true; break; @@ -1778,12 +1763,6 @@ struct CLIOptions { flushAndExit(FDB_EXIT_ERROR); } - if (role == ServerRole::NetworkTestClient && testServersStr.empty()) { - fprintf(stderr, "ERROR: please specify --testservers\n"); - printHelpTeaser(argv[0]); - flushAndExit(FDB_EXIT_ERROR); - } - if (role == ServerRole::ChangeClusterKey) { bool error = false; if (newClusterKey.empty()) { @@ -1985,8 +1964,7 @@ int main(int argc, char* argv[]) { FlowTransport::createInstance(false, 1, WLTOKEN_RESERVED_COUNT, &opts.allowList); opts.buildNetwork(argv[0]); - const bool expectsPublicAddress = - (role == ServerRole::FDBD || role == ServerRole::NetworkTestServer || role == ServerRole::MockS3Server); + const bool expectsPublicAddress = (role == ServerRole::FDBD || role == ServerRole::MockS3Server); if (opts.publicAddressStrs.empty()) { if (expectsPublicAddress) { fprintf(stderr, "ERROR: The -p or --public-address option is required\n"); @@ -2382,12 +2360,6 @@ int main(int argc, char* argv[]) { g_network->run(); } else if (role == ServerRole::CreateTemplateDatabase) { createTemplateDatabase(); - } else if (role == ServerRole::NetworkTestClient) { - f = stopAfter(networkTestClient(opts.testServersStr)); - g_network->run(); - } else if (role == ServerRole::NetworkTestServer) { - f = stopAfter(networkTestServer()); - g_network->run(); } else if (role == ServerRole::KVFileIntegrityCheck) { f = stopAfter(KVFileCheck(opts.kvFile, true)); g_network->run(); diff --git a/fdbserver/grvproxy/GrvProxyServer.cpp b/fdbserver/grvproxy/GrvProxyServer.cpp index d3b2775fe51..6b54b038a4e 100644 --- a/fdbserver/grvproxy/GrvProxyServer.cpp +++ b/fdbserver/grvproxy/GrvProxyServer.cpp @@ -411,13 +411,15 @@ Future getRate(UID myID, nextRequestTimer = Never(); bool detailed = now() - lastDetailedReply > SERVER_KNOBS->DETAILED_METRIC_UPDATE_RATE; + // Receive ratekeeper replies at socket priority so incoming GRVs cannot starve rate-lease updates. reply = brokenPromiseToNever( db->get().ratekeeper.get().getRateInfo.getReply(GetRateInfoRequest(myID, *inTransactionCount, *inBatchTransactionCount, proxyData->version, *transactionTagCounter, - detailed))); + detailed), + TaskPriority::ReadSocket)); transactionTagCounter->clear(); expectingDetailedReply = detailed; } else if (res.index() == 2) { @@ -713,9 +715,10 @@ Future getLiveCommittedVersion(std::vector spa double grvStart = now(); Optional debugID = getDebugID(debugIDs); Future replyFromMasterFuture; + // Receive master replies at socket priority so incoming GRVs cannot starve an already-arrived reply. replyFromMasterFuture = grvProxyData->master.getLiveCommittedVersion.getReply( GetRawCommittedVersionRequest(span.context, debugID, grvProxyData->ssVersionVectorCache.getMaxVersion()), - TaskPriority::GetLiveCommittedVersionReply); + TaskPriority::ReadSocket); if (!SERVER_KNOBS->ALWAYS_CAUSAL_READ_RISKY && !(flags & GetReadVersionRequest::FLAG_CAUSAL_READ_RISKY)) { co_await transformError(updateLastCommit(grvProxyData, debugID), broken_promise(), tlog_failed()); diff --git a/fdbserver/kvstore/CMakeLists.txt b/fdbserver/kvstore/CMakeLists.txt index 677fc092698..e3885a8240a 100644 --- a/fdbserver/kvstore/CMakeLists.txt +++ b/fdbserver/kvstore/CMakeLists.txt @@ -18,7 +18,7 @@ target_include_directories(fdbserver_kvstore ${CMAKE_CURRENT_SOURCE_DIR} ${CMAKE_SOURCE_DIR}/contrib/sqlite ${CMAKE_SOURCE_DIR}/fdbserver/include) -target_link_libraries(fdbserver_kvstore PUBLIC fdbserver_core sqlite) +target_link_libraries(fdbserver_kvstore PUBLIC fdbserver_core sqlite PRIVATE fdbserver_checkpoint) if(WITH_ROCKSDB) add_dependencies(fdbserver_kvstore rocksdb) diff --git a/fdbserver/kvstore/IPager.h b/fdbserver/kvstore/IPager.h index 8bd6b529f89..ad3f268217d 100644 --- a/fdbserver/kvstore/IPager.h +++ b/fdbserver/kvstore/IPager.h @@ -174,12 +174,12 @@ struct ArbitraryObject { onDestruct = nullptr; } +private: // ptr can be set to any arbitrary thing. If it is not null at destruct time then // onDestruct(ptr) will be called if onDestruct is not null. void* ptr = nullptr; void (*onDestruct)(void*) = nullptr; -private: // Call onDestruct(ptr) if needed but don't reset any state void destructOnly() { if (ptr != nullptr && onDestruct != nullptr) { diff --git a/fdbserver/kvstore/KeyValueStoreRocksDB.cpp b/fdbserver/kvstore/KeyValueStoreRocksDB.cpp index dd2e53cd160..c2a96e6bb14 100644 --- a/fdbserver/kvstore/KeyValueStoreRocksDB.cpp +++ b/fdbserver/kvstore/KeyValueStoreRocksDB.cpp @@ -68,7 +68,7 @@ #include "fdbserver/core/Knobs.h" #include "fdbserver/kvstore/IKeyValueStore.h" -#include "fdbserver/core/RocksDBCheckpointUtils.h" +#include "fdbserver/checkpoint/RocksDBCheckpointUtils.h" #include "RocksDBCommon.h" #include "flow/CoroUtils.h" diff --git a/fdbserver/kvstore/KeyValueStoreShardedRocksDB.cpp b/fdbserver/kvstore/KeyValueStoreShardedRocksDB.cpp index 5eec337f9a2..ee03399e270 100644 --- a/fdbserver/kvstore/KeyValueStoreShardedRocksDB.cpp +++ b/fdbserver/kvstore/KeyValueStoreShardedRocksDB.cpp @@ -44,7 +44,7 @@ #include "fdbserver/core/Knobs.h" #include "fdbserver/kvstore/IKeyValueStore.h" -#include "fdbserver/core/RocksDBCheckpointUtils.h" +#include "fdbserver/checkpoint/RocksDBCheckpointUtils.h" #include "RocksDBCommon.h" #ifdef WITH_ROCKSDB @@ -362,7 +362,7 @@ class CompactOnRangeDeletionCollectorFactory : public rocksdb::TablePropertiesCo // A factory of a table property collector that marks a SST file as need-compaction when the number of range // deletions exceeds the threshold. // @param numRangeDeletionsAllowed, triggers compaction range deletion count exceeds numRangeDeletionsAllowed. - explicit(false) CompactOnRangeDeletionCollectorFactory(uint64_t numRangeDeletionsAllowed) + explicit CompactOnRangeDeletionCollectorFactory(uint64_t numRangeDeletionsAllowed) : threshold(numRangeDeletionsAllowed), numFilesMarkedForCompaction(0) {} ~CompactOnRangeDeletionCollectorFactory() override = default; @@ -2515,7 +2515,7 @@ struct ShardedRocksDBKeyValueStore : IKeyValueStore { struct DeleteVisitor : public rocksdb::WriteBatch::Handler { std::vector>* deletes; - explicit(false) DeleteVisitor(std::vector>* deletes) : deletes(deletes) { + explicit DeleteVisitor(std::vector>* deletes) : deletes(deletes) { ASSERT(deletes); } diff --git a/fdbserver/kvstore/RadixTree.h b/fdbserver/kvstore/RadixTree.h index bb56c5d0edb..c2565675ca2 100644 --- a/fdbserver/kvstore/RadixTree.h +++ b/fdbserver/kvstore/RadixTree.h @@ -91,6 +91,9 @@ class radix_tree { node(const node&) = delete; // delete node& operator=(const node& other) { + if (this == &other) { + return *this; + } m_is_leaf = other.m_is_leaf; m_is_inline = other.m_is_inline; m_inline_length = other.m_inline_length; @@ -237,10 +240,7 @@ class radix_tree { iterator() : m_pointee(nullptr) {} iterator(const iterator& r) : m_pointee(r.m_pointee) {} explicit(false) iterator(node* p) : m_pointee(p) {} - iterator& operator=(const iterator& r) { - m_pointee = r.m_pointee; - return *this; - } + iterator& operator=(const iterator& r) = default; ~iterator() = default; const iterator& operator++(); diff --git a/fdbserver/kvstore/VFSAsync.cpp b/fdbserver/kvstore/VFSAsync.cpp index 5f7f5508bb9..103dfed5d7b 100644 --- a/fdbserver/kvstore/VFSAsync.cpp +++ b/fdbserver/kvstore/VFSAsync.cpp @@ -250,7 +250,7 @@ static int asyncLock(sqlite3_file* pFile, int eLock) { return eLock == EXCLUSIVE_LOCK ? SQLITE_BUSY : SQLITE_OK; } static int asyncUnlock(sqlite3_file* pFile, int eLock) { - assert(eLock <= SHARED_LOCK); + ASSERT_ABORT(eLock <= SHARED_LOCK); return SQLITE_OK; } @@ -347,7 +347,7 @@ static int asyncShmMap(sqlite3_file* fd, /* Handle open on database file */ ++memInfo->refcount; // printf("Shared memory for: '%s' (%d refs)\n", filename.c_str(), memInfo->refcount); } else { - assert(memInfo->regionSize == szRegion); + ASSERT_ABORT(memInfo->regionSize == szRegion); } if (iRegion >= memInfo->regions.size()) { @@ -380,11 +380,12 @@ static int asyncShmLock(sqlite3_file* fd, /* Database file holding the shared me int n, /* Number of locks to acquire or release */ int flags /* What to do with the lock */ ) { - assert(ofst >= 0 && ofst + n <= SQLITE_SHM_NLOCK); - assert(n >= 1); - assert(flags == (SQLITE_SHM_LOCK | SQLITE_SHM_SHARED) || flags == (SQLITE_SHM_LOCK | SQLITE_SHM_EXCLUSIVE) || - flags == (SQLITE_SHM_UNLOCK | SQLITE_SHM_SHARED) || flags == (SQLITE_SHM_UNLOCK | SQLITE_SHM_EXCLUSIVE)); - assert(n == 1 || (flags & SQLITE_SHM_EXCLUSIVE) != 0); + ASSERT_ABORT(ofst >= 0 && ofst + n <= SQLITE_SHM_NLOCK); + ASSERT_ABORT(n >= 1); + ASSERT_ABORT(flags == (SQLITE_SHM_LOCK | SQLITE_SHM_SHARED) || flags == (SQLITE_SHM_LOCK | SQLITE_SHM_EXCLUSIVE) || + flags == (SQLITE_SHM_UNLOCK | SQLITE_SHM_SHARED) || + flags == (SQLITE_SHM_UNLOCK | SQLITE_SHM_EXCLUSIVE)); + ASSERT_ABORT(n == 1 || (flags & SQLITE_SHM_EXCLUSIVE) != 0); MutexHolder hold(SharedMemoryInfo::mutex); @@ -617,9 +618,9 @@ static int asyncAccess(sqlite3_vfs* pVfs, const char* zPath, int flags, int* pRe int rc; /* access() return code */ int eAccess = F_OK; /* Second argument to access() */ - assert(flags == SQLITE_ACCESS_EXISTS /* access(zPath, F_OK) */ - || flags == SQLITE_ACCESS_READ /* access(zPath, R_OK) */ - || flags == SQLITE_ACCESS_READWRITE /* access(zPath, R_OK|W_OK) */ + ASSERT_ABORT(flags == SQLITE_ACCESS_EXISTS /* access(zPath, F_OK) */ + || flags == SQLITE_ACCESS_READ /* access(zPath, R_OK) */ + || flags == SQLITE_ACCESS_READWRITE /* access(zPath, R_OK|W_OK) */ ); if (flags == SQLITE_ACCESS_READWRITE) diff --git a/fdbserver/kvstore/VersionedBTree.cpp b/fdbserver/kvstore/VersionedBTree.cpp index 1a819a732d3..bffe295da01 100644 --- a/fdbserver/kvstore/VersionedBTree.cpp +++ b/fdbserver/kvstore/VersionedBTree.cpp @@ -1658,7 +1658,7 @@ class ObjectCache : NonCopyable { // must eventually give them back with moveIn() or remove them with reclaim(). class Evictor : NonCopyable { public: - explicit(false) Evictor(int64_t sizeLimit = 0) : sizeLimit(sizeLimit) {} + explicit Evictor(int64_t sizeLimit = 0) : sizeLimit(sizeLimit) {} // Evictors are normally singletons, either one per real process or one per virtual process in simulation static Evictor* getEvictor() { @@ -1791,7 +1791,7 @@ class ObjectCache : NonCopyable { int64_t movedOutCount = 0; }; - explicit(false) ObjectCache(Evictor* evictor = nullptr) : pEvictor(evictor) { + explicit ObjectCache(Evictor* evictor = nullptr) : pEvictor(evictor) { if (pEvictor == nullptr) { pEvictor = Evictor::getEvictor(); } @@ -2061,9 +2061,9 @@ class DWALPager final : public IPager2 { struct RemappedPage { enum Type { NONE = 'N', REMAP = 'R', FREE = 'F', DETACH = 'D' }; - explicit(false) RemappedPage(Version v = invalidVersion, - LogicalPageID o = invalidLogicalPageID, - LogicalPageID n = invalidLogicalPageID) + explicit RemappedPage(Version v = invalidVersion, + LogicalPageID o = invalidLogicalPageID, + LogicalPageID n = invalidLogicalPageID) : version(v), originalPageID(o), newPageID(n) {} Version version; @@ -5333,7 +5333,7 @@ class VersionedBTree { // Clear SingleKeyMutation() : op(MutationRef::ClearRange) {} // Set - explicit(false) SingleKeyMutation(Value val) : op(MutationRef::SetValue), value(val) {} + explicit SingleKeyMutation(Value val) : op(MutationRef::SetValue), value(val) {} // Atomic Op SingleKeyMutation(MutationRef::Type op, Value val) : op(op), value(val) {} @@ -11172,7 +11172,7 @@ struct KVSource { // TODO there is probably a better way to do this Prefix extraRangePrefix; - explicit(false) KVSource(const std::vector& desc, int numPrefixes = 0) : desc(desc) { + explicit KVSource(const std::vector& desc, int numPrefixes = 0) : desc(desc) { if (numPrefixes == 0) { numPrefixes = 1; for (auto& p : desc) { diff --git a/fdbserver/logsystem/ApplyMetadataMutation.cpp b/fdbserver/logsystem/ApplyMetadataMutation.cpp index ccc031bd3b2..71013155fef 100644 --- a/fdbserver/logsystem/ApplyMetadataMutation.cpp +++ b/fdbserver/logsystem/ApplyMetadataMutation.cpp @@ -573,9 +573,9 @@ class ApplyMetadataMutationsImpl { return; } if (cdcStreamKeys.contains(m.param1)) { - cdcRouting->setRange(decodeCDCStreamKey(m.param1), decodeCDCStreamKeysValue(m.param2)); + cdcRouting->setRanges(decodeCDCStreamKey(m.param1), decodeCDCStreamKeysValue(m.param2)); } else if (cdcTagHistoryKeys.contains(m.param1)) { - const CDCTagHistoryEntry history = decodeCDCTagHistoryKey(m.param1); + const CDCTagHistoryEntry history = decodeCDCTagHistoryEntry(m.param1, m.param2); cdcRouting->setTag(history.streamId, history.version, history.tag); } } diff --git a/fdbserver/logsystem/CDCRoutingTable.cpp b/fdbserver/logsystem/CDCRoutingTable.cpp index 240469352d6..9adf0a85b0f 100644 --- a/fdbserver/logsystem/CDCRoutingTable.cpp +++ b/fdbserver/logsystem/CDCRoutingTable.cpp @@ -28,8 +28,8 @@ CDCRoutingTable::CDCRoutingTable() { tagsByRange.insert(allKeys, std::set()); } -void CDCRoutingTable::updateRange(CDCStreamId streamId, KeyRangeRef const& keys) { - streams[streamId].keys = KeyRange(keys); +void CDCRoutingTable::updateRanges(CDCStreamId streamId, std::vector const& ranges) { + streams[streamId].ranges = ranges; } bool CDCRoutingTable::updateTag(CDCStreamId streamId, Version version, Tag tag) { @@ -45,18 +45,20 @@ bool CDCRoutingTable::updateTag(CDCStreamId streamId, Version version, Tag tag) void CDCRoutingTable::rebuildRanges() { tagsByRange.insert(allKeys, std::set()); for (const auto& [streamId, state] : streams) { - if (!state.keys.present() || !state.tag.present()) { + if (!state.tag.present()) { continue; } - for (auto range : tagsByRange.modify(state.keys.get())) { - range->value().insert(state.tag.get().second); + for (const auto& keys : state.ranges) { + for (auto range : tagsByRange.modify(keys)) { + range->value().insert(state.tag.get().second); + } } } tagsByRange.coalesce(allKeys); } -void CDCRoutingTable::setRange(CDCStreamId streamId, KeyRangeRef const& keys) { - updateRange(streamId, keys); +void CDCRoutingTable::setRanges(CDCStreamId streamId, std::vector const& ranges) { + updateRanges(streamId, ranges); rebuildRanges(); } @@ -70,11 +72,11 @@ void CDCRoutingTable::reload(IKeyValueStore* txnStateStore) { streams.clear(); const RangeResult streamRows = txnStateStore->readRange(cdcStreamKeys).get(); for (const auto& kv : streamRows) { - updateRange(decodeCDCStreamKey(kv.key), decodeCDCStreamKeysValue(kv.value)); + updateRanges(decodeCDCStreamKey(kv.key), decodeCDCStreamKeysValue(kv.value)); } const RangeResult tagHistoryRows = txnStateStore->readRange(cdcTagHistoryKeys).get(); for (const auto& kv : tagHistoryRows) { - const CDCTagHistoryEntry history = decodeCDCTagHistoryKey(kv.key); + const CDCTagHistoryEntry history = decodeCDCTagHistoryEntry(kv.key, kv.value); updateTag(history.streamId, history.version, history.tag); } rebuildRanges(); @@ -101,9 +103,9 @@ TEST_CASE("/NativeCDC/RoutingTable") { ASSERT(table.tagsForKey("b"_sr).empty()); ASSERT(table.tagsForRange(KeyRangeRef("b"_sr, "x"_sr)).empty()); - table.setRange(1, KeyRangeRef("a"_sr, "m"_sr)); + table.setRanges(1, { KeyRangeRef("a"_sr, "m"_sr) }); table.setTag(1, 100, ordersTag); - table.setRange(2, KeyRangeRef("g"_sr, "z"_sr)); + table.setRanges(2, { KeyRangeRef("g"_sr, "z"_sr) }); table.setTag(2, 100, overlappingTag); ASSERT_EQ(table.tagsForKey("b"_sr), std::set{ ordersTag }); @@ -125,14 +127,14 @@ TEST_CASE("/NativeCDC/RoutingTable/MetadataOrdering") { const Tag staleTag(tagLocalityCDC, 3); const Tag replacementTag(tagLocalityCDC, 4); - table.setRange(1, KeyRangeRef("a"_sr, "m"_sr)); + table.setRanges(1, { KeyRangeRef("a"_sr, "m"_sr) }); ASSERT(table.tagsForKey("b"_sr).empty()); table.setTag(2, 100, tagFirstTag); ASSERT(table.tagsForKey("n"_sr).empty()); table.setTag(1, 100, rangeFirstTag); - table.setRange(2, KeyRangeRef("m"_sr, "z"_sr)); + table.setRanges(2, { KeyRangeRef("m"_sr, "z"_sr) }); ASSERT_EQ(table.tagsForKey("b"_sr), std::set{ rangeFirstTag }); ASSERT_EQ(table.tagsForKey("n"_sr), std::set{ tagFirstTag }); @@ -149,18 +151,49 @@ TEST_CASE("/NativeCDC/RoutingTable/SharedTagRangeReplacement") { CDCRoutingTable table; const Tag sharedTag(tagLocalityCDC, 1); - table.setRange(1, KeyRangeRef("a"_sr, "m"_sr)); + table.setRanges(1, { KeyRangeRef("a"_sr, "m"_sr) }); table.setTag(1, 100, sharedTag); - table.setRange(2, KeyRangeRef("g"_sr, "z"_sr)); + table.setRanges(2, { KeyRangeRef("g"_sr, "z"_sr) }); table.setTag(2, 100, sharedTag); ASSERT_EQ(table.tagsForKey("h"_sr), std::set{ sharedTag }); ASSERT_EQ(table.tagsForRange(KeyRangeRef("a"_sr, "z"_sr)), std::set{ sharedTag }); - table.setRange(1, KeyRangeRef("n"_sr, "t"_sr)); + table.setRanges(1, { KeyRangeRef("n"_sr, "t"_sr) }); ASSERT(table.tagsForKey("b"_sr).empty()); ASSERT_EQ(table.tagsForKey("h"_sr), std::set{ sharedTag }); ASSERT_EQ(table.tagsForKey("p"_sr), std::set{ sharedTag }); return Void(); } + +TEST_CASE("/NativeCDC/RoutingTable/DisjointRanges") { + CDCRoutingTable table; + const Tag sharedTag(tagLocalityCDC, 1); + const Tag gapTag(tagLocalityCDC, 2); + const Tag rotatedTag(tagLocalityCDC, 3); + + table.setRanges(1, { KeyRangeRef("a"_sr, "c"_sr), KeyRangeRef("x"_sr, "z"_sr) }); + table.setTag(1, 100, sharedTag); + ASSERT_EQ(table.tagsForKey("a"_sr), std::set{ sharedTag }); + ASSERT_EQ(table.tagsForKey("x"_sr), std::set{ sharedTag }); + ASSERT(table.tagsForKey("c"_sr).empty()); + ASSERT(table.tagsForKey("m"_sr).empty()); + ASSERT(table.tagsForKey("z"_sr).empty()); + ASSERT(table.tagsForRange(KeyRangeRef("c"_sr, "x"_sr)).empty()); + ASSERT_EQ(table.tagsForRange(KeyRangeRef("b"_sr, "y"_sr)), std::set{ sharedTag }); + + table.setRanges(2, { KeyRangeRef("b"_sr, "d"_sr) }); + table.setTag(2, 100, sharedTag); + table.setRanges(3, { KeyRangeRef("m"_sr, "q"_sr) }); + table.setTag(3, 100, gapTag); + ASSERT_EQ(table.tagsForKey("b"_sr), std::set{ sharedTag }); + ASSERT_EQ(table.tagsForRange(KeyRangeRef("b"_sr, "y"_sr)), (std::set{ sharedTag, gapTag })); + + table.setTag(1, 200, rotatedTag); + ASSERT_EQ(table.tagsForKey("a"_sr), std::set{ rotatedTag }); + ASSERT_EQ(table.tagsForKey("b"_sr), (std::set{ sharedTag, rotatedTag })); + ASSERT_EQ(table.tagsForKey("x"_sr), std::set{ rotatedTag }); + ASSERT_EQ(table.tagsForRange(KeyRangeRef("b"_sr, "y"_sr)), (std::set{ sharedTag, gapTag, rotatedTag })); + return Void(); +} diff --git a/fdbserver/logsystem/LogSystem.cpp b/fdbserver/logsystem/LogSystem.cpp index 2d2ebdea049..ba5fb71d2c7 100644 --- a/fdbserver/logsystem/LogSystem.cpp +++ b/fdbserver/logsystem/LogSystem.cpp @@ -491,6 +491,11 @@ Reference LogSystem::fromOldLogSystemConfig(UID const& dbgid, } void LogSystem::purgeOldRecoveredGenerationsCoreState(DBCoreState& newState) { + // Anti-quorum recovery may still need these generations after a controller or router failure. + // Do not let incremental purging bypass the remote-prefix durability barrier. + if (!remoteLogPrefixRecovered()) { + return; + } Version oldestGenerationRecoverAtVersion = std::min(recoveredVersion->get(), remoteRecoveredVersion->get()); TraceEvent("ToCoreStateOldestGenerationRecoverAtVersion") .detail("RecoveredVersion", recoveredVersion->get()) @@ -537,6 +542,8 @@ void LogSystem::toCoreState(DBCoreState& newState) const { if (remoteRecoveryComplete.isValid() && remoteRecoveryComplete.isError()) throw remoteRecoveryComplete.getError(); + if (remoteLogPrefixComplete.isValid() && remoteLogPrefixComplete.isError()) + throw remoteLogPrefixComplete.getError(); newState.tLogs.clear(); newState.logRouterTags = logRouterTags; @@ -554,9 +561,7 @@ void LogSystem::toCoreState(DBCoreState& newState) const { } newState.oldTLogData.clear(); - if (!recoveryComplete.isValid() || !recoveryComplete.isReady() || - (repopulateRegionAntiQuorum == 0 && (!remoteRecoveryComplete.isValid() || !remoteRecoveryComplete.isReady())) || - epoch != oldestBackupEpoch) { + if (!storageRecovered() || !remoteLogPrefixRecovered()) { for (const auto& oldData : oldLogData) { newState.oldTLogData.push_back(toOldTLogCoreData(oldData)); TraceEvent("BWToCore") @@ -572,10 +577,23 @@ void LogSystem::toCoreState(DBCoreState& newState) const { newState.logSystemType = logSystemType; } +bool LogSystem::storageRecovered() const { + return recoveryComplete.isValid() && recoveryComplete.isReady() && !recoveryComplete.isError() && + (repopulateRegionAntiQuorum != 0 || (remoteStorageRecovered() && !remoteRecoveryComplete.isError())) && + epoch == oldestBackupEpoch; +} + bool LogSystem::remoteStorageRecovered() const { return remoteRecoveryComplete.isValid() && remoteRecoveryComplete.isReady(); } +bool LogSystem::remoteLogPrefixRecovered() const { + return repopulateRegionAntiQuorum == 0 || !hasRemoteServers || oldLogData.empty() || + (remoteStorageRecovered() && !remoteRecoveryComplete.isError()) || + (remoteLogPrefixComplete.isValid() && remoteLogPrefixComplete.isReady() && + !remoteLogPrefixComplete.isError()); +} + Future LogSystem::onCoreStateChanged() const { std::vector> changes; changes.push_back(Never()); @@ -588,6 +606,9 @@ Future LogSystem::onCoreStateChanged() const { if (remoteRecoveryComplete.isValid() && !remoteRecoveryComplete.isReady()) { changes.push_back(remoteRecoveryComplete); } + if (remoteLogPrefixComplete.isValid() && !remoteLogPrefixComplete.isReady()) { + changes.push_back(remoteLogPrefixComplete); + } changes.push_back(backupWorkerChanged.onTrigger()); // changes to oldestBackupEpoch changes.push_back(recoveredVersion->onChange()); changes.push_back(remoteRecoveredVersion->onChange()); @@ -596,6 +617,7 @@ Future LogSystem::onCoreStateChanged() const { void LogSystem::coreStateWritten(DBCoreState const& newState) { if (newState.oldTLogData.empty()) { + ASSERT(remoteLogPrefixRecovered()); recoveryCompleteWrittenToCoreState.set(true); } for (auto& t : newState.tLogs) { @@ -607,6 +629,87 @@ void LogSystem::coreStateWritten(DBCoreState const& newState) { } } +namespace { + +Future waitForRemoteTLogPrefixDurable(UID dbgid, + Reference>> tlog, + Version handoffVersion) { + TraceEvent("WaitForRemoteTLogPrefixDurable", dbgid) + .detail("TLogID", tlog->get().id()) + .detail("HandoffVersion", handoffVersion); + while (true) { + Future changed = tlog->onChange(); + if (!tlog->get().present()) { + co_await changed; + continue; + } + + auto result = co_await race(tlog->get().interf().getQueuingMetrics.getReplyUnlessFailedFor( + TLogQueuingMetricsRequest(), + SERVER_KNOBS->TLOG_TIMEOUT, + SERVER_KNOBS->MASTER_FAILURE_SLOPE_DURING_RECOVERY), + changed); + if (result.index() != 0 || changed.isReady()) { + continue; + } + ErrorOr reply = std::get<0>(result); + if (reply.isError()) { + if (reply.getError().code() == error_code_actor_cancelled) { + throw reply.getError(); + } + throw tlog_failed(); + } + + // Old-router pops advance from this remote log's durable acknowledgements. The current router's initial + // pop becomes visible only after the old prefix is consumed. Since the metric is durable known-committed + // progress and router pop positions are exclusive, require the handoff rather than its preceding version. + if (reply.get().v >= handoffVersion) { + TraceEvent("RemoteTLogPrefixDurable", dbgid) + .detail("TLogID", tlog->get().id()) + .detail("HandoffVersion", handoffVersion) + .detail("DurableKnownCommittedVersion", reply.get().v); + co_return; + } + co_await (delay(SERVER_KNOBS->METRIC_UPDATE_RATE) || changed); + } +} + +} // namespace + +Future LogSystem::onRemoteLogPrefixDurable() { + ASSERT(expectedLogSets > 0 && tLogs.size() == static_cast(expectedLogSets)); + if (!remoteLogPrefixComplete.isValid()) { + const Version handoffVersion = getMaxLocalStartVersion(tLogs); + std::vector> remotePrefixes; + if (!remoteLogPrefixRecovered()) { + for (const auto& logSet : tLogs) { + if (!logSet->isLocal && logSet->startVersion < handoffVersion) { + for (const auto& tlog : logSet->logServers) { + remotePrefixes.push_back(waitForRemoteTLogPrefixDurable(dbgid, tlog, handoffVersion)); + } + } + } + } + remoteLogPrefixComplete = waitForAll(remotePrefixes); + } + return remoteLogPrefixComplete; +} + +void LogSystem::retireOldLogRoles(DBCoreState const& finalState) { + ASSERT(finalState.recoveryCount == epoch); + ASSERT(finalState.oldTLogData.empty()); + ASSERT(expectedLogSets > 0 && finalState.tLogs.size() == static_cast(expectedLogSets)); + ASSERT(recoveryCompleteWrittenToCoreState.get()); + ASSERT(remoteLogPrefixRecovered()); + if (!oldLogRolesRetired) { + oldLogRolesRetired = true; + TraceEvent("RetireOldTLogRoles", dbgid) + .detail("RecoveryCount", epoch) + .detail("OldGenerations", oldLogData.size()); + logSystemConfigChanged.trigger(); + } +} + Future LogSystem::onError() const { // Never returns normally, but throws an error if the subsystem stops working while (true) { @@ -1026,7 +1129,7 @@ Future LogSystem::confirmEpochLive_internal(Reference logSet, Opti while (true) { for (int i = 0; i < alive.size(); i++) { if (!responded[i] && alive[i].isReady() && !alive[i].isError()) { - aliveEntries.push_back(logSet->logEntryArray[i]); + aliveEntries.push_back(logSet->getLogEntry(i)); responded[i] = true; } } @@ -1117,12 +1220,12 @@ LogSystemConfig LogSystem::getLogSystemConfig() const { } } - // ServerDBInfo uses oldTLogs to keep old-generation TLog roles from displacing - // themselves while this cluster controller is alive. Durable state/logsKey can - // drop recovered old generations earlier, but these roles may still be needed - // if this recovery has to run again. - for (const auto& oldData : oldLogData) { - logSystemConfig.oldTLogs.push_back(toOldTLogConf(oldData)); + // A storage-recovered state can still need old TLog locks and history for remote recruitment/catch-up. + // Keep those roles alive until the terminal recovery state is durable. + if (!oldLogRolesRetired) { + for (const auto& oldData : oldLogData) { + logSystemConfig.oldTLogs.push_back(toOldTLogConf(oldData)); + } } return logSystemConfig; } @@ -2097,6 +2200,40 @@ Future LogSystem::epochEnd(Reference>> outLo } } +namespace { + +Future initializeOldLogRouter(RequestStream worker, + InitializeLogRouterRequest request, + bool forRemote) { + Future firstReply = + transformErrors(throwErrorOr(worker.getReplyUnlessFailedFor( + request, SERVER_KNOBS->TLOG_TIMEOUT, SERVER_KNOBS->MASTER_FAILURE_SLOPE_DURING_RECOVERY)), + cluster_recovery_failed()); + if (forRemote) { + co_return co_await firstReply; + } + + auto first = co_await race(firstReply, delay(SERVER_KNOBS->CC_RERECRUIT_LOG_ROUTER_TIMEOUT)); + if (first.index() == 0) { + co_return std::get<0>(std::move(first)); + } + + // Reuse the live role if its initialization reply was lost. Keep the first reply valid, and let the + // transaction-system recovery monitor bound the wait after this single retry. + TraceEvent(SevWarn, "OldLogRouterInitializationRetry", request.reqId) + .detail("Locality", request.locality) + .detail("Tag", request.routerTag); + request.reply.reset(); + Future retryReply = + transformErrors(throwErrorOr(worker.getReplyUnlessFailedFor( + request, SERVER_KNOBS->TLOG_TIMEOUT, SERVER_KNOBS->MASTER_FAILURE_SLOPE_DURING_RECOVERY)), + cluster_recovery_failed()); + auto result = co_await race(firstReply, retryReply); + co_return result.index() == 0 ? std::get<0>(std::move(result)) : std::get<1>(std::move(result)); +} + +} // namespace + Future LogSystem::recruitOldLogRouters(std::vector workers, LogEpoch recoveryCount, int8_t locality, @@ -2157,10 +2294,7 @@ Future LogSystem::recruitOldLogRouters(std::vector worker req.knownLockedTLogIds = knownLockedTLogIds; req.allowDropInSim = SERVER_KNOBS->CC_RECOVERY_INIT_REQ_ALLOW_DROP_IN_SIM && !forRemote; req.isReplacement = false; - auto reply = transformErrors( - throwErrorOr(workers[nextRouter].logRouter.getReplyUnlessFailedFor( - req, SERVER_KNOBS->TLOG_TIMEOUT, SERVER_KNOBS->MASTER_FAILURE_SLOPE_DURING_RECOVERY)), - cluster_recovery_failed()); + auto reply = initializeOldLogRouter(workers[nextRouter].logRouter, req, forRemote); logRouterInitializationReplies.back().push_back(reply); allReplies.push_back(reply); nextRouter = (nextRouter + 1) % workers.size(); @@ -2213,10 +2347,7 @@ Future LogSystem::recruitOldLogRouters(std::vector worker req.recoverAt = old.recoverAt; req.allowDropInSim = SERVER_KNOBS->CC_RECOVERY_INIT_REQ_ALLOW_DROP_IN_SIM && !forRemote; req.isReplacement = false; - auto reply = transformErrors( - throwErrorOr(workers[nextRouter].logRouter.getReplyUnlessFailedFor( - req, SERVER_KNOBS->TLOG_TIMEOUT, SERVER_KNOBS->MASTER_FAILURE_SLOPE_DURING_RECOVERY)), - cluster_recovery_failed()); + auto reply = initializeOldLogRouter(workers[nextRouter].logRouter, req, forRemote); logRouterInitializationReplies.back().push_back(reply); allReplies.push_back(reply); nextRouter = (nextRouter + 1) % workers.size(); @@ -2494,6 +2625,7 @@ Future LogSystem::newRemoteEpoch(Reference oldLogSystem, remoteRecoveryComplete = waitForAll(recoveryComplete); remoteTrackTLogRecovery = LogSystem::trackTLogRecoveryActor(allRemoteTLogServers, remoteRecoveredVersion); tLogs.push_back(logSet); + onRemoteLogPrefixDurable(); TraceEvent("RemoteLogRecruitment_CompletingRecovery").log(); } diff --git a/fdbserver/logsystem/LogSystemPeekCursor.cpp b/fdbserver/logsystem/LogSystemPeekCursor.cpp index 2866bc353b8..46dc58d5aaf 100644 --- a/fdbserver/logsystem/LogSystemPeekCursor.cpp +++ b/fdbserver/logsystem/LogSystemPeekCursor.cpp @@ -887,7 +887,7 @@ void MergedPeekCursor::updateMessage(bool usePolicy) { locations.clear(); for (auto sortedVersion : versions) { - locations.push_back(logSet->logEntryArray[sortedVersion.second]); + locations.push_back(logSet->getLogEntry(sortedVersion.second)); if (locations.size() >= tLogReplicationFactor && logSet->satisfiesPolicy(locations)) { selectedVersion = sortedVersion.first; break; @@ -1016,7 +1016,12 @@ const LogMessageVersion& MergedPeekCursor::version() const { } Version MergedPeekCursor::getMinKnownCommittedVersion() const { - return serverCursors[currentCursor]->getMinKnownCommittedVersion(); + // A committed-version certificate is global, even when its replica has no next tagged message. + Version minKnownCommittedVersion = 0; + for (const auto& cursor : serverCursors) { + minKnownCommittedVersion = std::max(minKnownCommittedVersion, cursor->getMinKnownCommittedVersion()); + } + return minKnownCommittedVersion; } Version MergedPeekCursor::getMaxKnownVersion() const { @@ -1207,7 +1212,7 @@ void SetPeekCursor::updateMessage(int logIdx, bool usePolicy) { std::sort(versions.begin(), versions.end()); locations.clear(); for (auto sortedVersion : versions) { - locations.push_back(logSets[logIdx]->logEntryArray[sortedVersion.second]); + locations.push_back(logSets[logIdx]->getLogEntry(sortedVersion.second)); if (locations.size() >= logSets[logIdx]->tLogReplicationFactor && logSets[logIdx]->satisfiesPolicy(locations)) { selectedVersion = sortedVersion.first; @@ -1305,7 +1310,7 @@ Future setPeekGetMore(SetPeekCursor* self, LogMessageVersion startVersion, for (int i = 0; i < self->serverCursors[self->bestSet].size(); i++) { if (!self->serverCursors[self->bestSet][i]->isActive() && self->serverCursors[self->bestSet][i]->version() <= self->messageVersion) { - self->locations.push_back(self->logSets[self->bestSet]->logEntryArray[i]); + self->locations.push_back(self->logSets[self->bestSet]->getLogEntry(i)); } } bestSetValid = self->locations.size() < self->logSets[self->bestSet]->tLogReplicationFactor || @@ -1386,7 +1391,14 @@ const LogMessageVersion& SetPeekCursor::version() const { } Version SetPeekCursor::getMinKnownCommittedVersion() const { - return serverCursors[currentSet][currentCursor]->getMinKnownCommittedVersion(); + // Empty replies can certify progress independently of the current payload source. + Version minKnownCommittedVersion = 0; + for (const auto& cursors : serverCursors) { + for (const auto& cursor : cursors) { + minKnownCommittedVersion = std::max(minKnownCommittedVersion, cursor->getMinKnownCommittedVersion()); + } + } + return minKnownCommittedVersion; } Version SetPeekCursor::getMaxKnownVersion() const { @@ -1665,6 +1677,97 @@ TEST_CASE("/NativeCDC/ReplayPeekReplyAccounting") { return Void(); } +TEST_CASE("/NativeCDC/MergedPeekEmptyCommittedFrontier") { + auto makeServerCursor = []() { + return makeReference( + Reference>>(), Tag(tagLocalityCDC, 0), 0, 1000, false, false); + }; + std::vector> servers{ makeServerCursor(), makeServerCursor() }; + auto merged = makeReference( + servers, LogMessageVersion(0), 1, 2, Optional(), Reference(), 0); + ASSERT_EQ(merged->getMinKnownCommittedVersion(), 0); + + TLogPeekReply reply; + reply.end = 101; + reply.maxKnownVersion = 500; + reply.minKnownCommittedVersion = 100; + updateCursorWithReply(servers[1].getPtr(), reply); + merged->calcHasMessage(); + ASSERT(!merged->hasMessage()); + ASSERT_EQ(merged->version().version, 101); + ASSERT_EQ(merged->currentCursor, 0); + ASSERT_EQ(servers[0]->getMinKnownCommittedVersion(), 0); + ASSERT_EQ(merged->getMaxKnownVersion(), 500); + ASSERT_EQ(merged->getMinKnownCommittedVersion(), 100); + + // Another replica's stronger certificate does not advance the tagged read frontier. + reply.end = 151; + reply.maxKnownVersion = 800; + reply.minKnownCommittedVersion = 150; + updateCursorWithReply(servers[0].getPtr(), reply); + merged->calcHasMessage(); + ASSERT(!merged->hasMessage()); + ASSERT_EQ(merged->version().version, 101); + ASSERT_EQ(merged->currentCursor, 0); + ASSERT_EQ(merged->getMaxKnownVersion(), 800); + ASSERT_EQ(merged->getMinKnownCommittedVersion(), 150); + return Void(); +} + +TEST_CASE("/NativeCDC/SetPeekEmptyCommittedFrontier") { + std::vector> logSets; + std::vector>> servers(2); + for (int i = 0; i < servers.size(); ++i) { + auto logSet = makeReference(); + logSet->logServers.resize(2); + logSet->tLogReplicationFactor = 1; + logSet->tLogPolicy = makeReference(); + logSet->tLogLocalities.resize(2); + logSet->updateLocalitySet(logSet->tLogLocalities); + logSets.push_back(logSet); + for (int j = 0; j < 2; ++j) { + servers[i].push_back( + makeReference(Reference>>(), + Tag(tagLocalityCDC, 0), + 0, + 1000, + false, + false)); + } + } + auto merged = + makeReference(logSets, servers, LogMessageVersion(0), 0, 1, Optional(), true); + ASSERT_EQ(merged->getMinKnownCommittedVersion(), 0); + + TLogPeekReply reply; + reply.end = 101; + reply.maxKnownVersion = 500; + reply.minKnownCommittedVersion = 100; + updateCursorWithReply(servers[0][1].getPtr(), reply); + merged->calcHasMessage(); + ASSERT(!merged->hasMessage()); + ASSERT_EQ(merged->version().version, 101); + ASSERT_EQ(merged->currentSet, 0); + ASSERT_EQ(merged->currentCursor, 0); + ASSERT_EQ(servers[0][0]->getMinKnownCommittedVersion(), 0); + ASSERT_EQ(merged->getMaxKnownVersion(), 500); + ASSERT_EQ(merged->getMinKnownCommittedVersion(), 100); + + // A fallback set can know more is committed without changing the selected set's read frontier. + reply.end = 201; + reply.maxKnownVersion = 800; + reply.minKnownCommittedVersion = 200; + updateCursorWithReply(servers[1][1].getPtr(), reply); + merged->calcHasMessage(); + ASSERT(!merged->hasMessage()); + ASSERT_EQ(merged->version().version, 101); + ASSERT_EQ(merged->currentSet, 0); + ASSERT_EQ(merged->currentCursor, 0); + ASSERT_EQ(merged->getMaxKnownVersion(), 800); + ASSERT_EQ(merged->getMinKnownCommittedVersion(), 200); + return Void(); +} + TEST_CASE("/NativeCDC/ReplayPeekCommittedEpochBoundary") { auto makeServerCursor = [](Version committedVersion) { auto cursor = makeReference( diff --git a/fdbserver/logsystem/LogSystemRecoveryTests.cpp b/fdbserver/logsystem/LogSystemRecoveryTests.cpp index 70baf43d283..6fa107661a4 100644 --- a/fdbserver/logsystem/LogSystemRecoveryTests.cpp +++ b/fdbserver/logsystem/LogSystemRecoveryTests.cpp @@ -20,6 +20,8 @@ #include "fdbserver/logsystem/LogSystem.h" #include "fdbserver/logsystem/LogSystemConsumer.h" +#include "fdbserver/logsystem/LogSystemFactory.h" +#include "flow/CoroUtils.h" #include "flow/UnitTest.h" namespace { @@ -34,6 +36,171 @@ Reference makeSingleLogSet(const std::vector& tlogs, bool return logSet; } +Reference makeRemotePrefixLogSet(const std::vector& tlogs, + bool isLocal, + int8_t locality, + Version startVersion) { + auto logSet = makeSingleLogSet(tlogs, isLocal); + logSet->locality = locality; + logSet->startVersion = startVersion; + logSet->tLogVersion = TLogVersion::V6; + logSet->tLogReplicationFactor = 1; + logSet->tLogPolicy = makeReference(); + for (const auto& tlog : tlogs) { + logSet->tLogLocalities.push_back(tlog.filteredLocality); + } + return logSet; +} + +TLogInterface makeRemotePrefixRouterClient(TLogInterface router) { + // Streaming health checks and reply backpressure require transport-backed peek streams. + router.peekMessages = RequestStream(router.peekMessages.getEndpoint()); + router.peekStreamMessages = RequestStream(router.peekStreamMessages.getEndpoint()); + return router; +} + +Reference makeLaggingRemoteLogSystem(const std::vector& remoteLogs, + const TLogInterface& oldRouter, + const TLogInterface& currentRouter) { + constexpr LogEpoch epoch = 2; + LocalityData locality; + auto logSystem = makeReference(UID(), locality, epoch); + logSystem->logSystemType = LogSystemType::tagPartitioned; + logSystem->expectedLogSets = 2; + logSystem->oldestBackupEpoch = epoch; + logSystem->repopulateRegionAntiQuorum = 1; + logSystem->recoveryComplete = Void(); + logSystem->remoteRecovery = Void(); + logSystem->remoteRecoveryComplete = Never(); + logSystem->hasRemoteServers = true; + logSystem->logRouterTags = 1; + logSystem->tLogs.push_back(makeRemotePrefixLogSet({ TLogInterface(locality) }, true, 0, 100)); + auto remote = makeRemotePrefixLogSet(remoteLogs, false, 1, 60); + remote->logRouters.push_back(makeReference>>( + OptionalInterface(makeRemotePrefixRouterClient(currentRouter)))); + logSystem->tLogs.push_back(remote); + + OldLogData old; + old.epoch = epoch - 1; + old.epochBegin = 50; + old.epochEnd = 100; + old.recoverAt = 109; + old.logRouterTags = 1; + old.tLogs.push_back(makeRemotePrefixLogSet({ TLogInterface(locality) }, true, 0, 50)); + auto oldRemote = makeRemotePrefixLogSet({ TLogInterface(locality) }, false, 1, 60); + oldRemote->logRouters.push_back(makeReference>>( + OptionalInterface(makeRemotePrefixRouterClient(oldRouter)))); + old.tLogs.push_back(oldRemote); + logSystem->oldLogData.push_back(old); + // Make the ordinary generation-purge criteria eligible so the prefix barrier is the only retention gate. + logSystem->recoveredVersion->set(old.recoverAt + 1); + logSystem->remoteRecoveredVersion->set(old.recoverAt + 1); + return logSystem; +} + +DBCoreState makePendingRemotePrefixCoreState(const Reference& logSystem) { + DBCoreState pendingState; + logSystem->toCoreState(pendingState); + pendingState.recoveryCount = logSystem->epoch; + ASSERT(!pendingState.oldTLogData.empty()); + ASSERT_EQ(pendingState.oldTLogData.size(), logSystem->oldLogData.size()); + ASSERT_EQ(pendingState.tLogs.size(), logSystem->expectedLogSets); + const auto oldGenerations = pendingState.oldTLogData; + logSystem->purgeOldRecoveredGenerationsCoreState(pendingState); + ASSERT(pendingState.oldTLogData == oldGenerations); + logSystem->coreStateWritten(pendingState); + ASSERT(logSystem->remoteLogsWrittenToCoreState); + ASSERT(!logSystem->recoveryCompleteWrittenToCoreState.get()); + return pendingState; +} + +DBCoreState makeRecoveredRemotePrefixCoreState(const Reference& logSystem) { + DBCoreState finalState; + logSystem->toCoreState(finalState); + finalState.recoveryCount = logSystem->epoch; + ASSERT(finalState.oldTLogData.empty()); + ASSERT_EQ(finalState.tLogs.size(), logSystem->expectedLogSets); + logSystem->coreStateWritten(finalState); + ASSERT(logSystem->recoveryCompleteWrittenToCoreState.get()); + return finalState; +} + +void assertRemotePrefixCoreStateError(const Reference& logSystem, int errorCode) { + Optional error; + try { + DBCoreState state; + logSystem->toCoreState(state); + } catch (Error& e) { + error = e; + } + ASSERT(error.present()); + ASSERT_EQ(error.get().code(), errorCode); + ASSERT(!logSystem->recoveryCompleteWrittenToCoreState.get()); +} + +TLogQueuingMetricsReply makeRemotePrefixMetricsReply(Version version) { + TLogQueuingMetricsReply reply{}; + reply.localTime = now(); + reply.instanceID = 1; + reply.v = version; + return reply; +} + +Future serveRemotePrefixMetrics(TLogInterface tlog, + Reference> version, + PromiseStream reports) { + while (true) { + TLogQueuingMetricsRequest req = co_await tlog.getQueuingMetrics.getFuture(); + const Version reported = version->get(); + req.reply.send(makeRemotePrefixMetricsReply(reported)); + reports.send(reported); + } +} + +TLogPeekReply makeRemotePrefixPeekReply(Version begin, Optional requestedEnd, Version end) { + TLogPeekReply reply; + reply.begin = begin; + reply.end = std::min(end, requestedEnd.orDefault(end)); + ASSERT_LT(begin, reply.end); + reply.popped = begin; + reply.maxKnownVersion = reply.end - 1; + reply.minKnownCommittedVersion = reply.end - 1; + return reply; +} + +Future serveRemotePrefixRouter(TLogInterface router, + Version begin, + Version end, + PromiseStream requests) { + while (true) { + co_await Choose() + .When(router.peekMessages.getFuture(), + [&](const TLogPeekRequest& req) { + ASSERT_GE(req.begin, begin); + req.reply.send(makeRemotePrefixPeekReply(req.begin, req.end, end)); + requests.send(req.begin); + }) + .When(router.peekStreamMessages.getFuture(), + [&](const TLogPeekStreamRequest& req) { + ASSERT_GE(req.begin, begin); + req.reply.setByteLimit(req.limitBytes); + Future ready = req.reply.onReady(); + ASSERT(ready.isReady() && !ready.isError()); + req.reply.send(TLogPeekStreamReply(makeRemotePrefixPeekReply(req.begin, req.end, end))); + req.reply.sendError(end_of_stream()); + requests.send(req.begin); + }) + .run(); + } +} + +Future advanceRemotePrefixCursorTo(Reference cursor, Version end) { + while (cursor->version().version < end) { + co_await cursor->getMore(); + } + co_return; +} + std::tuple, bool> makeLogGroupResults( int replicationFactor, const std::vector>& perTLogUCV, @@ -58,6 +225,257 @@ std::tuple, bool> makeLogGroupResults( void forceLinkLogSystemRecoveryTests() {} +TEST_CASE("/LogSystem/RetireOldLogRoles/FinalCoreState") { + constexpr double timeoutSeconds = 30.0; + LocalityData locality; + TLogInterface currentRouter(locality); + auto logSystem = makeLaggingRemoteLogSystem({ TLogInterface(locality) }, TLogInterface(locality), currentRouter); + const LogEpoch epoch = logSystem->epoch; + logSystem->oldestBackupEpoch = epoch - 1; + Reference remoteSet = logSystem->tLogs.back(); + remoteSet->startVersion = 100; + logSystem->tLogs.pop_back(); + logSystem->remoteRecovery = Never(); + const auto oldLogSets = logSystem->oldLogData.front().tLogs; + const auto oldRoles = logSystem->getLogSystemConfig().oldTLogs; + + DBCoreState partialState; + logSystem->toCoreState(partialState); + partialState.recoveryCount = epoch; + ASSERT(!logSystem->storageRecovered()); + ASSERT_EQ(partialState.oldTLogData.size(), 1); + logSystem->coreStateWritten(partialState); + ASSERT(!logSystem->recoveryCompleteWrittenToCoreState.get()); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + + // Local recovery and backups can finish before the expected remote log set exists. + logSystem->oldestBackupEpoch = epoch; + logSystem->toCoreState(partialState); + ASSERT(logSystem->storageRecovered()); + ASSERT_EQ(partialState.oldTLogData.size(), 1); + ASSERT_EQ(partialState.tLogs.size(), 1); + logSystem->coreStateWritten(partialState); + ASSERT(!logSystem->recoveryCompleteWrittenToCoreState.get()); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + + logSystem->tLogs.push_back(remoteSet); + logSystem->remoteRecovery = Void(); + ASSERT(!logSystem->remoteRecoveryComplete.isReady()); + co_await timeoutError(logSystem->onRemoteLogPrefixDurable(), timeoutSeconds); + ASSERT(!logSystem->recoveryCompleteWrittenToCoreState.get()); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + const DBCoreState finalState = makeRecoveredRemotePrefixCoreState(logSystem); + LogSystemConfig expected = logSystem->getLogSystemConfig(); + ASSERT_EQ(expected.tLogs.size(), 2); + ASSERT(expected.oldTLogs == oldRoles); + expected.oldTLogs.clear(); + + Future configChanged = logSystem->onLogSystemConfigChange(); + ASSERT(!configChanged.isReady()); + logSystem->retireOldLogRoles(finalState); + ASSERT(configChanged.isReady() && !configChanged.isError()); + ASSERT(!logSystem->remoteRecoveryComplete.isReady()); + ASSERT(logSystem->getLogSystemConfig() == expected); + ASSERT_EQ(logSystem->oldLogData.size(), 1); + ASSERT(logSystem->oldLogData.front().tLogs == oldLogSets); + + Future unchanged = logSystem->onLogSystemConfigChange(); + logSystem->retireOldLogRoles(finalState); + ASSERT(!unchanged.isReady()); + ASSERT(logSystem->getLogSystemConfig() == expected); + + PromiseStream currentRequests; + Future currentRouterServer = serveRemotePrefixRouter(currentRouter, 100, 110, currentRequests); + auto currentConsumer = + makeLogSystemFromLogSystemConfig(UID(), locality, logSystem->getLogSystemConfig())->makeConsumer(); + auto currentCursor = currentConsumer->peek(UID(), 100, Optional(109), Tag(tagLocalityRemoteLog, 0), false); + co_await timeoutError(advanceRemotePrefixCursorTo(currentCursor, 110) || currentRouterServer, timeoutSeconds); + const Version currentBegin = co_await timeoutError(waitAndForward(currentRequests.getFuture()), timeoutSeconds); + ASSERT_EQ(currentBegin, 100); + ASSERT_EQ(currentCursor->version().version, 110); + co_return; +} + +TEST_CASE("/LogSystem/RemoteLogPrefix/TrackerInstallation") { + constexpr double timeoutSeconds = 30.0; + LocalityData locality; + TLogInterface remote(locality); + auto logSystem = makeLaggingRemoteLogSystem({ remote }, TLogInterface(locality), TLogInterface(locality)); + Reference remoteSet = logSystem->tLogs.back(); + logSystem->tLogs.pop_back(); + Promise remoteRecruitment; + logSystem->remoteRecovery = remoteRecruitment.getFuture(); + + DBCoreState partialState; + logSystem->toCoreState(partialState); + partialState.recoveryCount = logSystem->epoch; + ASSERT_EQ(partialState.tLogs.size(), 1); + ASSERT_EQ(partialState.oldTLogData.size(), 1); + logSystem->coreStateWritten(partialState); + Future beforeInstallation = logSystem->onCoreStateChanged(); + ASSERT(!beforeInstallation.isReady()); + + // The prefix tracker must be installed before remote recruitment reports completion. + logSystem->tLogs.push_back(remoteSet); + Future prefixDurable = logSystem->onRemoteLogPrefixDurable(); + TLogQueuingMetricsRequest initialRequest = + co_await timeoutError(waitAndForward(remote.getQueuingMetrics.getFuture()), timeoutSeconds); + initialRequest.reply.send(makeRemotePrefixMetricsReply(99)); + ASSERT(!prefixDurable.isReady()); + remoteRecruitment.send(Void()); + co_await timeoutError(beforeInstallation, timeoutSeconds); + + makePendingRemotePrefixCoreState(logSystem); + Future afterInstallation = logSystem->onCoreStateChanged(); + ASSERT(!afterInstallation.isReady()); + TLogQueuingMetricsRequest caughtUpRequest = + co_await timeoutError(waitAndForward(remote.getQueuingMetrics.getFuture()), timeoutSeconds); + ASSERT(!afterInstallation.isReady()); + caughtUpRequest.reply.send(makeRemotePrefixMetricsReply(100)); + co_await timeoutError(afterInstallation, timeoutSeconds); + co_await timeoutError(prefixDurable, timeoutSeconds); + ASSERT(!logSystem->remoteRecoveryComplete.isReady()); + makeRecoveredRemotePrefixCoreState(logSystem); + co_return; +} + +TEST_CASE("/LogSystem/RemoteLogPrefix/RemainsReadable") { + constexpr double timeoutSeconds = 30.0; + LocalityData locality; + TLogInterface remoteA(locality); + TLogInterface remoteB(locality); + TLogInterface oldRouter(locality); + TLogInterface currentRouter(locality); + auto versionA = makeReference>(100); + auto versionB = makeReference>(99); + PromiseStream reportsA; + PromiseStream reportsB; + PromiseStream oldRequests; + PromiseStream currentRequests; + Future mockActors = waitForAll(std::vector>{ + serveRemotePrefixMetrics(remoteA, versionA, reportsA), + serveRemotePrefixMetrics(remoteB, versionB, reportsB), + serveRemotePrefixRouter(oldRouter, 60, 100, oldRequests), + serveRemotePrefixRouter(currentRouter, 100, 110, currentRequests), + }); + auto logSystem = makeLaggingRemoteLogSystem({ remoteA, remoteB }, oldRouter, currentRouter); + const auto oldRoles = logSystem->getLogSystemConfig().oldTLogs; + ASSERT(logSystem->storageRecovered()); + const DBCoreState beforeTracking = makePendingRemotePrefixCoreState(logSystem); + Future prefixDurable = logSystem->onRemoteLogPrefixDurable(); + Future coreStateChanged = logSystem->onCoreStateChanged(); + Future configChanged = logSystem->onLogSystemConfigChange(); + const Version reportedA = co_await timeoutError(waitAndForward(reportsA.getFuture()), timeoutSeconds); + const Version reportedB = co_await timeoutError(waitAndForward(reportsB.getFuture()), timeoutSeconds); + ASSERT_EQ(reportedA, 100); + ASSERT_EQ(reportedB, 99); + ASSERT(!prefixDurable.isReady()); + ASSERT(!coreStateChanged.isReady()); + ASSERT(!configChanged.isReady()); + const DBCoreState pendingState = makePendingRemotePrefixCoreState(logSystem); + ASSERT(pendingState.oldTLogData == beforeTracking.oldTLogData); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + + // The real remote cursor must still reach the handoff through the old router. + auto oldConsumer = + makeLogSystemFromLogSystemConfig(UID(), locality, logSystem->getLogSystemConfig())->makeConsumer(); + auto oldCursor = oldConsumer->peek(UID(), 60, Optional(99), Tag(tagLocalityRemoteLog, 0), false); + co_await timeoutError(advanceRemotePrefixCursorTo(oldCursor, 100) || mockActors, timeoutSeconds); + const Version oldBegin = co_await timeoutError(waitAndForward(oldRequests.getFuture()), timeoutSeconds); + ASSERT_EQ(oldBegin, 60); + ASSERT_EQ(oldCursor->version().version, 100); + ASSERT(!prefixDurable.isReady()); + + versionB->set(100); + co_await timeoutError(prefixDurable || mockActors, timeoutSeconds); + co_await timeoutError(coreStateChanged || mockActors, timeoutSeconds); + ASSERT(!configChanged.isReady()); + ASSERT(!logSystem->remoteRecoveryComplete.isReady()); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + DBCoreState purgeCandidate = pendingState; + logSystem->purgeOldRecoveredGenerationsCoreState(purgeCandidate); + ASSERT(purgeCandidate.oldTLogData.empty()); + makeRecoveredRemotePrefixCoreState(logSystem); + ASSERT(!configChanged.isReady()); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + + auto currentConsumer = + makeLogSystemFromLogSystemConfig(UID(), locality, logSystem->getLogSystemConfig())->makeConsumer(); + auto currentCursor = currentConsumer->peek(UID(), 100, Optional(109), Tag(tagLocalityRemoteLog, 0), false); + co_await timeoutError(advanceRemotePrefixCursorTo(currentCursor, 110) || mockActors, timeoutSeconds); + const Version currentBegin = co_await timeoutError(waitAndForward(currentRequests.getFuture()), timeoutSeconds); + ASSERT_EQ(currentBegin, 100); + ASSERT_EQ(currentCursor->version().version, 110); + co_return; +} + +TEST_CASE("/LogSystem/RemoteLogPrefix/InterruptedWait") { + constexpr double timeoutSeconds = 30.0; + LocalityData locality; + { + TLogInterface remote(locality); + auto logSystem = makeLaggingRemoteLogSystem({ remote }, TLogInterface(locality), TLogInterface(locality)); + const auto oldRoles = logSystem->getLogSystemConfig().oldTLogs; + makePendingRemotePrefixCoreState(logSystem); + Future prefixDurable = logSystem->onRemoteLogPrefixDurable(); + Future coreStateChanged = logSystem->onCoreStateChanged(); + TLogQueuingMetricsRequest request = + co_await timeoutError(waitAndForward(remote.getQueuingMetrics.getFuture()), timeoutSeconds); + prefixDurable.cancel(); + ErrorOr result = co_await timeoutError(errorOr(prefixDurable), timeoutSeconds); + ASSERT(result.isError() && result.getError().code() == error_code_actor_cancelled); + ErrorOr changedResult = co_await timeoutError(errorOr(coreStateChanged), timeoutSeconds); + ASSERT(changedResult.isError() && changedResult.getError().code() == error_code_actor_cancelled); + request.reply.send(makeRemotePrefixMetricsReply(100)); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + assertRemotePrefixCoreStateError(logSystem, error_code_actor_cancelled); + } + { + TLogInterface remote(locality); + auto logSystem = makeLaggingRemoteLogSystem({ remote }, TLogInterface(locality), TLogInterface(locality)); + const auto oldRoles = logSystem->getLogSystemConfig().oldTLogs; + makePendingRemotePrefixCoreState(logSystem); + Future prefixDurable = logSystem->onRemoteLogPrefixDurable(); + Future coreStateChanged = logSystem->onCoreStateChanged(); + TLogQueuingMetricsRequest request = + co_await timeoutError(waitAndForward(remote.getQueuingMetrics.getFuture()), timeoutSeconds); + request.reply.sendError(operation_failed()); + ErrorOr result = co_await timeoutError(errorOr(prefixDurable), timeoutSeconds); + ASSERT(result.isError() && result.getError().code() == error_code_tlog_failed); + ErrorOr changedResult = co_await timeoutError(errorOr(coreStateChanged), timeoutSeconds); + ASSERT(changedResult.isError() && changedResult.getError().code() == error_code_tlog_failed); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + assertRemotePrefixCoreStateError(logSystem, error_code_tlog_failed); + } + + TLogInterface original(locality); + TLogInterface replacement(original.id(), original.getSharedTLogID(), locality); + auto logSystem = makeLaggingRemoteLogSystem({ original }, TLogInterface(locality), TLogInterface(locality)); + const auto oldRoles = logSystem->getLogSystemConfig().oldTLogs; + makePendingRemotePrefixCoreState(logSystem); + Future prefixDurable = logSystem->onRemoteLogPrefixDurable(); + TLogQueuingMetricsRequest staleRequest = + co_await timeoutError(waitAndForward(original.getQueuingMetrics.getFuture()), timeoutSeconds); + logSystem->tLogs[1]->logServers[0]->setUnconditional(OptionalInterface(replacement)); + TLogQueuingMetricsRequest replacementRequest = + co_await timeoutError(waitAndForward(replacement.getQueuingMetrics.getFuture()), timeoutSeconds); + staleRequest.reply.send(makeRemotePrefixMetricsReply(100)); + replacementRequest.reply.send(makeRemotePrefixMetricsReply(99)); + ASSERT(!prefixDurable.isReady()); + makePendingRemotePrefixCoreState(logSystem); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + + TLogQueuingMetricsRequest caughtUpRequest = + co_await timeoutError(waitAndForward(replacement.getQueuingMetrics.getFuture()), timeoutSeconds); + caughtUpRequest.reply.send(makeRemotePrefixMetricsReply(100)); + co_await timeoutError(prefixDurable, timeoutSeconds); + ASSERT(!logSystem->remoteRecoveryComplete.isReady()); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + makeRecoveredRemotePrefixCoreState(logSystem); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + co_return; +} + TEST_CASE("/LogSystem/GetPseudoPopTag/LogRouterWithoutMappedLocality") { LocalityData locality; auto logSystem = makeReference(UID(), locality, LogEpoch(1)); diff --git a/fdbserver/logsystem/include/fdbserver/logsystem/CDCRoutingTable.h b/fdbserver/logsystem/include/fdbserver/logsystem/CDCRoutingTable.h index 6fd8be3f19e..2d9dbb90685 100644 --- a/fdbserver/logsystem/include/fdbserver/logsystem/CDCRoutingTable.h +++ b/fdbserver/logsystem/include/fdbserver/logsystem/CDCRoutingTable.h @@ -23,6 +23,7 @@ #include #include #include +#include #include "fdbclient/FDBTypes.h" #include "fdbclient/KeyRangeMap.h" @@ -32,20 +33,20 @@ class IKeyValueStore; // Active CDC write routing reconstructed from durable stream and tag-history metadata. class CDCRoutingTable : NonCopyable { struct StreamState { - Optional keys; + std::vector ranges; Optional> tag; }; std::unordered_map streams; KeyRangeMap> tagsByRange; - void updateRange(CDCStreamId streamId, KeyRangeRef const& keys); + void updateRanges(CDCStreamId streamId, std::vector const& ranges); bool updateTag(CDCStreamId streamId, Version version, Tag tag); void rebuildRanges(); public: CDCRoutingTable(); - void setRange(CDCStreamId streamId, KeyRangeRef const& keys); + void setRanges(CDCStreamId streamId, std::vector const& ranges); void setTag(CDCStreamId streamId, Version version, Tag tag); void reload(IKeyValueStore* txnStateStore); bool empty() const { return streams.empty(); } diff --git a/fdbserver/logsystem/include/fdbserver/logsystem/LogSet.h b/fdbserver/logsystem/include/fdbserver/logsystem/LogSet.h index d0dc5b14ada..490005b519b 100644 --- a/fdbserver/logsystem/include/fdbserver/logsystem/LogSet.h +++ b/fdbserver/logsystem/include/fdbserver/logsystem/LogSet.h @@ -57,8 +57,13 @@ class LogSet : NonCopyable, public ReferenceCounted { TLogVersion tLogVersion; Reference tLogPolicy; Reference logServerSet; + +private: + // The locality map stores pointers to these indices; only updateLocalitySet may resize them. std::vector logIndexArray; std::vector logEntryArray; + +public: bool isLocal; int8_t locality; Version startVersion; @@ -83,6 +88,7 @@ class LogSet : NonCopyable, public ReferenceCounted { void checkSatelliteTagLocations(); int bestLocationFor(Tag tag); void updateLocalitySet(std::vector const& localities); + LocalityEntry getLogEntry(int location) const { return logEntryArray[location]; } bool satisfiesPolicy(const std::vector& locations); void getPushLocations( VectorRef tags, diff --git a/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h b/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h index 85c54965abf..b031e45117b 100644 --- a/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h +++ b/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h @@ -368,7 +368,13 @@ struct LogSystem : ReferenceCounted { // Convert LogSystem to DBCoreState and override input newState as return value void toCoreState(DBCoreState& newState) const; + // The storage/backup recovery policy can be satisfied before remote logs have copied their old generations' data. + // This permits STORAGE_RECOVERED under anti-quorum, but does not make old log history safe to discard. + bool storageRecovered() const; bool remoteStorageRecovered() const; + // Waits for durable remote TLog progress through the current local start version, not remote storage recovery. + // Requires every expected current log set. The returned future is shared by this recovery. + Future onRemoteLogPrefixDurable(); // Removes no-longer-needed older TLog generations from the outgoing core state. void purgeOldRecoveredGenerationsCoreState(DBCoreState&); @@ -377,6 +383,10 @@ struct LogSystem : ReferenceCounted { void coreStateWritten(DBCoreState const& newState); + // Requires the successfully committed terminal recovery state with every expected current log set. + // Finalizing a partial state for a coordinator change is insufficient. + void retireOldLogRoles(DBCoreState const& finalState); + Future onError() const; static Future pushResetChecker(Reference self, NetworkAddress addr); @@ -547,6 +557,11 @@ struct LogSystem : ReferenceCounted { static Future lockTLog(UID myID, Reference>> tlog); template static std::vector getReadyNonError(std::vector> const& futures); + +private: + bool remoteLogPrefixRecovered() const; + Future remoteLogPrefixComplete; + bool oldLogRolesRetired = false; }; // Recovery version calculation for version vector unicast diff --git a/fdbserver/mocks3/MockS3Server.cpp b/fdbserver/mocks3/MockS3Server.cpp index bc1b30c25f1..15a245b98c3 100644 --- a/fdbserver/mocks3/MockS3Server.cpp +++ b/fdbserver/mocks3/MockS3Server.cpp @@ -1590,11 +1590,7 @@ Future registerMockS3Server_impl(std::string ip, std::string port) { .detail("Address", serverKey) .detail("TotalRegistered", registeredServers.size()); } catch (Error& e) { - TraceEvent(SevError, "MockS3ServerRegistrationFailed") - .error(e) - .detail("Address", serverKey) - .detail("ErrorCode", e.code()) - .detail("ErrorName", e.name()); + TraceEvent(SevError, "MockS3ServerRegistrationFailed").error(e).detail("Address", serverKey); throw; } } diff --git a/fdbserver/mocks3/MockS3ServerChaos.cpp b/fdbserver/mocks3/MockS3ServerChaos.cpp index 1eaac2c86d3..62826783bfd 100644 --- a/fdbserver/mocks3/MockS3ServerChaos.cpp +++ b/fdbserver/mocks3/MockS3ServerChaos.cpp @@ -342,11 +342,7 @@ Future registerMockS3ChaosServer(std::string ip, std::string port) { .detail("TotalRegistered", registeredMockS3ChaosServers().size()); } catch (Error& e) { - TraceEvent(SevError, "MockS3ChaosServerRegistrationFailed") - .error(e) - .detail("Address", serverKey) - .detail("ErrorCode", e.code()) - .detail("ErrorName", e.name()); + TraceEvent(SevError, "MockS3ChaosServerRegistrationFailed").error(e).detail("Address", serverKey); throw; } } diff --git a/fdbserver/ratekeeper/Ratekeeper.cpp b/fdbserver/ratekeeper/Ratekeeper.cpp index 8254dcfb2b5..eae4dbd374f 100644 --- a/fdbserver/ratekeeper/Ratekeeper.cpp +++ b/fdbserver/ratekeeper/Ratekeeper.cpp @@ -292,12 +292,18 @@ Future Ratekeeper::monitorHotShards(Reference const } UID ssi = ssHighWriteQueue.get(); + auto interface = storageServerInterfaces.find(ssi); + // The selected server may have left since the last rate update. + if (interface == storageServerInterfaces.end()) { + CODE_PROBE(true, "Hot shard storage server removed before monitoring"); + continue; + } SetThrottledShardRequest setReq; // TraceEvent(SevDebug, "SendGetHotShardsRequest"); try { GetHotShardsRequest getReq; - GetHotShardsReply reply = co_await storageServerInterfaces[ssi].getHotShards.getReply(getReq); + GetHotShardsReply reply = co_await interface->second.getHotShards.getReply(getReq); // Backup's restore range can't be throttled, otherwise restore would fail, // i.e., "ApplyMutationsError". diff --git a/fdbserver/ratekeeper/RkTagThrottleCollection.cpp b/fdbserver/ratekeeper/RkTagThrottleCollection.cpp index 27022ef68fe..4ed89fcf217 100644 --- a/fdbserver/ratekeeper/RkTagThrottleCollection.cpp +++ b/fdbserver/ratekeeper/RkTagThrottleCollection.cpp @@ -54,19 +54,6 @@ Optional RkTagThrottleCollection::RkTagThrottleData::updateAndGetClientR } } -RkTagThrottleCollection::RkTagThrottleCollection(RkTagThrottleCollection&& other) { - autoThrottledTags = std::move(other.autoThrottledTags); - manualThrottledTags = std::move(other.manualThrottledTags); - tagData = std::move(other.tagData); -} - -RkTagThrottleCollection& RkTagThrottleCollection::RkTagThrottleCollection::operator=(RkTagThrottleCollection&& other) { - autoThrottledTags = std::move(other.autoThrottledTags); - manualThrottledTags = std::move(other.manualThrottledTags); - tagData = std::move(other.tagData); - return *this; -} - double RkTagThrottleCollection::computeTargetTpsRate(double currentBusyness, double targetBusyness, double requestRate) { @@ -250,7 +237,6 @@ PrioritizedTransactionTagMap RkTagThrottleCollection::g if (manualItr->second.empty()) { CODE_PROBE(true, "All manual throttles expired"); manualThrottledTags.erase(manualItr); - break; } } diff --git a/fdbserver/ratekeeper/RkTagThrottleCollection.h b/fdbserver/ratekeeper/RkTagThrottleCollection.h index 803fd249cd7..d20267c1ca3 100644 --- a/fdbserver/ratekeeper/RkTagThrottleCollection.h +++ b/fdbserver/ratekeeper/RkTagThrottleCollection.h @@ -55,8 +55,8 @@ class RkTagThrottleCollection : NonCopyable { public: RkTagThrottleCollection() = default; - RkTagThrottleCollection(RkTagThrottleCollection&& other); - RkTagThrottleCollection& operator=(RkTagThrottleCollection&& other); + RkTagThrottleCollection(RkTagThrottleCollection&& other) = default; + RkTagThrottleCollection& operator=(RkTagThrottleCollection&& other) = default; Optional autoThrottleTag(UID id, TransactionTag const& tag, diff --git a/fdbserver/sequencer/ResolutionBalancer.h b/fdbserver/sequencer/ResolutionBalancer.h index 12a3a4e25ae..bfff88fe5f2 100644 --- a/fdbserver/sequencer/ResolutionBalancer.h +++ b/fdbserver/sequencer/ResolutionBalancer.h @@ -41,7 +41,7 @@ struct ResolutionBalancer { std::vector resolvers; AsyncTrigger triggerResolution; - explicit(false) ResolutionBalancer(Version* version) : pVersion(version) {} + explicit ResolutionBalancer(Version* version) : pVersion(version) {} Future resolutionBalancing(); diff --git a/fdbserver/storageserver/CMakeLists.txt b/fdbserver/storageserver/CMakeLists.txt index 9a7d8c76d36..056ff505dcc 100644 --- a/fdbserver/storageserver/CMakeLists.txt +++ b/fdbserver/storageserver/CMakeLists.txt @@ -16,4 +16,4 @@ target_include_directories(fdbserver_storageserver ${CMAKE_CURRENT_SOURCE_DIR}/include PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}) -target_link_libraries(fdbserver_storageserver PRIVATE fdbserver_core fdbserver_kvstore fdbserver_logsystem) +target_link_libraries(fdbserver_storageserver PRIVATE fdbserver_core fdbserver_checkpoint fdbserver_kvstore fdbserver_logsystem) diff --git a/fdbserver/storageserver/storageserver.cpp b/fdbserver/storageserver/storageserver.cpp index cf57b94ae26..4a4fe2cf192 100644 --- a/fdbserver/storageserver/storageserver.cpp +++ b/fdbserver/storageserver/storageserver.cpp @@ -42,6 +42,8 @@ #include "fdbserver/core/AccumulativeChecksumUtil.h" #include "fdbserver/core/BulkDumpUtil.h" #include "fdbserver/core/BulkLoadUtil.h" +#include "fdbserver/checkpoint/BulkSstFiles.h" +#include "fdbserver/checkpoint/Checkpoint.h" #include "fdbserver/core/FDBSimulationPolicy.h" #include "fdbserver/core/FDBRocksDBVersion.h" #include "fdbserver/kvstore/IKeyValueStore.h" @@ -55,8 +57,7 @@ #include "MappedKeyPlan.h" #include "ReadLatencySamples.h" #include "fdbserver/core/RecoveryState.h" -#include "fdbserver/core/RocksDBCheckpointUtils.h" -#include "fdbserver/core/ServerCheckpoint.h" +#include "fdbserver/checkpoint/RocksDBCheckpointUtils.h" #include "fdbserver/core/SpanContextMessage.h" #include "fdbserver/storageserver/StorageCorruptionBug.h" #include "fdbserver/core/StorageMetrics.h" @@ -884,12 +885,6 @@ struct StorageServer : public IStorageMetricsService { VersionedData versionedData; std::map> mutationLog; // versions (durableVersion, version] - using WatchMapKey = Key; - using WatchMapKeyHasher = boost::hash; - using WatchMapValue = Reference; - using WatchMap_t = std::unordered_map; - WatchMap_t watchMap; // keep track of server watches - public: struct PendingNewShard { PendingNewShard(uint64_t shardId, KeyRangeRef range) : shardId(format("%016llx", shardId)), range(range) {} @@ -1194,6 +1189,8 @@ struct StorageServer : public IStorageMetricsService { Reference const> db; Database cx; + // counters must be declared before every member that can own an actor (actors, watchMap, …): cancelling those + // actors runs CountedSection destructors that touch these counters, so counters must outlive them struct Counters : CommonStorageCounters { Counter allQueries, systemKeyQueries, getKeyQueries, getValueQueries, getRangeQueries, getRangeSystemKeyQueries, @@ -1349,6 +1346,14 @@ struct StorageServer : public IStorageMetricsService { } } counters; +private: + using WatchMapKey = Key; + using WatchMapKeyHasher = boost::hash; + using WatchMapValue = Reference; + using WatchMap_t = std::unordered_map; + WatchMap_t watchMap; // keep track of server watches + +public: class GetValueQuery { public: GetValueQuery(GetValueRequest request, Counters& counters) @@ -11214,9 +11219,10 @@ Future updateStorage(StorageServer* data) { ++data->counters.kvCommits; recentCommitStats.back().seqId = data->counters.kvCommits.getValue(); - // If the mutation bytes budget was not fully used then wait some time before the next commit - durableDelay = - (bytesLeft > 0) ? delay(SERVER_KNOBS->STORAGE_COMMIT_INTERVAL, TaskPriority::UpdateStorage) : Void(); + // Batch only while both budgets have capacity; otherwise keep draining pending mutations. + durableDelay = (bytesLeft > 0 && clearRangesLeft > 0) + ? delay(SERVER_KNOBS->STORAGE_COMMIT_INTERVAL, TaskPriority::UpdateStorage) + : Void(); recentCommitStats.back().whenCommit = now(); try { diff --git a/fdbserver/tester/DatabaseMaintenance.cpp b/fdbserver/tester/DatabaseMaintenance.cpp index fe957260870..357081eaf6b 100644 --- a/fdbserver/tester/DatabaseMaintenance.cpp +++ b/fdbserver/tester/DatabaseMaintenance.cpp @@ -67,7 +67,7 @@ Future clearData(Database cx) { break; } catch (Error& e) { TraceEvent(SevWarn, "TesterClearingDatabaseError", tr.trState->readOptions.get().debugID.get()).error(e); - TraceEvent("ClearData_Loop1_Catch").detail("Phase", "Loop1_Error").detail("ErrorCode", e.code()); + TraceEvent("ClearData_Loop1_Catch").error(e).detail("Phase", "Loop1_Error"); err = e; } @@ -109,7 +109,7 @@ Future clearData(Database cx) { } catch (Error& e) { TraceEvent(SevWarn, "TesterCheckDatabaseClearedError", tr.trState->readOptions.get().debugID.get()) .error(e); - TraceEvent("ClearData_Loop2_Catch").detail("Phase", "Loop2_Error").detail("ErrorCode", e.code()); + TraceEvent("ClearData_Loop2_Catch").error(e).detail("Phase", "Loop2_Error"); caughtError = e; needsErrorHandling = true; } diff --git a/fdbserver/tester/include/fdbserver/tester/workloads.h b/fdbserver/tester/include/fdbserver/tester/workloads.h index ca227251b08..d275e573cea 100644 --- a/fdbserver/tester/include/fdbserver/tester/workloads.h +++ b/fdbserver/tester/include/fdbserver/tester/workloads.h @@ -108,7 +108,7 @@ struct TestWorkloadImpl : Workload { static_assert(std::is_same_v, "Workload must not override TestWorkload::description"); - explicit(false) TestWorkloadImpl(WorkloadContext const& wcx) : Workload(wcx) {} + explicit TestWorkloadImpl(WorkloadContext const& wcx) : Workload(wcx) {} template requires(E) TestWorkloadImpl(WorkloadContext const& wcx, NoOptions o) : Workload(wcx, o) {} @@ -120,7 +120,7 @@ struct CompoundWorkload; class DeterministicRandom; struct FailureInjectionWorkload : TestWorkload { - explicit(false) FailureInjectionWorkload(WorkloadContext const&); + explicit FailureInjectionWorkload(WorkloadContext const&); ~FailureInjectionWorkload() override = default; virtual void initFailureInjectionMode(DeterministicRandom& random); virtual bool shouldInject(DeterministicRandom& random, const WorkloadRequest& work, const unsigned count) const; @@ -154,7 +154,7 @@ struct CompoundWorkload : TestWorkload { std::vector> workloads; std::vector> failureInjection; - explicit(false) CompoundWorkload(WorkloadContext& wcx); + explicit CompoundWorkload(WorkloadContext& wcx); CompoundWorkload* add(Reference&& w); void addFailureInjection(WorkloadRequest& work); bool shouldInjectFailure(DeterministicRandom& random, @@ -254,7 +254,7 @@ struct WorkloadFactory : IWorkloadFactory { "Each workload must have a Workload::NAME member"); using WorkloadType = TestWorkloadImpl; bool runInUntrustedClient; - explicit(false) WorkloadFactory(UntrustedMode runInUntrustedClient = UntrustedMode::False) + explicit WorkloadFactory(UntrustedMode runInUntrustedClient = UntrustedMode::False) : runInUntrustedClient(runInUntrustedClient) { auto& f = factories(); std::string name = WorkloadType::NAME; diff --git a/fdbserver/tlog/TLogServer.cpp b/fdbserver/tlog/TLogServer.cpp index 8c65a7be306..782aab81c9b 100644 --- a/fdbserver/tlog/TLogServer.cpp +++ b/fdbserver/tlog/TLogServer.cpp @@ -58,6 +58,30 @@ FDB_BOOLEAN_PARAM(NothingPersistent); FDB_BOOLEAN_PARAM(PoppedRecently); FDB_BOOLEAN_PARAM(UnpoppedRecovered); +// Commit progress can advance without appending another message to this log. Keep its wakeup separate from +// LogData::version, and do not allocate a new notification on foreground commits when no reader is waiting. +class CommittedVersion : NonCopyable { +public: + Version get() const { return value; } + Future onAdvance() const { return changed.getFuture(); } + + void advance(Version version) { + if (version <= value) { + return; + } + value = version; + if (changed.getFutureReferenceCount() > 0) { + Promise notification; + notification.swap(changed); + notification.send(Void()); + } + } + +private: + Version value = 0; + Promise changed; +}; + } // namespace struct TLogQueueEntryRef { @@ -560,7 +584,7 @@ struct LogData : NonCopyable, public ReferenceCounted { Version knownCommittedVersion; // The maximum version that a proxy has told us that is committed (all TLogs have // ack'd a commit for this version). Version durableKnownCommittedVersion; - Version minKnownCommittedVersion; + CommittedVersion minKnownCommittedVersion; Version queuePoppedVersion; // The disk queue has been popped up until the location which represents this version. Version minPoppedTagVersion; Tag minPoppedTag; // The tag (locality >= 0) that makes tLog hold its data and cause tLog's disk queue increasing. @@ -707,15 +731,15 @@ struct LogData : NonCopyable, public ReferenceCounted { std::vector tags, std::string context) : initialized(false), queueCommittingVersion(0), knownCommittedVersion(0), durableKnownCommittedVersion(0), - minKnownCommittedVersion(0), queuePoppedVersion(0), minPoppedTagVersion(0), minPoppedTag(invalidTag), - minSysPopTagVersion(0), minSysPopTag(invalidTag), unpoppedRecoveredTagCount(0), - cc("TLog", interf.id().toString()), bytesInput("BytesInput", cc), tagMessageCount("tagMessageCount", cc), - bytesDurable("BytesDurable", cc), blockingPeeks("BlockingPeeks", cc), - blockingPeekTimeouts("BlockingPeekTimeouts", cc), emptyPeeks("EmptyPeeks", cc), - nonEmptyPeeks("NonEmptyPeeks", cc), persistentDataUpdateBatches("PersistentDataUpdateBatches", cc), - dirtyTagsProcessed("DirtyTagsProcessed", cc), logId(interf.id()), protocolVersion(protocolVersion), - newPersistentDataVersion(invalidVersion), tLogData(tLogData), unrecoveredBefore(1), recoveredAt(1), - recoveryTxnVersion(1), logSystem(new AsyncVar>()), + queuePoppedVersion(0), minPoppedTagVersion(0), minPoppedTag(invalidTag), minSysPopTagVersion(0), + minSysPopTag(invalidTag), unpoppedRecoveredTagCount(0), cc("TLog", interf.id().toString()), + bytesInput("BytesInput", cc), tagMessageCount("tagMessageCount", cc), bytesDurable("BytesDurable", cc), + blockingPeeks("BlockingPeeks", cc), blockingPeekTimeouts("BlockingPeekTimeouts", cc), + emptyPeeks("EmptyPeeks", cc), nonEmptyPeeks("NonEmptyPeeks", cc), + persistentDataUpdateBatches("PersistentDataUpdateBatches", cc), dirtyTagsProcessed("DirtyTagsProcessed", cc), + logId(interf.id()), protocolVersion(protocolVersion), newPersistentDataVersion(invalidVersion), + tLogData(tLogData), unrecoveredBefore(1), recoveredAt(1), recoveryTxnVersion(1), + logSystem(new AsyncVar>()), logSystemConsumer(new AsyncVar>()), remoteTag(remoteTag), isPrimary(isPrimary), logRouterTags(logRouterTags), logRouterPoppedVersion(0), logRouterPopToVersion(0), locality(tagLocalityInvalid), recruitmentID(recruitmentID), logSpillType(logSpillType), allTags(tags.begin(), tags.end()), @@ -1899,6 +1923,26 @@ Future waitForMessagesForTag(Reference self, Tag reqTag, Version } } +Future waitForCommittedVersion(Reference self, Version begin, double deadline) { + if (self->stopped() || self->minKnownCommittedVersion.get() >= begin) { + co_return; + } + ++self->blockingPeeks; + Future expired = delay(std::max(0.0, deadline - now()), TaskPriority::TLogPeekReply); + while (!self->stopped() && self->minKnownCommittedVersion.get() < begin) { + // No suspension separates the level check from registration. A canceled peek removes its callback; + // no threshold entry remains waiting for some future commit. + auto ready = + co_await race(self->minKnownCommittedVersion.onAdvance(), expired, self->stoppedPromise.getFuture()); + // Promise notification is synchronous. Never scan/materialize a reply on the tLogCommit or stop stack. + co_await delay(0, TaskPriority::TLogPeekReply); + if (ready.index() == 1) { + ++self->blockingPeekTimeouts; + co_return; + } + } +} + void peekMessagesFromMemory(Reference self, Tag tag, Version begin, @@ -2275,6 +2319,21 @@ Future tLogPeekMessages(PromiseType replyPromise, co_await delay(0, TaskPriority::TLogSpilledPeekReply); } + // Capped CDC peeks feed a proxy that can only consume the committed prefix; uncapped peeks and other + // tag consumers retain their existing behavior. Nonblocking requests must return without waiting, and + // spill-only reads must drain persisted data without waiting for live commits. A stopped TLog cannot + // advance its frontier, and finite recovery/tag-history ranges must drain without requiring new commits. + // Only an unbounded live-tail peek can use commit progress to wake this wait. + const bool waitForCommittedFrontier = replyByteLimit > 0 && reqTag.locality == tagLocalityCDC && + !reqReturnIfBlocked && !reqOnlySpilled && !logData->stopped() && + (!reqEnd.present() || reqEnd.get() == std::numeric_limits::max()); + if (waitForCommittedFrontier && poppedVersion(logData, reqTag) <= reqBegin) { + // A speculative message (or an empty tail) cannot advance native CDC until this frontier reaches begin. + // Waiting here lets commit progress wake the same capped peek instead of returning a stale frontier to + // the proxy. + co_await waitForCommittedVersion(logData, reqBegin, now() + SERVER_KNOBS->BLOCKING_PEEK_TIMEOUT); + } + double workStart = now(); Version poppedVer{ 0 }; Version endVersion{ 0 }; @@ -2305,7 +2364,7 @@ Future tLogPeekMessages(PromiseType replyPromise, auto tagData = logData->getTagData(reqTag); bool tagRecovered = tagData && !tagData->unpoppedRecovered; - if (SERVER_KNOBS->ENABLE_VERSION_VECTOR && poppedVer <= reqBegin && + if (!waitForCommittedFrontier && SERVER_KNOBS->ENABLE_VERSION_VECTOR && poppedVer <= reqBegin && reqBegin > logData->persistentDataDurableVersion && !reqOnlySpilled && (reqTag.locality >= 0 || reqTag.locality == tagLocalityCDC) && !reqReturnIfBlocked && tagRecovered) { double startTime = now(); @@ -2330,7 +2389,7 @@ Future tLogPeekMessages(PromiseType replyPromise, if (poppedVer > reqBegin) { TLogPeekReply rep; rep.maxKnownVersion = logData->version.get(); - rep.minKnownCommittedVersion = logData->minKnownCommittedVersion; + rep.minKnownCommittedVersion = logData->minKnownCommittedVersion.get(); rep.popped = poppedVer; rep.end = poppedVer; rep.onlySpilled = false; @@ -2545,7 +2604,8 @@ Future tLogPeekMessages(PromiseType replyPromise, // - Have data return to the caller, or // - Batching empty peek is disabled, or // - Batching empty peek interval has been reached. - if (replyByteLimitReached || messages.getLength() > 0 || !SERVER_KNOBS->PEEK_BATCHING_EMPTY_MSG || + if (waitForCommittedFrontier || replyByteLimitReached || messages.getLength() > 0 || + !SERVER_KNOBS->PEEK_BATCHING_EMPTY_MSG || (now() - blockStart > SERVER_KNOBS->PEEK_BATCHING_EMPTY_MSG_INTERVAL)) { break; } @@ -2577,7 +2637,7 @@ Future tLogPeekMessages(PromiseType replyPromise, TLogPeekReply reply; reply.maxKnownVersion = logData->version.get(); - reply.minKnownCommittedVersion = logData->minKnownCommittedVersion; + reply.minKnownCommittedVersion = logData->minKnownCommittedVersion.get(); auto messagesValue = messages.toValue(); reply.arena.dependsOn(messagesValue.arena()); reply.messages = messagesValue; @@ -2852,7 +2912,7 @@ Future tLogCommit(TLogData* self, req.spanContext.spanID); } - logData->minKnownCommittedVersion = std::max(logData->minKnownCommittedVersion, req.minKnownCommittedVersion); + logData->minKnownCommittedVersion.advance(req.minKnownCommittedVersion); co_await logData->version.whenAtLeast(req.prevVersion); // Time until now has been spent waiting in the queue to do actual work. @@ -3212,7 +3272,7 @@ void getQueuingMetrics(TLogData* self, Reference logData, TLogQueuingMe reply.bytesInput = self->bytesInput; reply.bytesDurable = self->bytesDurable; reply.storageBytes = self->persistentData->getStorageBytes(); - // FIXME: Add the knownCommittedVersion to this message and change ratekeeper to use that version. + // FIXME: Add a separate knownCommittedVersion field for ratekeeper; v must remain durable for recovery. reply.v = logData->durableKnownCommittedVersion; req.reply.send(reply); } @@ -3588,8 +3648,7 @@ Future pullAsyncData(TLogData* self, if (poppedIsKnownCommitted) { logData->knownCommittedVersion = std::max(logData->knownCommittedVersion, r->popped()); - logData->minKnownCommittedVersion = - std::max(logData->minKnownCommittedVersion, r->getMinKnownCommittedVersion()); + logData->minKnownCommittedVersion.advance(r->getMinKnownCommittedVersion()); } co_await waitUntilTLogAcceptsNewData(self, logData, ver, endVersion); @@ -3629,8 +3688,7 @@ Future pullAsyncData(TLogData* self, if (poppedIsKnownCommitted) { logData->knownCommittedVersion = std::max(logData->knownCommittedVersion, r->popped()); - logData->minKnownCommittedVersion = - std::max(logData->minKnownCommittedVersion, r->getMinKnownCommittedVersion()); + logData->minKnownCommittedVersion.advance(r->getMinKnownCommittedVersion()); } co_await waitUntilTLogAcceptsNewData(self, logData, ver, endVersion); @@ -4451,6 +4509,333 @@ Future tLog(IKeyValueStore* persistentData, } // UNIT TESTS +namespace { + +class CommittedPeekTestKnobs : NonCopyable { +public: + explicit CommittedPeekTestKnobs(bool batchEmpty = true) + : versionVector(SERVER_KNOBS->ENABLE_VERSION_VECTOR), + recoveryReply(SERVER_KNOBS->ENABLE_VERSION_VECTOR_REPLY_RECOVERY), + batching(SERVER_KNOBS->PEEK_BATCHING_EMPTY_MSG), timeout(SERVER_KNOBS->BLOCKING_PEEK_TIMEOUT), + batchingInterval(SERVER_KNOBS->PEEK_BATCHING_EMPTY_MSG_INTERVAL) { + auto* knobs = const_cast(SERVER_KNOBS); + knobs->ENABLE_VERSION_VECTOR = true; + knobs->ENABLE_VERSION_VECTOR_REPLY_RECOVERY = false; + knobs->PEEK_BATCHING_EMPTY_MSG = batchEmpty; + knobs->BLOCKING_PEEK_TIMEOUT = 0.4; + knobs->PEEK_BATCHING_EMPTY_MSG_INTERVAL = 0.4; + } + + ~CommittedPeekTestKnobs() { + auto* knobs = const_cast(SERVER_KNOBS); + knobs->ENABLE_VERSION_VECTOR = versionVector; + knobs->ENABLE_VERSION_VECTOR_REPLY_RECOVERY = recoveryReply; + knobs->PEEK_BATCHING_EMPTY_MSG = batching; + knobs->BLOCKING_PEEK_TIMEOUT = timeout; + knobs->PEEK_BATCHING_EMPTY_MSG_INTERVAL = batchingInterval; + } + +private: + const bool versionVector; + const bool recoveryReply; + const bool batching; + const double timeout; + const double batchingInterval; +}; + +class CommittedPeekTestFixture : NonCopyable { +public: + explicit CommittedPeekTestFixture(bool batchEmpty = true, Version initialCommitted = 99) + : knobs(batchEmpty), shared(deterministicRandom()->randomUniqueID(), + deterministicRandom()->randomUniqueID(), + nullptr, + nullptr, + makeReference>(ServerDBInfo()), + makeReference>(false), + makeReference>(false), + "", + makeReference>(false)), + queue(shared.persistentQueue), logData(makeReference(&shared, + TLogInterface(LocalityData()), + invalidTag, + true, + 0, + 0, + deterministicRandom()->randomUniqueID(), + g_network->protocolVersion(), + TLogSpillType::VALUE, + std::vector{ cdcTag() }, + "CommittedPeekTest")) { + logData->version.set(100); + logData->queueCommittedVersion.set(100); + logData->knownCommittedVersion = initialCommitted; + logData->durableKnownCommittedVersion = initialCommitted; + logData->minKnownCommittedVersion.advance(initialCommitted); + logData->persistentDataVersion = 0; + logData->persistentDataDurableVersion = 0; + logData->locality = 0; + logData->getTagData(cdcTag()); + logData->createTagData(cdcTag(), 0, NothingPersistent::True, PoppedRecently::False, UnpoppedRecovered::False); + } + + ~CommittedPeekTestFixture() { + // This fixture only exercises in-memory peeks and duplicate commits. Termination prevents + // LogData's normal persistent-state removal from dereferencing the absent disk stores. + shared.terminated.send(Void()); + } + + static Tag cdcTag() { return Tag(tagLocalityCDC, 0); } + + static TLogPeekRequest request() { return TLogPeekRequest(100, cdcTag(), false, false, {}, {}, {}, 4096); } + + Future peek(Promise reply, const TLogPeekRequest& req = request()) { + return tLogPeekMessages(reply, + &shared, + logData, + req.begin, + req.tag, + req.returnIfBlocked, + req.onlySpilled, + req.sequence, + req.end, + req.returnEmptyIfStopped, + req.replyByteLimit); + } + + void addMessage(Tag tag = cdcTag()) { + BinaryWriter message(AssumeVersion(g_network->protocolVersion())); + message << int32_t(0) << uint32_t(1) << uint16_t(1) << tag; + message << MutationRef(MutationRef::SetValue, "frontier-key"_sr, "frontier-value"_sr); + *reinterpret_cast(message.getData()) = message.getLength() - sizeof(int32_t); + auto bytes = message.toValue(); + commitMessages(&shared, logData, 100, bytes.arena(), bytes); + } + + Future commitFrontier(Version frontier, Version previous = 99) { + TLogCommitRequest req; + // A real request is timestamped by transport delivery; this fixture invokes the handler directly. + req.setRequestTime(g_network->timer()); + req.prevVersion = previous; + req.version = previous + 1; + req.knownCommittedVersion = std::max(99, frontier); + req.minKnownCommittedVersion = frontier; + req.seqPrevVersion = previous; + req.tLogCount = 1; + return tLogCommit(&shared, req, logData, PromiseStream()); + } + + void advance(Version frontier) { logData->minKnownCommittedVersion.advance(frontier); } + void stop() { logData->stop(); } + void popPastBegin() { logData->getTagData(cdcTag())->popped = 101; } + Version frontier() const { return logData->minKnownCommittedVersion.get(); } + int64_t timedOutPeeks() const { return logData->blockingPeekTimeouts.getValue(); } + void assertNoDiskCommit() const { + ASSERT(shared.diskQueueCommitBytes == 0); + ASSERT(logData->version.get() == 100); + } + +private: + CommittedPeekTestKnobs knobs; + TLogData shared; + std::unique_ptr queue; + Reference logData; +}; + +Future testCommittedPeekProgress(bool speculative) { + CommittedPeekTestFixture fixture; + if (speculative) { + fixture.addMessage(); + } + Promise reply; + Future peek = fixture.peek(reply); + ASSERT(!reply.getFuture().isReady()); + co_await delay(0.02); + ASSERT(!reply.getFuture().isReady()); + Future commit = fixture.commitFrontier(100); + ASSERT(!reply.getFuture().isReady()); + co_await commit; + TLogPeekReply result = co_await timeoutError(reply.getFuture(), 0.2); + co_await peek; + ASSERT(result.minKnownCommittedVersion == 100); + ASSERT(result.maxKnownVersion == 100); + ASSERT(result.end == 101); + ASSERT(result.messages.empty() == !speculative); + if (speculative) { + BinaryReader reader(result.messages, Unversioned()); + int32_t header; + Version version; + reader >> header >> version; + ASSERT(header == VERSION_HEADER); + ASSERT(version == 100); + } + fixture.assertNoDiskCommit(); +} + +} // namespace + +TEST_CASE("/NativeCDC/TLogCommittedFrontier/Empty") { + co_await testCommittedPeekProgress(false); +} + +TEST_CASE("/NativeCDC/TLogCommittedFrontier/Speculative") { + co_await testCommittedPeekProgress(true); +} + +TEST_CASE("/NativeCDC/TLogCommittedFrontier/AlreadyCommitted") { + CommittedPeekTestFixture fixture; + fixture.advance(100); + Promise reply; + Future peek = fixture.peek(reply); + TLogPeekReply result = co_await timeoutError(reply.getFuture(), 0.2); + co_await peek; + ASSERT(result.minKnownCommittedVersion == 100); + ASSERT(result.messages.empty()); +} + +TEST_CASE("/NativeCDC/TLogCommittedFrontier/Timeout") { + CommittedPeekTestFixture fixture; + Promise reply; + double started = now(); + Future peek = fixture.peek(reply); + TLogPeekReply result = co_await timeoutError(reply.getFuture(), 1.0); + co_await peek; + ASSERT(now() - started >= 0.35); + ASSERT(result.minKnownCommittedVersion == 99); + ASSERT(result.messages.empty()); + ASSERT(fixture.timedOutPeeks() == 1); +} + +TEST_CASE("/NativeCDC/TLogCommittedFrontier/Stop") { + CommittedPeekTestFixture fixture; + Promise reply; + Future peek = fixture.peek(reply); + co_await delay(0.02); + ASSERT(!reply.getFuture().isReady()); + fixture.stop(); + ASSERT(!reply.getFuture().isReady()); + TLogPeekReply result = co_await timeoutError(reply.getFuture(), 0.2); + co_await peek; + ASSERT(result.minKnownCommittedVersion == 99); + ASSERT(result.messages.empty()); +} + +TEST_CASE("/NativeCDC/TLogCommittedFrontier/AbsoluteDeadline") { + CommittedPeekTestFixture fixture(true, 90); + Promise reply; + double started = now(); + Future peek = fixture.peek(reply); + for (Version frontier = 91; frontier <= 93; ++frontier) { + co_await delay(0.1); + fixture.advance(frontier); + } + TLogPeekReply result = co_await timeoutError(reply.getFuture(), 0.2); + co_await peek; + ASSERT(now() - started >= 0.35 && now() - started < 0.6); + ASSERT(result.minKnownCommittedVersion == 93); + ASSERT(fixture.timedOutPeeks() == 1); +} + +TEST_CASE("/NativeCDC/TLogCommittedFrontier/Cancellation") { + CommittedPeekTestFixture fixture; + Promise abandonedReply; + Future abandoned = fixture.peek(abandonedReply); + co_await delay(0.02); + abandoned.cancel(); + ASSERT(abandoned.isReady() && abandoned.isError()); + ASSERT(abandoned.getError().code() == error_code_actor_cancelled); + Promise reply; + Future peek = fixture.peek(reply); + fixture.advance(100); + TLogPeekReply result = co_await timeoutError(reply.getFuture(), 0.2); + co_await peek; + ASSERT(result.minKnownCommittedVersion == 100); + ASSERT(!abandonedReply.getFuture().isReady()); +} + +TEST_CASE("/NativeCDC/TLogCommittedFrontier/MultipleWaiters") { + CommittedPeekTestFixture fixture; + std::vector> replies(8); + std::vector> peeks; + peeks.reserve(replies.size()); + for (auto& reply : replies) { + peeks.push_back(fixture.peek(reply)); + } + co_await delay(0.02); + co_await fixture.commitFrontier(98); + co_await fixture.commitFrontier(99); + ASSERT(fixture.frontier() == 99); + for (auto& reply : replies) { + ASSERT(!reply.getFuture().isReady()); + } + Future commit = fixture.commitFrontier(100); + for (auto& reply : replies) { + ASSERT(!reply.getFuture().isReady()); + } + co_await commit; + for (auto& reply : replies) { + TLogPeekReply result = co_await timeoutError(reply.getFuture(), 0.2); + ASSERT(result.minKnownCommittedVersion == 100); + } + co_await waitForAll(peeks); + fixture.assertNoDiskCommit(); +} + +TEST_CASE("/NativeCDC/TLogCommittedFrontier/OutOfOrderCommit") { + CommittedPeekTestFixture fixture; + Promise reply; + Future peek = fixture.peek(reply); + co_await delay(0.02); + Future commit = fixture.commitFrontier(100, 101); + ASSERT(!commit.isReady()); + ASSERT(!reply.getFuture().isReady()); + TLogPeekReply result = co_await timeoutError(reply.getFuture(), 0.2); + co_await peek; + ASSERT(result.minKnownCommittedVersion == 100); + ASSERT(!commit.isReady()); + commit.cancel(); + ASSERT(commit.isReady() && commit.isError()); + ASSERT(commit.getError().code() == error_code_actor_cancelled); + fixture.assertNoDiskCommit(); +} + +TEST_CASE("/NativeCDC/TLogCommittedFrontier/Bypass") { + for (int scenario = 0; scenario < 6; ++scenario) { + CommittedPeekTestFixture fixture(false); + TLogPeekRequest req = fixture.request(); + if (scenario == 0) { + req.end = 101; + } else if (scenario == 1) { + req.returnIfBlocked = true; + } else if (scenario == 2) { + req.onlySpilled = true; + } else if (scenario == 3) { + req.replyByteLimit = 0; + } else if (scenario == 4) { + req.tag = Tag(0, 0); + } else { + fixture.stop(); + } + fixture.addMessage(req.tag); + Promise reply; + Future peek = fixture.peek(reply, req); + TLogPeekReply result = co_await timeoutError(reply.getFuture(), 0.2); + co_await peek; + ASSERT(result.minKnownCommittedVersion == 99); + ASSERT(fixture.timedOutPeeks() == 0); + } +} + +TEST_CASE("/NativeCDC/TLogCommittedFrontier/Popped") { + CommittedPeekTestFixture fixture; + fixture.popPastBegin(); + Promise reply; + Future peek = fixture.peek(reply); + TLogPeekReply result = co_await timeoutError(reply.getFuture(), 0.2); + co_await peek; + ASSERT(result.popped.present() && result.popped.get() == 101); + ASSERT(result.minKnownCommittedVersion == 99); +} + TEST_CASE("/NativeCDC/TLogPeekReplyLimit") { ASSERT(tLogPeekReplyByteLimit(Tag(tagLocalityCDC, 0), 0) == 0); ASSERT(tLogPeekReplyByteLimit(Tag(tagLocalityCDC, 0), SERVER_KNOBS->MAXIMUM_PEEK_BYTES) == diff --git a/fdbserver/tlog/TestTLogServer.h b/fdbserver/tlog/TestTLogServer.h index bc179a8954d..e89af65cb38 100644 --- a/fdbserver/tlog/TestTLogServer.h +++ b/fdbserver/tlog/TestTLogServer.h @@ -102,7 +102,7 @@ struct TLogTestContext : NonCopyable, public ReferenceCounted { static Future peekCommitMessages(TLogTestContext* pTLogTestContext, uint16_t logGroupID, uint32_t tag); - explicit(false) TLogTestContext(TestTLogOptions& tLogOptions) : tLogOptions(tLogOptions), epoch(1) {} + explicit TLogTestContext(TestTLogOptions& tLogOptions) : tLogOptions(tLogOptions), epoch(1) {} // paramaters std::string diskQueueBasename; diff --git a/fdbserver/worker/CMakeLists.txt b/fdbserver/worker/CMakeLists.txt index 74b8e01d846..0fd50242475 100644 --- a/fdbserver/worker/CMakeLists.txt +++ b/fdbserver/worker/CMakeLists.txt @@ -70,3 +70,7 @@ target_link_libraries(fdbserver_worker fdbserver_tester fdbserver_tlog fdbserver_core) + +if(WITH_ROCKSDB) + target_compile_definitions(fdbserver_worker PRIVATE WITH_ROCKSDB) +endif() diff --git a/fdbserver/worker/MetricLogger.cpp b/fdbserver/worker/MetricLogger.cpp index aa8127534c0..ca2a66c59a0 100644 --- a/fdbserver/worker/MetricLogger.cpp +++ b/fdbserver/worker/MetricLogger.cpp @@ -45,7 +45,7 @@ namespace { struct MetricsRule { - explicit(false) MetricsRule(bool enabled = false, int minLevel = 0, StringRef const& name = StringRef()) + explicit MetricsRule(bool enabled = false, int minLevel = 0, StringRef const& name = StringRef()) : namePattern(name), enabled(enabled), minLevel(minLevel) {} Standalone typePattern; diff --git a/fdbserver/worker/worker.cpp b/fdbserver/worker/worker.cpp index d3d65f908c9..5dd0bf244a1 100644 --- a/fdbserver/worker/worker.cpp +++ b/fdbserver/worker/worker.cpp @@ -19,6 +19,7 @@ */ #include +#include #include #include #include @@ -29,6 +30,7 @@ #include "flow/Buggify.h" #include "flow/CodeProbe.h" #include "flow/IAsyncFile.h" +#include "fdbrpc/FailureMonitor.h" #include "fdbrpc/Locality.h" #include "fdbclient/GlobalConfig.h" #include "fdbclient/ProcessInterface.h" @@ -638,15 +640,7 @@ Future registrationClient(Referencebegin(); it != peers->end();) { - if (now() - it->second.second > FLOW_KNOBS->INCOMPATIBLE_PEER_DELAY_BEFORE_LOGGING) { - request.incompatiblePeers.push_back(it->first); - it = peers->erase(it); - } else { - it++; - } - } + request.incompatiblePeers = FlowTransport::transport().consumeReportableIncompatiblePeers(); bool ccInterfacePresent = ccInterface->get().present(); if (ccInterfacePresent) { @@ -1970,6 +1964,21 @@ bool skipInitRspInSim(const UID workerInterfID, const bool allowDropInSim) { return skip; } +Promise cacheLogRouterInitialization(WorkerCache& cache, + InitializeLogRouterRequest const& request) { + Promise ready; + cache.set(request.reqId, ready.getFuture()); + return ready; +} + +bool replyToCachedLogRouter(WorkerCache& cache, InitializeLogRouterRequest const& request) { + if (!cache.exists(request.reqId)) { + return false; + } + forwardPromise(Uncancellable{}, request.reply, cache.get(request.reqId)); + return true; +} + #ifdef FLOW_GRPC_ENABLED Future registerWorkerGrpcServices(UID id, Reference ccr) { if (GrpcServer::instance() == nullptr) { @@ -2636,7 +2645,7 @@ class WorkerServerCore { while (true) { InitializeLogRouterRequest req = co_await interf.logRouter.getFuture(); - if (!logRouterCache.exists(req.reqId)) { + if (!replyToCachedLogRouter(logRouterCache, req)) { LocalLineage _; getCurrentLineage()->modify(&RoleLineage::role) = recruitment::LogRouter; TLogInterface recruited(locality); @@ -2658,19 +2667,18 @@ class WorkerServerCore { DUMPTOKEN(recruited.enablePopRequest); DUMPTOKEN(recruited.snapRequest); - ReplyPromise logRouterReady = req.reply; - logRouterCache.set(req.reqId, logRouterReady.getFuture()); + Promise logRouterReady = cacheLogRouterInitialization(logRouterCache, req); Future logRouterProcess = logRouter(recruited, req, dbInfo); logRouterProcess = logRouterCache.removeOnReady(req.reqId, logRouterProcess); errorForwarders.add( zombie(recruited, forwardError(errors, Role::LOG_ROUTER, recruited.id(), logRouterProcess))); TraceEvent("LogRouterInitRequest", req.reqId).detail("LogRouterId", recruited.id()); + // A lost response must not leave duplicate requests waiting on that response's promise. + logRouterReady.send(recruited); if (!skipInitRspInSim(interf.id(), req.allowDropInSim)) { - logRouterReady.send(recruited); + req.reply.send(recruited); } - } else { - forwardPromise(Uncancellable{}, req.reply, logRouterCache.get(req.reqId)); } } } diff --git a/fdbserver/workloads/AsyncFile.h b/fdbserver/workloads/AsyncFile.h index b85cf928bd3..b8ff9389875 100644 --- a/fdbserver/workloads/AsyncFile.h +++ b/fdbserver/workloads/AsyncFile.h @@ -67,7 +67,7 @@ struct AsyncFileWorkload : TestWorkload { std::string path; - explicit(false) AsyncFileWorkload(WorkloadContext const&); + explicit AsyncFileWorkload(WorkloadContext const&); ~AsyncFileWorkload() override = default; // Allocates a buffer of a given size. If necessary, the buffer will be aligned to 4K diff --git a/fdbserver/workloads/BulkLoading.cpp b/fdbserver/workloads/BulkLoading.cpp index 4a04b6930e0..3dd3d7fd28f 100644 --- a/fdbserver/workloads/BulkLoading.cpp +++ b/fdbserver/workloads/BulkLoading.cpp @@ -24,7 +24,7 @@ #include "fdbclient/RangeLock.h" #include "fdbclient/SystemData.h" #include "fdbserver/core/BulkLoadUtil.h" -#include "fdbserver/core/RocksDBCheckpointUtils.h" +#include "fdbserver/checkpoint/RocksDBCheckpointUtils.h" #include "fdbserver/core/StorageMetrics.h" #include "fdbserver/tester/workloads.h" diff --git a/fdbserver/workloads/CMakeLists.txt b/fdbserver/workloads/CMakeLists.txt index 76564bbaea6..dcd55b71919 100644 --- a/fdbserver/workloads/CMakeLists.txt +++ b/fdbserver/workloads/CMakeLists.txt @@ -10,10 +10,14 @@ target_sources(fdbserver_workloads_test PRIVATE ../MemoryTrackerTest.cpp ../Glob configure_fdbserver_common_includes(fdbserver_workloads) target_include_directories(fdbserver_workloads PRIVATE - ${CMAKE_CURRENT_SOURCE_DIR}) + ${CMAKE_CURRENT_SOURCE_DIR} + ${CMAKE_SOURCE_DIR}/fdbclient) target_link_libraries(fdbserver_workloads PRIVATE + fdbserver_clustercontroller fdbserver_consistencyscan fdbserver_core + fdbserver_logsystem + fdbserver_checkpoint fdbserver_kvstore fdbserver_worker fdbserver_tester @@ -21,3 +25,7 @@ target_link_libraries(fdbserver_workloads PRIVATE fdbserver_resolver fdbserver_storageserver fdbserver_mocks3) + +if(WITH_ROCKSDB) + target_compile_definitions(fdbserver_workloads PRIVATE WITH_ROCKSDB) +endif() diff --git a/fdbserver/workloads/ClientMetric.cpp b/fdbserver/workloads/ClientMetric.cpp index 4a0586a3105..6cb90929a8d 100644 --- a/fdbserver/workloads/ClientMetric.cpp +++ b/fdbserver/workloads/ClientMetric.cpp @@ -158,6 +158,8 @@ struct ClientMetricWorkload : TestWorkload { break; } ++cnt; + // Independent writes must not inherit the previous transaction's retry backoff. + tr.fullReset(); } catch (Error& e) { err = e; } diff --git a/fdbserver/workloads/ConsistencyCheck.cpp b/fdbserver/workloads/ConsistencyCheck.cpp index 1c65ecead4e..23b0dbe1540 100644 --- a/fdbserver/workloads/ConsistencyCheck.cpp +++ b/fdbserver/workloads/ConsistencyCheck.cpp @@ -1026,10 +1026,13 @@ struct ConsistencyCheckWorkload : TestWorkload { // Check DataDistributor recruitment::Fitness fitnessLowerBound = recruitment::machineClassFitness( allWorkerProcessMap[db.master.address()].processClass, recruitment::DataDistributor); + // Bulk-load simulation can deliberately retain a DD with suboptimal fitness to let data moves finish. + const bool allowUnfitDistributor = g_network->isSimulated() && SERVER_KNOBS->CC_ENFORCE_USE_UNFIT_DD_IN_SIM; if (db.distributor.present() && (!nonExcludedWorkerProcessMap.contains(db.distributor.get().address()) || - recruitment::machineClassFitness(nonExcludedWorkerProcessMap[db.distributor.get().address()].processClass, - recruitment::DataDistributor) > fitnessLowerBound)) { + (!allowUnfitDistributor && + recruitment::machineClassFitness(nonExcludedWorkerProcessMap[db.distributor.get().address()].processClass, + recruitment::DataDistributor) > fitnessLowerBound))) { TraceEvent("ConsistencyCheck_DistributorNotBest") .detail("DataDistributorFitnessLowerBound", fitnessLowerBound) .detail("ExistingDistributorFitness", diff --git a/fdbserver/workloads/NativeCdcEndToEnd.cpp b/fdbserver/workloads/NativeCdcEndToEnd.cpp index f74e7b0fe7d..e24a96f881a 100644 --- a/fdbserver/workloads/NativeCdcEndToEnd.cpp +++ b/fdbserver/workloads/NativeCdcEndToEnd.cpp @@ -21,17 +21,24 @@ #include #include #include +#include #include #include #include #include +#include "NativeCdcInternal.h" +#include "fdbserver/core/NativeCdcMetadata.h" #include "fdbclient/DatabaseContext.h" #include "fdbclient/NativeCdc.h" #include "fdbclient/SystemData.h" +#include "fdbserver/clustercontroller/NativeCdcProxyBalancer.h" +#include "fdbserver/core/Knobs.h" #include "fdbserver/core/RecoveryState.h" #include "fdbserver/core/ServerDBInfo.h" +#include "fdbserver/logsystem/LogSystemConsumer.h" +#include "fdbserver/logsystem/LogSystemFactory.h" #include "fdbserver/tester/workloads.h" #include "fdbrpc/simulator.h" #include "flow/DeterministicRandom.h" @@ -61,6 +68,110 @@ class NativeCdcEndToEndWorkload : public TestWorkload { std::unordered_map, ExpectedWrite, KeyValueHash> expected; }; + class RetagMarkerLedger : public ReferenceCounted { + Key markerKey; + std::unordered_map writes; + std::unordered_map markerKeys; + std::unordered_map> epochObservations; + Version committedThrough = invalidVersion; + Version acknowledgedThrough = invalidVersion; + int nextValue = 0; + int replayedMutations = 0; + + public: + explicit RetagMarkerLedger(Key key) : markerKey(std::move(key)) {} + const Key& key() const { return markerKey; } + Version lastCommittedVersion() const { return committedThrough; } + int replayCount() const { return replayedMutations; } + // Every unacknowledged marker must be delivered again after replacement, independently of earlier observations. + void allowReplay() { epochObservations.clear(); } + + Value expectWrite(int valueBytes, Optional key = Optional()) { + std::string bytes = format("retag/%010d/", nextValue++); + bytes.resize(valueBytes, 'x'); + Value value{ StringRef(bytes) }; + ASSERT(writes.emplace(value, ExpectedWrite{ invalidVersion, {} }).second); + markerKeys.emplace(value, key.present() ? key.get() : markerKey); + return value; + } + + void committed(Value const& value, Version version) { + writes.at(value).committedVersion = version; + committedThrough = std::max(committedThrough, version); + } + + void observe(CDCConsumeReply const& reply) { + Version previousGroup = invalidVersion; + for (const auto& versioned : reply.mutations) { + ASSERT_GT(versioned.version, previousGroup); + ASSERT_GT(versioned.version, acknowledgedThrough); + ASSERT_LE(versioned.version, reply.lastConsumedVersion); + previousGroup = versioned.version; + for (const auto& mutation : versioned.mutations) { + ASSERT_EQ(mutation.type, MutationRef::SetValue); + const Value value(mutation.param2); + auto expected = writes.find(value); + ASSERT(expected != writes.end()); + ASSERT_EQ(mutation.param1, markerKeys.at(value)); + ASSERT(epochObservations[value].insert(versioned.version).second); + if (expected->second.committedVersion != invalidVersion) { + ASSERT_LE(versioned.version, expected->second.committedVersion); + } + if (!expected->second.observedVersions.insert(versioned.version).second) { + ++replayedMutations; + } + } + } + } + + void verifyThrough(Version version) const { + for (const auto& [value, expected] : writes) { + ASSERT_NE(expected.committedVersion, invalidVersion); + if (expected.committedVersion > acknowledgedThrough && expected.committedVersion <= version) { + const auto observed = epochObservations.find(value); + ASSERT(observed != epochObservations.end()); + ASSERT(observed->second.contains(expected.committedVersion)); + } + } + } + + void verifyBoundary(Version boundary) const { + bool before = false; + bool after = false; + for (const auto& [value, expected] : writes) { + before |= expected.committedVersion < boundary; + after |= expected.committedVersion >= boundary; + } + ASSERT(before && after); + verifyThrough(committedThrough); + } + + void acknowledged(Version version) { + verifyThrough(version); + acknowledgedThrough = version; + } + }; + + struct RetagSnapshot { + NativeCdcTagState state; + std::vector history; + }; + + struct RetagFixtureAttempt { + Version readVersion; + Version gapVersion; + Version committedVersion = invalidVersion; + }; + + struct RetagRestartMarkers { + CDCStreamId streamId; + Version before; + Version cutover; + Version after; + Tag oldTag; + Tag newTag; + }; + int initialStreamCount; int minStreamCount; int maxStreamCount; @@ -69,18 +180,25 @@ class NativeCdcEndToEndWorkload : public TestWorkload { int rounds; int assignmentPublicationChecks; bool testProxyReplacement; + bool testProxyRebalance; + bool testProxyRebalanceAutomatic; bool testTagOwnership; bool injectUndeliveredProxyHalt; bool testMemoryBound; bool testReplyChunking; + bool testMultipleRanges; bool testOversizedPeek; bool testDurableAckScan; bool testDelayedRetention; bool testRetiredRecovery; bool blockRetiredPopWithLiveStream; bool testRetiredSharedTagSnapshot; + bool testRetagCompatibility; + bool testRetaggingMemoryBound; bool prepareRestartDrain; bool drainAfterRestart; + bool testRetaggedRestart; + bool testRetagTransactionRetries; int memoryTestValueBytes; double retentionValidationDelay; double drainProbability; @@ -225,7 +343,9 @@ class NativeCdcEndToEndWorkload : public TestWorkload { StreamState stream; stream.name = Key(StringRef(format("native-cdc-e2e/stream/%04d", nextStreamNumber++))); stream.keys = std::move(keys); - co_await timeoutError(registerNativeCdcStreamClient(cx, stream.name, stream.keys), operationTimeout); + // GCC 13 cannot lower initializer-list vector arguments inside these awaited expressions. + const std::vector ranges{ stream.keys }; + co_await timeoutError(registerNativeCdcStreamClient(cx, stream.name, ranges), operationTimeout); stream.consumer = co_await timeoutError(createNativeCdcConsumer(cx, stream.name), operationTimeout); streams.push_back(std::move(stream)); } @@ -233,6 +353,12 @@ class NativeCdcEndToEndWorkload : public TestWorkload { Future addStream(Database cx) { return addStream(cx, randomOverlappingRange()); } Future initializeStreams(Database cx) { + if (testProxyRebalance || testProxyRebalanceAutomatic) { + for (int i = 0; i < initialStreamCount; ++i) { + co_await addStream(cx, KeyRange(KeyRangeRef(keyForIndex(0), keyForIndex(keyCount)))); + } + co_return; + } for (int i = 0; i < initialStreamCount; ++i) { co_await addStream(cx); } @@ -255,18 +381,426 @@ class NativeCdcEndToEndWorkload : public TestWorkload { co_await addStream(cx, KeyRange(KeyRangeRef(keyForIndex(0), keyForIndex(keyCount)))); } + Future initializeRetaggingStreams(Database cx) { + for (int i = 0; i < initialStreamCount; ++i) { + const Key key = keyForIndex(i); + const Key end = testRetaggingMemoryBound ? keyForIndex(i + 1) : keyAfter(key); + co_await addStream(cx, KeyRange(KeyRangeRef(key, end))); + } + } + + Future readRetagSnapshot(Transaction* tr, int index, int maxStreams) { + const CDCStreamId streamId = streams[index].consumer->position().streamId; + const auto states = co_await readNativeCdcTagStates(tr, maxStreams); + ASSERT(states.present()); + const auto found = std::find_if(states.get().begin(), states.get().end(), [streamId](const auto& state) { + return state.streamId == streamId; + }); + ASSERT(found != states.get().end()); + RetagSnapshot result; + result.state = *found; + ASSERT(result.state.ranges == std::vector{ streams[index].keys }); + const RangeResult history = co_await tr->getRange(cdcTagHistoryRangeFor(streamId), 3); + ASSERT(!history.more && !history.empty() && history.size() <= 2); + for (const auto& row : history) { + result.history.push_back(decodeCDCTagHistoryEntry(row.key, row.value)); + } + co_return result; + } + + Future readRetagSnapshot(Database cx, int index, int maxStreams = 16) { + RetagSnapshot result; + // NOLINTNEXTLINE(cppcoreguidelines-avoid-capturing-lambda-coroutines) Database::run owns the closure. + co_await cx.run([this, &result, index, maxStreams](Transaction* tr) -> Future { + tr->setOption(FDBTransactionOptions::READ_LOCK_AWARE); + tr->setOption(FDBTransactionOptions::READ_SYSTEM_KEYS); + result = co_await readRetagSnapshot(tr, index, maxStreams); + }); + co_return result; + } + + Future writeRetagMarkers(Database cx, + std::vector indices, + std::vector> ledgers) { + std::vector> values; + values.reserve(indices.size()); + for (const int index : indices) { + values.emplace_back(ledgers[index]->key(), ledgers[index]->expectWrite(memoryTestValueBytes)); + } + const Version committed = co_await writeValues(cx, values); + for (int i = 0; i < static_cast(indices.size()); ++i) { + ledgers[indices[i]]->committed(values[i].second, committed); + } + co_return committed; + } + + Future drainRetagMarkers(int index, Reference ledger, bool acknowledge) { + const double deadline = now() + operationTimeout; + while (streams[index].consumer->position().lastConsumedVersion < ledger->lastCommittedVersion()) { + ledger->observe(co_await timeoutError(streams[index].consumer->consume(), operationTimeout)); + ASSERT_LT(now(), deadline); + co_await delay(0.01); + } + ledger->verifyThrough(ledger->lastCommittedVersion()); + if (acknowledge) { + const Version position = streams[index].consumer->position().lastConsumedVersion; + co_await timeoutError(streams[index].consumer->acknowledge(), operationTimeout); + ledger->acknowledged(position); + } + } + + Future waitForCanonicalRetag(Database cx, int index, CDCTagHistoryEntry assignment) { + const double deadline = now() + operationTimeout; + while (true) { + const RetagSnapshot snapshot = co_await readRetagSnapshot(cx, index); + ASSERT_EQ(snapshot.state.assignment.tag, assignment.tag); + ASSERT_EQ(snapshot.state.assignment.version, assignment.version); + if (!snapshot.state.pending) { + ASSERT_EQ(snapshot.history.size(), 1); + co_return; + } + ASSERT_LT(now(), deadline); + co_await delay(0.05); + } + } + + Future commitRetagFixture(Database cx, + int index, + RetagSnapshot original, + Tag destination, + int maxStreams, + std::vector> gapLedgers) { + ASSERT(!original.state.pending); + ASSERT_EQ(original.history.size(), 1); + std::unordered_map> attempts; + bool readRetryInjected = false; + bool commitRetryInjected = false; + bool ambiguousCommit = false; + Transaction tr(cx); + while (true) { + Error err; + try { + tr.setOption(FDBTransactionOptions::LOCK_AWARE); + tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); + if (testRetagTransactionRetries && !readRetryInjected) { + readRetryInjected = true; + throw future_version(); + } + RetagSnapshot snapshot = co_await readRetagSnapshot(&tr, index, maxStreams); + ASSERT_EQ(snapshot.state.streamId, original.state.streamId); + ASSERT(snapshot.state.ranges == original.state.ranges); + ASSERT_EQ(snapshot.state.minVersion, original.state.minVersion); + if (snapshot.state.pending) { + ASSERT_EQ(snapshot.history.size(), 2); + ASSERT_EQ(snapshot.history.front().tag, original.state.assignment.tag); + ASSERT_EQ(snapshot.history.front().version, original.state.assignment.version); + ASSERT_EQ(snapshot.state.assignment.tag, destination); + // An ambiguous commit may already have installed our exact history row. Never turn that retry into + // another move, or accept an unrelated move merely because it chose the same destination. + const auto submitted = attempts.find(snapshot.state.historyKey); + ASSERT(submitted != attempts.end()); + const Version cutover = snapshot.state.assignment.version; + ASSERT_LT(snapshot.state.minVersion, cutover); + bool matchesAttempt = false; + for (const auto& attempt : submitted->second) { + if (attempt.committedVersion != invalidVersion) { + ASSERT_EQ(cutover, attempt.committedVersion); + } + matchesAttempt |= cutover > attempt.readVersion && + (attempt.gapVersion == invalidVersion || + (attempt.gapVersion > attempt.readVersion && cutover > attempt.gapVersion)); + } + ASSERT(matchesAttempt); + ASSERT(!testRetagTransactionRetries || (readRetryInjected && ambiguousCommit)); + CODE_PROBE(ambiguousCommit, "Native CDC retag fixture recognizes its own ambiguous committed move"); + co_return snapshot; + } + ASSERT_EQ(snapshot.history.size(), 1); + ASSERT_EQ(snapshot.state.historyKey, original.state.historyKey); + ASSERT_EQ(snapshot.state.assignment.tag, original.state.assignment.tag); + ASSERT_EQ(snapshot.state.assignment.version, original.state.assignment.version); + const bool prepared = co_await retagNativeCdcStream(&tr, snapshot.state, destination); + ASSERT(prepared); + const Version readVersion = co_await tr.getReadVersion(); + Version gapVersion = invalidVersion; + if (!gapLedgers.empty()) { + // Each retried attempt needs its own separately committed marker after its fresh read version. + // Failed attempts remain in the ledger and must still be delivered. + gapVersion = co_await writeRetagMarkers(cx, { index }, gapLedgers); + ASSERT_GT(gapVersion, readVersion); + } + const Key historyKey = cdcTagHistoryKeyFor(snapshot.state.streamId, readVersion, destination); + auto& attempt = attempts[historyKey].emplace_back(RetagFixtureAttempt{ readVersion, gapVersion }); + co_await tr.commit(); + attempt.committedVersion = tr.getCommittedVersion(); + if (testRetagTransactionRetries && !commitRetryInjected) { + commitRetryInjected = true; + throw commit_unknown_result(); + } + tr.reset(); + continue; + } catch (Error& e) { + ambiguousCommit |= e.code() == error_code_commit_unknown_result; + err = e; + } + co_await tr.onError(err); + } + } + + Future assertRetagRejected(Database cx, NativeCdcTagState expected, Tag destination) { + // NOLINTNEXTLINE(cppcoreguidelines-avoid-capturing-lambda-coroutines) Database::run owns the closure. + co_await cx.run([expected = std::move(expected), destination](Transaction* tr) -> Future { + tr->setOption(FDBTransactionOptions::LOCK_AWARE); + tr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); + const bool prepared = co_await retagNativeCdcStream(tr, expected, destination); + ASSERT(!prepared); + }); + } + + Future retagAcrossConcurrentWrite(Database cx, + int index, + Tag destination, + std::vector> ledgers) { + const RetagSnapshot original = co_await readRetagSnapshot(cx, index); + RetagSnapshot snapshot = co_await commitRetagFixture(cx, index, original, destination, 16, ledgers); + co_await assertRetagRejected(cx, original.state, destination); + co_await assertRetagRejected(cx, snapshot.state, original.state.assignment.tag); + CODE_PROBE(true, "Native CDC live retag uses its commit boundary and rejects stale or pending moves"); + co_return snapshot; + } + + Future writeRetagBatch(Database cx, Reference ledger) { + std::vector> values; + // Small independent mutations fit in one raw peek but expand beyond it when materialized. + for (int i = 0; i < 12; ++i) { + Key key = ledger->key().withSuffix(StringRef(format("/%02d", i))); + values.emplace_back(key, ledger->expectWrite(32, key)); + } + const Version committed = co_await writeValues(cx, values); + for (const auto& [key, value] : values) { + ledger->committed(value, committed); + } + co_return committed; + } + + Future consumeRetagWithAckPause(int index, + Reference ledger, + Version through, + Reference> firstBatches, + Future releaseAcknowledgements) { + bool first = true; + while (streams[index].consumer->position().lastConsumedVersion < through) { + const CDCConsumeReply reply = co_await streams[index].consumer->consume(); + ledger->observe(reply); + ledger->verifyThrough(reply.lastConsumedVersion); + if (first) { + if (reply.mutations.empty()) { + continue; + } + first = false; + firstBatches->set(firstBatches->get() + 1); + co_await releaseAcknowledgements; + } + co_await streams[index].consumer->acknowledge(); + ledger->acknowledged(reply.lastConsumedVersion); + } + ASSERT(!first); + ledger->verifyThrough(ledger->lastCommittedVersion()); + } + + Future retainedTagBytes(Tag tag, Version begin, Version end) { + Reference logs = makeLogSystemConsumerFromServerDBInfo(UID(), dbInfo->get()); + Reference cursor = logs->peekSingle(UID(), begin, tag); + int64_t bytes = 0; + while (cursor->version().version <= end) { + if (!cursor->hasMessage()) { + co_await cursor->getMore(); + ASSERT_LE(cursor->popped(), begin); + continue; + } + bytes += cursor->getMessageWithTags().size(); + cursor->nextMessage(); + } + co_return bytes; + } + + void checkRetagBufferStatus(CDCProxyBufferStatus const& status) const { + ASSERT_GE(status.bufferedBytes, 0); + ASSERT_LE(status.bufferedBytes, status.activePermits); + ASSERT_LE(status.activePermits, status.bufferLimit); + ASSERT_LE(status.peakActivePermits, status.bufferLimit); + } + + Future getRetagBufferStatus(Database cx, UID owner) { + const auto result = co_await timeoutError( + getAssignedProxyStatus(cx, streams.front().consumer->position().streamId), operationTimeout); + ASSERT_EQ(result.first.id(), owner); + checkRetagBufferStatus(result.second); + co_return result.second; + } + + Future validateRetaggingMemoryBound(Database cx) { + ASSERT_EQ(streams.size(), 2); + const RetagSnapshot original = co_await readRetagSnapshot(cx, 0); + const RetagSnapshot destination = co_await readRetagSnapshot(cx, 1); + ASSERT_NE(original.state.assignment.tag, destination.state.assignment.tag); + ASSERT_EQ(original.state.proxyId, destination.state.proxyId); + const Tag oldTag = original.state.assignment.tag; + const Tag newTag = destination.state.assignment.tag; + auto moving = makeReference(streams[0].keys.begin); + auto active = makeReference(streams[1].keys.begin); + const Version before = co_await writeRetagBatch(cx, moving); + const double oldestCommittedAt = now(); + const RetagSnapshot pending = co_await commitRetagFixture(cx, 0, original, newTag, 16, {}); + const Version destinationVersion = co_await writeRetagBatch(cx, active); + const Version after = co_await writeRetagBatch(cx, moving); + ASSERT_LT(before, pending.state.assignment.version); + ASSERT_GE(after, pending.state.assignment.version); + + ASSERT_LT(operationTimeout, SERVER_KNOBS->CDC_PROXY_CONSUME_POLL_TIMEOUT); + const double consumeStarted = now(); + Promise releaseAcknowledgements; + auto firstBatches = makeReference>(0); + std::vector> consumers{ + consumeRetagWithAckPause(0, moving, after, firstBatches, releaseAcknowledgements.getFuture()), + consumeRetagWithAckPause(1, active, after, firstBatches, releaseAcknowledgements.getFuture()) + }; + // Both real readers must deliver before either acknowledges, and well before a consume lease can expire. + while (firstBatches->get() < 2) { + co_await timeoutError(firstBatches->onChange(), std::max(0.0, consumeStarted + operationTimeout - now())); + } + ASSERT_LT(now() - consumeStarted, SERVER_KNOBS->CDC_PROXY_CONSUME_POLL_TIMEOUT); + auto status = co_await getRetagBufferStatus(cx, original.state.proxyId); + ASSERT_GT(status.bufferedBytes, 0); + const int64_t oldBytes = co_await timeoutError(retainedTagBytes(oldTag, before, before), operationTimeout); + const int64_t newBytes = + co_await timeoutError(retainedTagBytes(newTag, destinationVersion, after), operationTimeout); + ASSERT_GT(oldBytes, 0); + ASSERT_GT(newBytes, 0); + const double pauseStarted = now(); + co_await delay(retentionValidationDelay); + ASSERT_EQ(co_await timeoutError(retainedTagBytes(oldTag, before, before), operationTimeout), oldBytes); + ASSERT_EQ(co_await timeoutError(retainedTagBytes(newTag, destinationVersion, after), operationTimeout), + newBytes); + const RetagSnapshot held = co_await readRetagSnapshot(cx, 0); + ASSERT(held.state.pending); + ASSERT_EQ(held.state.minVersion, original.state.minVersion); + status = co_await getRetagBufferStatus(cx, original.state.proxyId); + TraceEvent("NativeCdcRetagRetentionPause") + .detail("OldTagBytes", oldBytes) + .detail("DestinationTagBytes", newBytes) + .detail("PauseSeconds", now() - pauseStarted) + .detail("OldestCommitAgeSeconds", now() - oldestCommittedAt) + .detail("BufferedBytes", status.bufferedBytes) + .detail("PeakActivePermits", status.peakActivePermits); + releaseAcknowledgements.send(Void()); + co_await timeoutError(waitForAll(consumers), operationTimeout); + co_await waitForCanonicalRetag(cx, 0, pending.state.assignment); + Reference logs = makeLogSystemConsumerFromServerDBInfo(UID(), dbInfo->get()); + co_await timeoutError(logs->waitForPopped(pending.state.assignment.version, oldTag), operationTimeout); + co_await timeoutError(logs->waitForPopped(after + 1, newTag), operationTimeout); + status = co_await getRetagBufferStatus(cx, original.state.proxyId); + ASSERT_EQ(status.bufferedBytes, 0); + CODE_PROBE(true, "Native CDC retagging delivers both streams before acknowledgement within a bounded budget"); + CODE_PROBE(true, "Native CDC retains both retag histories during an acknowledgement pause then drains"); + for (const auto& stream : streams) { + co_await removeNativeCdcStreamClient(cx, stream.name); + } + streams.clear(); + co_await timeoutError(waitForRetiredTagCleanup(cx), operationTimeout); + } + + Future validateRetagCompatibility(Database cx) { + ASSERT_EQ(streams.size(), 4); + ASSERT_EQ(cx->clientInfo->get().nativeCdcTagCount, 2); + std::vector> ledgers; + std::vector initial; + for (int i = 0; i < static_cast(streams.size()); ++i) { + ledgers.push_back(makeReference(streams[i].keys.begin)); + initial.push_back(co_await readRetagSnapshot(cx, i)); + ASSERT(!initial.back().state.pending); + ASSERT_EQ(initial.back().state.proxyId, initial.front().state.proxyId); + } + const Tag originalTag = initial.front().state.assignment.tag; + const Tag destination(tagLocalityCDC, originalTag.id == 0 ? 1 : 0); + int sibling = -1; + for (int i = 1; i < static_cast(streams.size()); ++i) { + if (initial[i].state.assignment.tag == originalTag) { + sibling = i; + break; + } + } + ASSERT_GE(sibling, 0); + co_await writeRetagMarkers(cx, { 0, 1, 2, 3 }, ledgers); + const RetagSnapshot pending = co_await retagAcrossConcurrentWrite(cx, 0, destination, ledgers); + co_await writeRetagMarkers(cx, { 0, 1, 2, 3 }, ledgers); + co_await drainRetagMarkers(0, ledgers[0], false); + ledgers[0]->verifyBoundary(pending.state.assignment.version); + const RetagSnapshot unacknowledged = co_await readRetagSnapshot(cx, 0); + ASSERT_EQ(unacknowledged.history.size(), 2); + + const CDCProxyInterface originalOwner = + co_await timeoutError(waitForAssignedProxy(cx, pending.state.streamId), operationTimeout); + ledgers[0]->allowReplay(); + const int previousReplays = ledgers[0]->replayCount(); + co_await timeoutError(haltProxyUntilReplaced(cx, originalOwner, false), operationTimeout); + co_await timeoutError(waitForAssignedProxy(cx, pending.state.streamId, originalOwner.id()), operationTimeout); + co_await forceTransactionSystemRecovery(); + co_await writeRetagMarkers(cx, { 0, sibling }, ledgers); + co_await drainRetagMarkers(0, ledgers[0], false); + ASSERT_GT(ledgers[0]->replayCount(), previousReplays); + ledgers[0]->verifyBoundary(pending.state.assignment.version); + const RetagSnapshot recovered = co_await readRetagSnapshot(cx, 0); + ASSERT_EQ(recovered.state.minVersion, initial[0].state.minVersion); + co_await drainRetagMarkers(0, ledgers[0], true); + co_await waitForCanonicalRetag(cx, 0, pending.state.assignment); + CODE_PROBE(true, "Native CDC compatibility reader replays both retag intervals after recovery"); + + // An unacknowledged sibling still protects the retired tag after this stream's transition is canonical. + const RetagSnapshot lagging = co_await readRetagSnapshot(cx, sibling); + ASSERT_EQ(lagging.state.minVersion, initial[sibling].state.minVersion); + ASSERT_EQ(lagging.state.assignment.tag, originalTag); + // NOLINTNEXTLINE(cppcoreguidelines-avoid-capturing-lambda-coroutines) Database::run owns the closure. + co_await cx.run([originalTag, pending](Transaction* tr) -> Future { + tr->setOption(FDBTransactionOptions::READ_LOCK_AWARE); + tr->setOption(FDBTransactionOptions::READ_SYSTEM_KEYS); + const Optional retired = co_await tr->get(cdcRetiredTagPopKeyFor(originalTag)); + const Optional watermark = co_await tr->get(cdcRetiredTagPopVersionKeyFor(originalTag)); + ASSERT(retired.present() && watermark.present()); + ASSERT_GE(decodeCDCMinVersionValue(watermark.get()), pending.state.assignment.version); + }); + const RetagSnapshot returned = co_await retagAcrossConcurrentWrite(cx, 0, originalTag, ledgers); + co_await writeRetagMarkers(cx, { 0 }, ledgers); + co_await drainRetagMarkers(0, ledgers[0], true); + ledgers[0]->verifyBoundary(returned.state.assignment.version); + co_await waitForCanonicalRetag(cx, 0, returned.state.assignment); + for (int i = 1; i < static_cast(streams.size()); ++i) { + co_await drainRetagMarkers(i, ledgers[i], true); + } + CODE_PROBE(true, + "Native CDC retag compatibility preserves shared-tag data and supports returning to an old tag"); + for (const auto& stream : streams) { + co_await timeoutError(removeNativeCdcStreamClient(cx, stream.name), operationTimeout); + } + streams.clear(); + co_await timeoutError(waitForRetiredTagCleanup(cx), operationTimeout); + co_await timeoutError(waitForFullyRecovered(), operationTimeout); + } + Future validatePublicLifecycle(Database cx) { const Key name = "native-cdc-e2e/lifecycle"_sr; const KeyRange keys(KeyRangeRef("native-cdc-e2e/lifecycle/"_sr, "native-cdc-e2e/lifecycle0"_sr)); const KeyRange conflictingKeys(KeyRangeRef("native-cdc-e2e/lifecycle/"_sr, "native-cdc-e2e/lifecycle1"_sr)); + const std::vector ranges{ keys }; const CDCStreamId streamId = - co_await timeoutError(registerNativeCdcStreamClient(cx, name, keys), operationTimeout); - ASSERT_EQ(co_await timeoutError(registerNativeCdcStreamClient(cx, name, keys), operationTimeout), streamId); + co_await timeoutError(registerNativeCdcStreamClient(cx, name, ranges), operationTimeout); + ASSERT_EQ(co_await timeoutError(registerNativeCdcStreamClient(cx, name, ranges), operationTimeout), streamId); bool conflictingRegistrationRejected = false; try { - co_await timeoutError(registerNativeCdcStreamClient(cx, name, conflictingKeys), operationTimeout); + const std::vector conflictingRanges{ conflictingKeys }; + co_await timeoutError(registerNativeCdcStreamClient(cx, name, conflictingRanges), operationTimeout); } catch (Error& e) { if (e.code() != error_code_client_invalid_operation) { throw; @@ -281,7 +815,8 @@ class NativeCdcEndToEndWorkload : public TestWorkload { listed.begin(), listed.end(), [&](NativeCdcStreamInfo const& stream) { return stream.name == name; }); ASSERT_EQ(found != listed.end(), true); ASSERT_EQ(found->streamId, streamId); - ASSERT_EQ(found->keys, keys); + ASSERT_EQ(found->ranges.size(), 1); + ASSERT_EQ(found->ranges.front(), keys); bool futureConsumeRejected = false; try { @@ -345,7 +880,8 @@ class NativeCdcEndToEndWorkload : public TestWorkload { const KeyRange expectedLower(KeyRangeRef(keys.begin, lowerClear.end)); const KeyRange expectedUpper(KeyRangeRef(upperClear.begin, keys.end)); - co_await timeoutError(registerNativeCdcStreamClient(cx, name, keys), operationTimeout); + const std::vector ranges{ keys }; + co_await timeoutError(registerNativeCdcStreamClient(cx, name, ranges), operationTimeout); Reference consumer = co_await timeoutError(createNativeCdcConsumer(cx, name), operationTimeout); @@ -399,8 +935,9 @@ class NativeCdcEndToEndWorkload : public TestWorkload { const Value value = Value(StringRef(format("assignment-value/%04d", check))); co_await delay(0.1); + const std::vector ranges{ keys }; const CDCStreamId streamId = - co_await timeoutError(registerNativeCdcStreamClient(cx, name, keys), operationTimeout); + co_await timeoutError(registerNativeCdcStreamClient(cx, name, ranges), operationTimeout); Reference consumer = co_await timeoutError(createNativeCdcConsumer(cx, name), operationTimeout); ASSERT_EQ(consumer->position().streamId, streamId); @@ -410,6 +947,181 @@ class NativeCdcEndToEndWorkload : public TestWorkload { co_await timeoutError(removeNativeCdcStreamClient(cx, name), operationTimeout); } + Key multipleRangeKey(StringRef suffix) const { + return suffix.withPrefix("native-cdc-e2e/multiple-ranges/data/"_sr); + } + + KeyRange multipleRangeKeys(StringRef begin, StringRef end) const { + return KeyRange(KeyRangeRef(multipleRangeKey(begin), multipleRangeKey(end))); + } + + Future consumeExpectedVersion(Reference consumer, + Version committed, + Standalone> expected) { + const double deadline = now() + operationTimeout; + bool observed = false; + while (!observed) { + ASSERT_LT(now(), deadline); + CDCConsumeReply reply = co_await timeoutError(consumer->consume(), deadline - now()); + for (const auto& versioned : reply.mutations) { + ASSERT_LE(versioned.version, reply.lastConsumedVersion); + if (versioned.version != committed) { + continue; + } + ASSERT(!observed); + ASSERT_EQ(versioned.mutations.size(), expected.size()); + for (int i = 0; i < expected.size(); ++i) { + ASSERT_EQ(versioned.mutations[i].type, expected[i].type); + ASSERT_EQ(versioned.mutations[i].param1, expected[i].param1); + ASSERT_EQ(versioned.mutations[i].param2, expected[i].param2); + } + observed = true; + } + } + ASSERT_GE(consumer->position().lastConsumedVersion, committed); + } + + Future writeMultipleRangeClear(Database cx) { + Transaction tr(cx); + while (true) { + Error err; + try { + tr.set(multipleRangeKey("e"_sr), "before-left"_sr); + tr.set(multipleRangeKey("q"_sr), "before-middle"_sr); + tr.set(multipleRangeKey("w"_sr), "before-right"_sr); + tr.set(multipleRangeKey("h"_sr), "before-gap"_sr); + tr.clear(multipleRangeKeys("d"_sr, "x"_sr)); + tr.set(multipleRangeKey("e"_sr), "after-left"_sr); + tr.set(multipleRangeKey("q"_sr), "after-middle"_sr); + tr.set(multipleRangeKey("y"_sr), "after-right"_sr); + tr.clear(multipleRangeKeys("g"_sr, "j"_sr)); + co_await tr.commit(); + co_return tr.getCommittedVersion(); + } catch (Error& e) { + err = e; + } + co_await tr.onError(err); + } + } + + Future validateMultipleRanges(Database cx) { + ASSERT(streams.empty()); + ASSERT_EQ(cx->clientInfo->get().nativeCdcTagCount, 1); + const Key name = "native-cdc-e2e/multiple-ranges"_sr; + const Key gapName = "native-cdc-e2e/multiple-ranges-gap"_sr; + const std::vector ranges{ multipleRangeKeys("b"_sr, "f"_sr), + multipleRangeKeys("m"_sr, "r"_sr), + multipleRangeKeys("w"_sr, "z"_sr) }; + const std::vector registrationRanges{ + multipleRangeKeys("m"_sr, "r"_sr), multipleRangeKeys("c"_sr, "f"_sr), multipleRangeKeys("b"_sr, "d"_sr), + multipleRangeKeys("b"_sr, "c"_sr), multipleRangeKeys("x"_sr, "z"_sr), multipleRangeKeys("w"_sr, "x"_sr), + multipleRangeKeys("m"_sr, "r"_sr) + }; + const CDCStreamId streamId = + co_await timeoutError(registerNativeCdcStreamClient(cx, name, registrationRanges), operationTimeout); + ASSERT_EQ(co_await timeoutError(registerNativeCdcStreamClient(cx, name, ranges), operationTimeout), streamId); + bool changedGapsRejected = false; + try { + const std::vector mergedRanges{ multipleRangeKeys("b"_sr, "z"_sr) }; + co_await timeoutError(registerNativeCdcStreamClient(cx, name, mergedRanges), operationTimeout); + } catch (Error& e) { + if (e.code() != error_code_client_invalid_operation) { + throw; + } + changedGapsRejected = true; + } + ASSERT(changedGapsRejected); + const std::vector listed = + co_await timeoutError(listNativeCdcStreamsClient(cx), operationTimeout); + const auto found = std::find_if( + listed.begin(), listed.end(), [&](NativeCdcStreamInfo const& stream) { return stream.name == name; }); + ASSERT(found != listed.end()); + ASSERT_EQ(found->streamId, streamId); + ASSERT_EQ(found->ranges, ranges); + + // Route excluded keys onto the same tag so the proxy must filter shared-tag false positives. + const std::vector gapRanges{ multipleRangeKeys("a"_sr, "zz"_sr) }; + co_await timeoutError(registerNativeCdcStreamClient(cx, gapName, gapRanges), operationTimeout); + Reference consumer = + co_await timeoutError(createNativeCdcConsumer(cx, name), operationTimeout); + ASSERT_EQ(consumer->position().streamId, streamId); + std::vector> values; + for (StringRef suffix : + { "a"_sr, "b"_sr, "e"_sr, "f"_sr, "h"_sr, "m"_sr, "q"_sr, "r"_sr, "s"_sr, "w"_sr, "y"_sr, "z"_sr }) { + values.emplace_back(multipleRangeKey(suffix), suffix); + } + Standalone> expected; + for (StringRef suffix : { "b"_sr, "e"_sr, "m"_sr, "q"_sr, "w"_sr, "y"_sr }) { + expected.push_back_deep(expected.arena(), + MutationRef(MutationRef::SetValue, multipleRangeKey(suffix), suffix)); + } + const Version written = co_await writeValues(cx, values); + co_await consumeExpectedVersion(consumer, written, expected); + co_await timeoutError(consumer->acknowledge(), operationTimeout); + + expected = Standalone>(); + for (const auto& [suffix, value] : { std::pair("e"_sr, "before-left"_sr), + std::pair("q"_sr, "before-middle"_sr), + std::pair("w"_sr, "before-right"_sr) }) { + expected.push_back_deep(expected.arena(), + MutationRef(MutationRef::SetValue, multipleRangeKey(suffix), value)); + } + for (const auto& range : { multipleRangeKeys("d"_sr, "f"_sr), ranges[1], multipleRangeKeys("w"_sr, "x"_sr) }) { + expected.push_back_deep(expected.arena(), MutationRef(MutationRef::ClearRange, range.begin, range.end)); + } + for (const auto& [suffix, value] : { std::pair("e"_sr, "after-left"_sr), + std::pair("q"_sr, "after-middle"_sr), + std::pair("y"_sr, "after-right"_sr) }) { + expected.push_back_deep(expected.arena(), + MutationRef(MutationRef::SetValue, multipleRangeKey(suffix), value)); + } + const Version cleared = co_await writeMultipleRangeClear(cx); + co_await consumeExpectedVersion(consumer, cleared, expected); + + CDCProxyInterface original = co_await timeoutError(waitForAssignedProxy(cx, streamId), operationTimeout); + co_await timeoutError(haltProxyUntilReplaced(cx, original, false), operationTimeout); + CDCProxyInterface replacement = + co_await timeoutError(waitForAssignedProxy(cx, streamId, original.id()), operationTimeout); + ASSERT_NE(original.id(), replacement.id()); + // One unacknowledged version must be replayed in full, including all disjoint clear fragments. + co_await consumeExpectedVersion(consumer, cleared, expected); + co_await timeoutError(consumer->acknowledge(), operationTimeout); + const CDCCursor checkpoint = consumer->position(); + ASSERT_EQ(checkpoint.streamId, streamId); + const std::vector acknowledged = + co_await timeoutError(listNativeCdcStreamsClient(cx), operationTimeout); + const auto acknowledgedStream = std::find_if( + acknowledged.begin(), acknowledged.end(), [&](const auto& stream) { return stream.name == name; }); + ASSERT(acknowledgedStream != acknowledged.end()); + ASSERT_EQ(acknowledgedStream->ranges, ranges); + ASSERT_EQ(acknowledgedStream->minVersion, checkpoint.lastConsumedVersion + 1); + // The multi-range stream must retain unread history without another stream holding its tag back. + co_await timeoutError(removeNativeCdcStreamClient(cx, gapName), operationTimeout); + + values.clear(); + expected = Standalone>(); + for (StringRef suffix : { "b"_sr, "m"_sr, "w"_sr }) { + values.emplace_back(multipleRangeKey(suffix), "recovered-range"_sr); + expected.push_back_deep(expected.arena(), + MutationRef(MutationRef::SetValue, values.back().first, values.back().second)); + } + values.emplace_back(multipleRangeKey("h"_sr), "excluded-gap"_sr); + const Version retained = co_await writeValues(cx, values); + co_await timeoutError(forceTransactionSystemRecovery(), operationTimeout); + consumer = resumeNativeCdcConsumer(cx, checkpoint); + co_await consumeExpectedVersion(consumer, retained, expected); + co_await timeoutError(consumer->acknowledge(), operationTimeout); + + // A new commit must use the recovered range routing, including exclusion of the untracked gap. + const Version resumed = co_await writeValues(cx, values); + co_await consumeExpectedVersion(consumer, resumed, expected); + co_await timeoutError(consumer->acknowledge(), operationTimeout); + co_await timeoutError(removeNativeCdcStreamClient(cx, name), operationTimeout); + co_await timeoutError(waitForRetiredTagCleanup(cx), operationTimeout); + co_await timeoutError(waitForFullyRecovered(), operationTimeout); + CODE_PROBE(true, "Native CDC consumes and recovers one stream covering multiple disjoint ranges"); + } + Future validateAssignmentPublication(Database cx) { for (int check = 0; check < assignmentPublicationChecks; ++check) { co_await validateAssignmentPublicationOnce(cx, check); @@ -469,6 +1181,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { const Key name = "native-cdc-e2e/stale-initialization"_sr; const KeyRange keys( KeyRangeRef("native-cdc-e2e/stale-initialization/"_sr, "native-cdc-e2e/stale-initialization0"_sr)); + const std::vector ranges{ keys }; const double deadline = now() + operationTimeout; bool recovering = false; bool streamRegistered = false; @@ -488,7 +1201,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { co_await timeoutError(setAllProxyPopsPaused(cx, true), operationTimeout); streamRegistered = true; const CDCStreamId streamId = - co_await timeoutError(registerNativeCdcStreamClient(cx, name, keys), operationTimeout); + co_await timeoutError(registerNativeCdcStreamClient(cx, name, ranges), operationTimeout); CDCProxyInterface proxy = co_await timeoutError(waitForAssignedProxy(cx, streamId), operationTimeout); const CDCCursor cursor(streamId, invalidVersion); Future> pendingConsume = proxy.consume.tryGetReply(CDCConsumeRequest(cursor)); @@ -540,8 +1253,9 @@ class NativeCdcEndToEndWorkload : public TestWorkload { const Key key = "native-cdc-e2e/proxy-replacement/value"_sr; const Value value = "replacement-value"_sr; + const std::vector ranges{ keys }; const CDCStreamId streamId = - co_await timeoutError(registerNativeCdcStreamClient(cx, name, keys), operationTimeout); + co_await timeoutError(registerNativeCdcStreamClient(cx, name, ranges), operationTimeout); Reference consumer = co_await timeoutError(createNativeCdcConsumer(cx, name), operationTimeout); CDCProxyInterface original = co_await timeoutError(waitForAssignedProxy(cx, streamId), operationTimeout); @@ -569,6 +1283,244 @@ class NativeCdcEndToEndWorkload : public TestWorkload { co_await timeoutError(removeNativeCdcStreamClient(cx, name), operationTimeout); } + Future validateProxyRebalance(Database cx) { + ASSERT_EQ(streams.size(), 3); + const CDCStreamId firstId = streams[0].consumer->position().streamId; + const CDCStreamId otherTagId = streams[1].consumer->position().streamId; + const CDCStreamId sharedTagId = streams[2].consumer->position().streamId; + const auto proxies = cx->clientInfo->get().cdcProxies; + ASSERT_EQ(proxies.size(), 2); + // Registration commits durable ownership before the controller publishes each assignment. + for (const auto& stream : streams) { + co_await timeoutError(waitForAssignedProxy(cx, stream.consumer->position().streamId), operationTimeout); + } + const NativeCdcStatus initial = co_await timeoutError(getNativeCdcStatus(cx), operationTimeout); + ASSERT(initial.metadataComplete); + ASSERT_EQ(initial.tagCount, 2); + ASSERT_EQ(initial.streams.size(), 3); + const auto findStream = [](NativeCdcStatus const& status, CDCStreamId id) -> NativeCdcStreamStatus const& { + const auto stream = std::find_if(status.streams.begin(), status.streams.end(), [&](auto const& candidate) { + return candidate.info.streamId == id; + }); + ASSERT(stream != status.streams.end()); + return *stream; + }; + const auto& first = findStream(initial, firstId); + const auto& otherTag = findStream(initial, otherTagId); + const auto& shared = findStream(initial, sharedTagId); + ASSERT_EQ(first.tags.size(), 1); + ASSERT_EQ(otherTag.tags.size(), 1); + ASSERT_EQ(shared.tags.size(), 1); + const Tag tag = first.tags.front(); + ASSERT_EQ(shared.tags.front(), tag); + ASSERT_NE(otherTag.tags.front(), tag); + co_await checkTagOwner(cx, tag, firstId); + co_await checkTagOwner(cx, otherTag.tags.front(), otherTagId); + + const CDCProxyInterface source = co_await timeoutError(waitForAssignedProxy(cx, firstId), operationTimeout); + const CDCProxyInterface target = proxies[proxies.front().id() == source.id() ? 1 : 0]; + ASSERT_NE(source.id(), target.id()); + for (const auto& stream : initial.streams) { + ASSERT(stream.owner.present()); + ASSERT_EQ(stream.owner.get(), source.id()); + ASSERT(stream.ownerPublished); + } + + const Key key = keyForIndex(keyCount / 2); + const Value beforeMove = "native-cdc-rebalance-before"_sr; + const Version beforeVersion = co_await writeValue(cx, key, beforeMove); + co_await consumeThroughValue(streams[0].consumer, beforeVersion, key, beforeMove); + co_await consumeThroughValue(streams[1].consumer, beforeVersion, key, beforeMove); + const auto containsValue = [](CDCConsumeReply const& reply, Version version, KeyRef key, ValueRef value) { + return std::any_of(reply.mutations.begin(), reply.mutations.end(), [&](auto const& versioned) { + return versioned.version == version && + std::any_of(versioned.mutations.begin(), versioned.mutations.end(), [&](auto const& mutation) { + return mutation.type == MutationRef::SetValue && mutation.param1 == key && + mutation.param2 == value; + }); + }); + }; + bool primed = false; + const double primeDeadline = now() + operationTimeout; + while (streams[2].consumer->position().lastConsumedVersion < beforeVersion) { + CDCConsumeReply reply = co_await timeoutError(streams[2].consumer->consume(), operationTimeout); + primed |= containsValue(reply, beforeVersion, key, beforeMove); + ASSERT_LT(now(), primeDeadline); + } + ASSERT(primed); + // Leave this stream unacknowledged so its old tag data must remain readable by the new owner. + Future pending = streams[0].consumer->consume(); + ASSERT(!pending.isReady()); + + std::vector availableProxies{ proxies[0].id(), proxies[1].id() }; + ASSERT(co_await timeoutError(rebalanceNativeCdcProxyAssignments(cx, availableProxies, [] { return true; }), + operationTimeout)); + const CDCProxyInterface moved = + co_await timeoutError(waitForAssignedProxy(cx, firstId, source.id()), operationTimeout); + ASSERT_EQ(moved.id(), target.id()); + const ClientDBInfo& published = cx->clientInfo->get(); + ASSERT_EQ(published.cdcProxies, proxies); + ASSERT_EQ(published.streamToCDCProxyId.at(firstId), target.id()); + ASSERT_EQ(published.streamToCDCProxyId.at(sharedTagId), target.id()); + ASSERT_EQ(published.streamToCDCProxyId.at(otherTagId), source.id()); + const NativeCdcStatus afterMove = co_await timeoutError(getNativeCdcStatus(cx), operationTimeout); + ASSERT(afterMove.metadataComplete); + for (CDCStreamId id : { firstId, sharedTagId }) { + const auto& stream = findStream(afterMove, id); + ASSERT(stream.owner.present()); + ASSERT_EQ(stream.owner.get(), target.id()); + ASSERT(stream.ownerPublished); + } + const auto& untouched = findStream(afterMove, otherTagId); + ASSERT(untouched.owner.present()); + ASSERT_EQ(untouched.owner.get(), source.id()); + ASSERT(untouched.ownerPublished); + co_await checkTagOwner(cx, tag, firstId); + const auto blockingTag = std::find_if( + afterMove.tags.begin(), afterMove.tags.end(), [&](auto const& state) { return state.tag == tag; }); + ASSERT(blockingTag != afterMove.tags.end()); + ASSERT_LE(blockingTag->safePopVersion, beforeVersion); + ASSERT(std::find(blockingTag->blockingStreams.begin(), blockingTag->blockingStreams.end(), sharedTagId) != + blockingTag->blockingStreams.end()); + ASSERT_EQ(afterMove.proxies.size(), 2); + for (const auto& proxy : afterMove.proxies) { + ASSERT(proxy.sample.present()); + } + ASSERT(!(co_await timeoutError(rebalanceNativeCdcProxyAssignments(cx, availableProxies, [] { return true; }), + operationTimeout))); + + const ErrorOr staleAck = + co_await timeoutError(source.ack.tryGetReply(CDCAckRequest(firstId, beforeVersion)), operationTimeout); + ASSERT(!staleAck.present()); + ASSERT_EQ(staleAck.getError().code(), error_code_wrong_shard_server); + bool replayed = false; + const double replayDeadline = now() + operationTimeout; + do { + ASSERT_LT(now(), replayDeadline); + CDCConsumeReply replay = co_await timeoutError(streams[2].consumer->consume(), replayDeadline - now()); + replayed |= containsValue(replay, beforeVersion, key, beforeMove); + } while (streams[2].consumer->position().lastConsumedVersion < beforeVersion); + ASSERT(replayed); + co_await timeoutError(streams[2].consumer->acknowledge(), operationTimeout); + const NativeCdcStatus afterAck = co_await timeoutError(getNativeCdcStatus(cx), operationTimeout); + const auto advancedTag = std::find_if( + afterAck.tags.begin(), afterAck.tags.end(), [&](auto const& state) { return state.tag == tag; }); + ASSERT(advancedTag != afterAck.tags.end()); + ASSERT_GT(advancedTag->safePopVersion, beforeVersion); + + const Value afterMoveValue = "native-cdc-rebalance-after"_sr; + const Version afterVersion = co_await writeValue(cx, key, afterMoveValue); + bool pendingObserved = false; + const double deliveryDeadline = now() + operationTimeout; + while (!pendingObserved) { + CDCConsumeReply reply = co_await timeoutError(pending, operationTimeout); + pendingObserved = containsValue(reply, afterVersion, key, afterMoveValue); + co_await timeoutError(streams[0].consumer->acknowledge(), operationTimeout); + ASSERT_LT(now(), deliveryDeadline); + if (!pendingObserved) { + pending = streams[0].consumer->consume(); + } + } + co_await consumeThroughValue(streams[1].consumer, afterVersion, key, afterMoveValue); + co_await consumeThroughValue(streams[2].consumer, afterVersion, key, afterMoveValue); + for (const auto& stream : streams) { + co_await timeoutError(removeNativeCdcStreamClient(cx, stream.name), operationTimeout); + } + streams.clear(); + co_await timeoutError(waitForRetiredTagCleanup(cx), operationTimeout); + CODE_PROBE(true, "Native CDC rebalances a shared tag across live proxies without losing delivery"); + } + + Future validateAutomaticProxyRebalance(Database cx) { + ASSERT_EQ(streams.size(), 3); + const CDCStreamId firstId = streams[0].consumer->position().streamId; + const CDCStreamId otherTagId = streams[1].consumer->position().streamId; + const CDCStreamId sharedTagId = streams[2].consumer->position().streamId; + const auto proxies = cx->clientInfo->get().cdcProxies; + ASSERT_EQ(proxies.size(), 2); + const NativeCdcStatus initial = co_await timeoutError(getNativeCdcStatus(cx), operationTimeout); + ASSERT(initial.metadataComplete); + ASSERT_EQ(initial.tagCount, 2); + ASSERT_EQ(initial.streams.size(), 3); + const auto findStream = [](NativeCdcStatus const& status, CDCStreamId id) -> NativeCdcStreamStatus const& { + const auto stream = std::find_if(status.streams.begin(), status.streams.end(), [&](auto const& candidate) { + return candidate.info.streamId == id; + }); + ASSERT(stream != status.streams.end()); + return *stream; + }; + const auto& first = findStream(initial, firstId); + const auto& otherTag = findStream(initial, otherTagId); + const auto& shared = findStream(initial, sharedTagId); + ASSERT_EQ(first.tags.size(), 1); + ASSERT_EQ(otherTag.tags.size(), 1); + ASSERT_EQ(shared.tags.size(), 1); + ASSERT_EQ(first.tags.front(), shared.tags.front()); + ASSERT_NE(first.tags.front(), otherTag.tags.front()); + for (const auto& stream : initial.streams) { + ASSERT(stream.owner.present()); + ASSERT(std::any_of( + proxies.begin(), proxies.end(), [&](auto const& proxy) { return proxy.id() == stream.owner.get(); })); + } + ASSERT_EQ(first.owner.get(), shared.owner.get()); + + UID groupOwner; + UID otherOwner; + const double deadline = now() + operationTimeout; + while (true) { + Future changed = cx->clientInfo->onChange(); + const ClientDBInfo& published = cx->clientInfo->get(); + ASSERT_EQ(published.cdcProxies, proxies); + const auto group = published.streamToCDCProxyId.find(firstId); + const auto sharedGroup = published.streamToCDCProxyId.find(sharedTagId); + const auto other = published.streamToCDCProxyId.find(otherTagId); + const auto isLiveProxy = [&](UID id) { + return std::any_of(proxies.begin(), proxies.end(), [&](auto const& proxy) { return proxy.id() == id; }); + }; + if (group != published.streamToCDCProxyId.end() && sharedGroup != published.streamToCDCProxyId.end() && + other != published.streamToCDCProxyId.end() && group->second == sharedGroup->second && + group->second != other->second && isLiveProxy(group->second) && isLiveProxy(other->second)) { + groupOwner = group->second; + otherOwner = other->second; + break; + } + ASSERT_LT(now(), deadline); + co_await timeoutError(changed, deadline - now()); + } + if (first.owner.get() == otherTag.owner.get()) { + ASSERT_NE(groupOwner, first.owner.get()); + ASSERT_EQ(otherOwner, otherTag.owner.get()); + } else { + ASSERT_EQ(groupOwner, first.owner.get()); + ASSERT_EQ(otherOwner, otherTag.owner.get()); + } + const NativeCdcStatus afterMove = co_await timeoutError(getNativeCdcStatus(cx), operationTimeout); + ASSERT(afterMove.metadataComplete); + ASSERT_EQ(afterMove.streams.size(), 3); + for (const auto& stream : afterMove.streams) { + ASSERT(stream.owner.present()); + ASSERT_EQ(stream.owner.get(), stream.info.streamId == otherTagId ? otherOwner : groupOwner); + ASSERT(stream.ownerPublished); + } + co_await checkTagOwner(cx, first.tags.front(), firstId); + co_await checkTagOwner(cx, otherTag.tags.front(), otherTagId); + ASSERT_EQ(afterMove.proxies.size(), 2); + for (const auto& proxy : afterMove.proxies) { + ASSERT(proxy.sample.present()); + } + + const Key key = keyForIndex(keyCount / 2); + const Value value = "native-cdc-automatic-rebalance"_sr; + const Version committed = co_await writeValue(cx, key, value); + for (const auto& stream : streams) { + co_await consumeThroughValue(stream.consumer, committed, key, value); + co_await timeoutError(removeNativeCdcStreamClient(cx, stream.name), operationTimeout); + } + streams.clear(); + co_await timeoutError(waitForRetiredTagCleanup(cx), operationTimeout); + CODE_PROBE(true, "Native CDC controller rebalances a whole tag without proxy replacement"); + } + Future checkTagOwner(Database cx, Tag tag, Optional expected) { Transaction tr(cx); while (true) { @@ -614,6 +1566,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { Key name, KeyRange keys, Optional sharedTagStream = Optional()) { + const std::vector ranges{ keys }; const double deadline = now() + operationTimeout; CDCStreamId streamId = 0; if (sharedTagStream.present()) { @@ -637,7 +1590,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { ASSERT_NE(proxy.id(), anchor->second); try { Future> request = - proxy.registerStream.tryGetReply(CDCRegisterStreamRequest(name, keys)); + proxy.registerStream.tryGetReply(CDCRegisterStreamRequest(name, ranges)); // A different stream's assignment publication must not discard an in-flight registration. while (true) { auto result = @@ -666,7 +1619,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { co_await delay(0.1); } } else { - streamId = co_await timeoutError(registerNativeCdcStreamClient(cx, name, keys), deadline - now()); + streamId = co_await timeoutError(registerNativeCdcStreamClient(cx, name, ranges), deadline - now()); } co_await timeoutError(waitForAssignedProxy(cx, streamId), deadline - now()); // Recovery can replace the owner while registration or consumption is in flight. Compare live streams @@ -691,8 +1644,9 @@ class NativeCdcEndToEndWorkload : public TestWorkload { const Key thirdName = "native-cdc-e2e/tag-owner/third"_sr; const Key key = "native-cdc-e2e/tag-owner/data"_sr; const KeyRange keys(KeyRangeRef(key, keyAfter(key))); + const std::vector ranges{ keys }; const CDCStreamId firstId = - co_await timeoutError(registerNativeCdcStreamClient(cx, firstName, keys), operationTimeout); + co_await timeoutError(registerNativeCdcStreamClient(cx, firstName, ranges), operationTimeout); CDCProxyInterface owner = co_await timeoutError(waitForAssignedProxy(cx, firstId), operationTimeout); ASSERT_EQ(cx->clientInfo->get().cdcProxies.size(), 2); co_await checkTagOwner(cx, tag, firstId); @@ -793,8 +1747,9 @@ class NativeCdcEndToEndWorkload : public TestWorkload { const Key key = "native-cdc-e2e/stale-ownership/value"_sr; const Value value = "replacement-survives-stale-removal"_sr; const double deadline = now() + operationTimeout; + const std::vector ranges{ keys }; CDCStreamId expectedStreamId = - co_await timeoutError(registerNativeCdcStreamClient(cx, name, keys), operationTimeout); + co_await timeoutError(registerNativeCdcStreamClient(cx, name, ranges), operationTimeout); const auto rejectedWrongOwner = [](ErrorOr const& result) { ASSERT(!result.present()); @@ -890,7 +1845,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { co_await timeoutError(removeNativeCdcStreamClient(cx, name), operationTimeout); const CDCStreamId replacementId = - co_await timeoutError(registerNativeCdcStreamClient(cx, name, keys), operationTimeout); + co_await timeoutError(registerNativeCdcStreamClient(cx, name, ranges), operationTimeout); ASSERT_NE(replacementId, streamId); expectedStreamId = replacementId; const NativeCdcRemoveResult guardedRemoval = @@ -1030,29 +1985,6 @@ class NativeCdcEndToEndWorkload : public TestWorkload { co_return proxyStatus.second; } - Future startBlockedConsume(Database cx, - CDCStreamId streamId, - Reference consumer, - CDCProxyInterface proxy, - Future* outstanding) { - *outstanding = consumer->consume(); - const double deadline = now() + operationTimeout; - while (true) { - CDCProxyBufferStatus status = co_await getCurrentProxyStatus(cx, streamId, &proxy); - if (outstanding->isReady()) { - co_await *outstanding; - co_await timeoutError(consumer->acknowledge(), operationTimeout); - *outstanding = consumer->consume(); - continue; - } - if (status.activeConsumeRequests > 0 && status.readDemand > 0) { - co_return; - } - ASSERT_LT(now(), deadline); - co_await delay(0.01); - } - } - Future waitForNoActiveConsumes(Database cx, CDCStreamId streamId, CDCProxyInterface* proxy) { const double deadline = now() + operationTimeout; while (true) { @@ -1073,7 +2005,9 @@ class NativeCdcEndToEndWorkload : public TestWorkload { error = e; } ASSERT(error.present()); - ASSERT_EQ(error.get().code(), error_code_client_invalid_operation); + if (error.get().code() != error_code_client_invalid_operation) { + throw error.get(); + } } Future validateConsumeLeaseAndExclusivity(Database cx, CDCStreamId streamId, CDCProxyInterface* proxy) { @@ -1082,8 +2016,10 @@ class NativeCdcEndToEndWorkload : public TestWorkload { // the tracked consumer so later workload phases do not retain a cursor behind acknowledgements made here. Reference idleConsumer = streams.front().consumer; const Version idleStartVersion = idleConsumer->position().lastConsumedVersion; - Future idleConsume; - co_await startBlockedConsume(cx, streamId, idleConsumer, *proxy, &idleConsume); + // Check client exclusivity before yielding: committed-version progress can complete a consume before + // a status request observes read demand, even when the client correctly rejects overlapping operations. + Future idleConsume = idleConsumer->consume(); + ASSERT(!idleConsume.isReady()); Future overlappingConsume = idleConsumer->consume(); ASSERT(overlappingConsume.isReady() && overlappingConsume.isError()); ASSERT_EQ(overlappingConsume.getError().code(), error_code_client_invalid_operation); @@ -1096,8 +2032,8 @@ class NativeCdcEndToEndWorkload : public TestWorkload { for (int i = 0; i < 4; ++i) { Key name = Key(StringRef(format("native-cdc-e2e/lease/%04d", i))); Key key = keyForIndex(keyCount / 2); - co_await timeoutError(registerNativeCdcStreamClient(cx, name, KeyRangeRef(key, keyAfter(key))), - operationTimeout); + const std::vector ranges{ KeyRangeRef(key, keyAfter(key)) }; + co_await timeoutError(registerNativeCdcStreamClient(cx, name, ranges), operationTimeout); CDCProxyBufferStatus status = co_await getCurrentProxyStatus(cx, streamId, proxy); ASSERT_LE(status.activeConsumeRequests, 1); ASSERT_LE(status.readDemand, 1); @@ -1120,13 +2056,49 @@ class NativeCdcEndToEndWorkload : public TestWorkload { co_await waitForNoActiveConsumes(cx, streamId, proxy); CDCCursor currentCursor = idleConsumer->position(); - // Send both requests without yielding. The first request marks the stream active before its metadata read, so - // the second request deterministically exercises server-side exclusivity even while versions advance. - co_await getCurrentProxyStatus(cx, streamId, proxy); - Future> first = proxy->consume.tryGetReply(CDCConsumeRequest(currentCursor)); - co_await expectConcurrentConsumeRejected(*proxy, currentCursor); - first.cancel(); - co_await waitForNoActiveConsumes(cx, streamId, proxy); + const double deadline = now() + operationTimeout; + while (true) { + ASSERT_LT(now(), deadline); + try { + // Send both requests without yielding. The first request marks the stream active before its metadata + // read, so the second request deterministically exercises server-side exclusivity even while versions + // advance. + co_await getCurrentProxyStatus(cx, streamId, proxy); + Future> first = proxy->consume.tryGetReply(CDCConsumeRequest(currentCursor)); + co_await timeoutError(expectConcurrentConsumeRejected(*proxy, currentCursor), operationTimeout); + first.cancel(); + co_await waitForNoActiveConsumes(cx, streamId, proxy); + + // The first request may finish before the retry reaches the proxy. A pending request is superseded, + // while an already-completed request retains its reply; either ordering must allow the same consumer to + // retry. + const UID consumerId = deterministicRandom()->randomUniqueID(); + Future> original = + proxy->consume.tryGetReply(CDCConsumeRequest(currentCursor, consumerId)); + Future> retry = + proxy->consume.tryGetReply(CDCConsumeRequest(currentCursor, consumerId)); + const ErrorOr firstReply = co_await timeoutError(original, operationTimeout); + if (firstReply.isError()) { + if (firstReply.getError().code() != error_code_request_maybe_delivered) { + throw firstReply.getError(); + } + } else { + ASSERT_GE(firstReply.get().lastConsumedVersion, currentCursor.lastConsumedVersion); + } + co_await timeoutError(throwErrorOr(retry), operationTimeout); + co_await waitForNoActiveConsumes(cx, streamId, proxy); + co_return; + } catch (Error& e) { + if (e.code() != error_code_wrong_shard_server && e.code() != error_code_broken_promise && + e.code() != error_code_connection_failed && e.code() != error_code_request_maybe_delivered) { + throw; + } + // A status reply cannot prevent a proxy replacement or disconnect before the consume replies. + // Retry both requests so transport failure cannot count as evidence of exclusivity. + CODE_PROBE(true, "Native CDC server consume validation retries after proxy request failure"); + } + co_await waitForNoActiveConsumes(cx, streamId, proxy); + } } Future requestPopsUntilStopped(Database cx, Reference> stopped) { @@ -1142,14 +2114,18 @@ class NativeCdcEndToEndWorkload : public TestWorkload { auto initialProxyStatus = co_await timeoutError(getAssignedProxyStatus(cx, streamId), operationTimeout); bool followedProxyReplacement = proxy->id() != initialProxyStatus.first.id(); updateObservedProxy(*proxy, initialProxyStatus.first); - const CDCProxyBufferStatus initial = initialProxyStatus.second; + CDCProxyBufferStatus initial = initialProxyStatus.second; auto stopped = makeReference>(false); Future requester = requestPopsUntilStopped(cx, stopped); const double deadline = now() + operationTimeout; while (true) { const UID previousProxy = proxy->id(); const CDCProxyBufferStatus status = co_await getCurrentProxyStatus(cx, streamId, proxy); - followedProxyReplacement |= previousProxy != proxy->id(); + if (previousProxy != proxy->id()) { + // Pop counters belong to one proxy instance; require fresh progress after replacement. + initial = status; + followedProxyReplacement = true; + } if (status.popCompletions > initial.popCompletions) { ASSERT_GT(status.popRequests, initial.popRequests); break; @@ -1593,7 +2569,8 @@ class NativeCdcEndToEndWorkload : public TestWorkload { const NativeCdcStreamStatus& blocker = blockedStatus.streams.front(); ASSERT_EQ(blocker.info.streamId, streams.front().consumer->position().streamId); ASSERT_EQ(blocker.info.name, streams.front().name); - ASSERT_EQ(blocker.info.keys, keys); + ASSERT_EQ(blocker.info.ranges.size(), 1); + ASSERT_EQ(blocker.info.ranges.front(), keys); ASSERT_GT(blocker.info.minVersion, invalidVersion); ASSERT_LE(blocker.info.minVersion, committed); ASSERT_EQ(blocker.tags.size(), 1); @@ -1755,9 +2732,84 @@ class NativeCdcEndToEndWorkload : public TestWorkload { CODE_PROBE(true, "Native CDC retired tag cleanup allows recovery to complete"); } + Future prepareRetaggedRestartState(Database cx, Version before) { + ASSERT_EQ(cx->clientInfo->get().nativeCdcTagCount, 2); + const RetagSnapshot original = co_await readRetagSnapshot(cx, 0, 1); + const Tag destination(tagLocalityCDC, original.state.assignment.tag.id == 0 ? 1 : 0); + const RetagSnapshot committed = co_await commitRetagFixture(cx, 0, original, destination, 1, {}); + const Version cutover = committed.state.assignment.version; + const Version after = co_await writeValue(cx, keyForIndex(keyCount / 2), "native-cdc-restart-after-retag"_sr); + ASSERT_LT(before, cutover); + ASSERT_GT(after, cutover); + + // Keep exact marker versions outside the tracked range so the restarted reader can verify both log intervals. + BinaryWriter fixture{ Unversioned() }; + fixture << original.state.streamId << before << cutover << after << original.state.assignment.tag + << destination; + co_await writeValue(cx, "native-cdc-e2e/restart-retag-state"_sr, fixture.toValue()); + const RetagSnapshot pending = co_await readRetagSnapshot(cx, 0); + ASSERT(pending.state.pending); + ASSERT_EQ(pending.history.size(), 2); + ASSERT_EQ(pending.state.assignment.version, cutover); + ASSERT_LT(pending.state.minVersion, cutover); + CODE_PROBE(true, "Native CDC restart preserves an unacknowledged retag boundary"); + } + + Future loadRetaggedRestartState(Database cx, Key name, Reference consumer) { + const std::vector listed = + co_await timeoutError(listNativeCdcStreamsClient(cx), operationTimeout); + ASSERT_EQ(listed.size(), 1); + ASSERT_EQ(listed.front().name, name); + ASSERT_EQ(listed.front().streamId, consumer->position().streamId); + StreamState stream; + stream.name = name; + ASSERT_EQ(listed.front().ranges.size(), 1); + stream.keys = listed.front().ranges.front(); + stream.consumer = consumer; + streams.push_back(std::move(stream)); + + RetagRestartMarkers markers; + // NOLINTNEXTLINE(cppcoreguidelines-avoid-capturing-lambda-coroutines) Database::run owns the closure. + co_await cx.run([this, &markers, &consumer, &cx](Transaction* tr) -> Future { + const Optional fixture = co_await tr->get("native-cdc-e2e/restart-retag-state"_sr); + ASSERT(fixture.present()); + BinaryReader reader(fixture.get(), Unversioned()); + reader >> markers.streamId >> markers.before >> markers.cutover >> markers.after >> markers.oldTag >> + markers.newTag; + ASSERT_EQ(markers.streamId, consumer->position().streamId); + const RetagSnapshot pending = co_await readRetagSnapshot(cx, 0); + ASSERT(pending.state.pending); + ASSERT_EQ(pending.history.size(), 2); + ASSERT_EQ(pending.history.front().tag, markers.oldTag); + ASSERT_EQ(pending.state.assignment.tag, markers.newTag); + ASSERT_EQ(pending.state.assignment.version, markers.cutover); + ASSERT_LT(pending.state.minVersion, markers.cutover); + }); + co_return markers; + } + + Future finishRetaggedRestartState(Database cx, RetagRestartMarkers markers) { + const RetagSnapshot acknowledged = co_await readRetagSnapshot(cx, 0); + co_await waitForCanonicalRetag(cx, 0, acknowledged.state.assignment); + // NOLINTNEXTLINE(cppcoreguidelines-avoid-capturing-lambda-coroutines) Database::run owns the closure. + co_await cx.run([this, markers](Transaction* tr) -> Future { + tr->setOption(FDBTransactionOptions::LOCK_AWARE); + tr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); + const RetagSnapshot completed = co_await readRetagSnapshot(tr, 0, 1); + ASSERT(!completed.state.pending); + ASSERT_EQ(completed.state.assignment.version, markers.cutover); + const bool prepared = co_await retagNativeCdcStream(tr, completed.state, markers.oldTag); + ASSERT(!prepared); + }); + CODE_PROBE(true, "Native CDC disabled admission finishes pending retags without admitting new moves"); + } + Future prepareRestartDrainState(Database cx) { ASSERT_EQ(streams.size(), 1); - co_await writeValue(cx, keyForIndex(keyCount / 2), "native-cdc-restart-drain"_sr); + const Version before = co_await writeValue(cx, keyForIndex(keyCount / 2), "native-cdc-restart-drain"_sr); + if (testRetaggedRestart) { + co_await timeoutError(prepareRetaggedRestartState(cx, before), operationTimeout); + } CODE_PROBE(true, "Native CDC restart marker is durable before save and kill"); } @@ -1779,18 +2831,44 @@ class NativeCdcEndToEndWorkload : public TestWorkload { ASSERT_EQ(activeStatus.streams.size(), 1); ASSERT_EQ(activeStatus.streams.front().info.name, name); ASSERT_EQ(activeStatus.streams.front().info.streamId, consumer->position().streamId); + Optional retagMarkers; + if (testRetaggedRestart) { + retagMarkers = co_await timeoutError(loadRetaggedRestartState(cx, name, consumer), operationTimeout); + } bool observed = false; - while (!observed) { + bool observedAfterRetag = !testRetaggedRestart; + const double retagDeadline = now() + operationTimeout; + while (!observed || !observedAfterRetag) { + if (testRetaggedRestart) { + ASSERT_LT(now(), retagDeadline); + } CDCConsumeReply reply = co_await timeoutError(consumer->consume(), operationTimeout); for (const auto& versioned : reply.mutations) { for (const auto& mutation : versioned.mutations) { if (mutation.type == MutationRef::SetValue && mutation.param1 == keyForIndex(keyCount / 2) && mutation.param2 == "native-cdc-restart-drain"_sr) { - observed = true; + if (retagMarkers.present()) { + ASSERT_LT(versioned.version, retagMarkers.get().cutover); + observed |= versioned.version == retagMarkers.get().before; + } else { + observed = true; + } + } + if (retagMarkers.present() && mutation.type == MutationRef::SetValue && + mutation.param1 == keyForIndex(keyCount / 2) && + mutation.param2 == "native-cdc-restart-after-retag"_sr) { + ASSERT_GE(versioned.version, retagMarkers.get().cutover); + observedAfterRetag |= versioned.version == retagMarkers.get().after; } } } + if (!testRetaggedRestart) { + co_await timeoutError(consumer->acknowledge(), operationTimeout); + } + } + if (testRetaggedRestart) { co_await timeoutError(consumer->acknowledge(), operationTimeout); + co_await timeoutError(finishRetaggedRestartState(cx, retagMarkers.get()), operationTimeout); } const NativeCdcRemoveResult removed = co_await timeoutError( removeNativeCdcStreamGuarded(cx, name, consumer->position().streamId), operationTimeout); @@ -1809,9 +2887,9 @@ class NativeCdcEndToEndWorkload : public TestWorkload { Optional registrationError; try { - co_await timeoutError( - registerNativeCdcStreamClient(cx, "native-cdc-e2e/disabled-registration"_sr, normalKeys), - operationTimeout); + const std::vector ranges{ normalKeys }; + co_await timeoutError(registerNativeCdcStreamClient(cx, "native-cdc-e2e/disabled-registration"_sr, ranges), + operationTimeout); } catch (Error& e) { registrationError = e; } @@ -1892,6 +2970,26 @@ class NativeCdcEndToEndWorkload : public TestWorkload { } Future run(Database cx) { + if (testRetagCompatibility) { + co_await validateRetagCompatibility(cx); + co_return; + } + if (testRetaggingMemoryBound) { + co_await validateRetaggingMemoryBound(cx); + co_return; + } + if (testProxyRebalanceAutomatic) { + co_await timeoutError(validateAutomaticProxyRebalance(cx), operationTimeout); + co_return; + } + if (testProxyRebalance) { + co_await timeoutError(validateProxyRebalance(cx), operationTimeout); + co_return; + } + if (testMultipleRanges) { + co_await validateMultipleRanges(cx); + co_return; + } if (testRetiredSharedTagSnapshot) { co_await validateRetiredSharedTagSnapshot(cx); co_return; @@ -1987,18 +3085,25 @@ class NativeCdcEndToEndWorkload : public TestWorkload { rounds = getOption(options, "rounds"_sr, 30); assignmentPublicationChecks = getOption(options, "assignmentPublicationChecks"_sr, 0); testProxyReplacement = getOption(options, "testProxyReplacement"_sr, false); + testProxyRebalance = getOption(options, "testProxyRebalance"_sr, false); + testProxyRebalanceAutomatic = getOption(options, "testProxyRebalanceAutomatic"_sr, false); testTagOwnership = getOption(options, "testTagOwnership"_sr, false); injectUndeliveredProxyHalt = getOption(options, "injectUndeliveredProxyHalt"_sr, false); testMemoryBound = getOption(options, "testMemoryBound"_sr, false); testReplyChunking = getOption(options, "testReplyChunking"_sr, false); + testMultipleRanges = getOption(options, "testMultipleRanges"_sr, false); testOversizedPeek = getOption(options, "testOversizedPeek"_sr, false); testDurableAckScan = getOption(options, "testDurableAckScan"_sr, false); testDelayedRetention = getOption(options, "testDelayedRetention"_sr, false); testRetiredRecovery = getOption(options, "testRetiredRecovery"_sr, false); blockRetiredPopWithLiveStream = getOption(options, "blockRetiredPopWithLiveStream"_sr, false); testRetiredSharedTagSnapshot = getOption(options, "testRetiredSharedTagSnapshot"_sr, false); + testRetagCompatibility = getOption(options, "testRetagCompatibility"_sr, false); + testRetaggingMemoryBound = getOption(options, "testRetaggingMemoryBound"_sr, false); prepareRestartDrain = getOption(options, "prepareRestartDrain"_sr, false); drainAfterRestart = getOption(options, "drainAfterRestart"_sr, false); + testRetaggedRestart = getOption(options, "testRetaggedRestart"_sr, false); + testRetagTransactionRetries = getOption(options, "testRetagTransactionRetries"_sr, false); memoryTestValueBytes = getOption(options, "memoryTestValueBytes"_sr, 1024); retentionValidationDelay = getOption(options, "retentionValidationDelay"_sr, 0.0); drainProbability = getOption(options, "drainProbability"_sr, 0.25); @@ -2011,13 +3116,24 @@ class NativeCdcEndToEndWorkload : public TestWorkload { ASSERT_GE(writesPerRound, 1); ASSERT_LE(writesPerRound, keyCount); ASSERT_GE(assignmentPublicationChecks, 0); + ASSERT(!(testProxyRebalance && testProxyRebalanceAutomatic)); + ASSERT(!(testProxyRebalance || testProxyRebalanceAutomatic) || initialStreamCount == 3); ASSERT(!injectUndeliveredProxyHalt || testProxyReplacement); ASSERT_GT(memoryTestValueBytes, 0); ASSERT_GE(retentionValidationDelay, 0.0); ASSERT(!(prepareRestartDrain && drainAfterRestart)); + ASSERT(!testRetaggedRestart || prepareRestartDrain || drainAfterRestart); + ASSERT(!testRetagTransactionRetries || testRetagCompatibility || testRetaggedRestart); ASSERT(!(testReplyChunking && (testOversizedPeek || testDurableAckScan))); ASSERT(!(testOversizedPeek && testDurableAckScan)); ASSERT(!(testRetiredSharedTagSnapshot && testRetiredRecovery)); + ASSERT(!(testRetagCompatibility && testMemoryBound)); + ASSERT(!testRetaggingMemoryBound || (!testRetagCompatibility && !testMemoryBound && !prepareRestartDrain && + !drainAfterRestart && initialStreamCount == 2 && keyCount >= 2)); + ASSERT(!testRetagCompatibility || + (initialStreamCount == 4 && keyCount >= 4 && memoryTestValueBytes >= 32 && !prepareRestartDrain && + !drainAfterRestart && !testRetiredSharedTagSnapshot && !testOversizedPeek && !testReplyChunking && + !testDurableAckScan)); ASSERT(!blockRetiredPopWithLiveStream || testRetiredRecovery); } @@ -2031,9 +3147,15 @@ class NativeCdcEndToEndWorkload : public TestWorkload { if (drainAfterRestart) { return Void(); } + if (testMultipleRanges) { + return Void(); + } if (prepareRestartDrain) { return prepareRestartDrainSetup(cx); } + if (testRetagCompatibility || testRetaggingMemoryBound) { + return initializeRetaggingStreams(cx); + } if (testRetiredSharedTagSnapshot) { return Void(); } diff --git a/fdbserver/workloads/PhysicalShardMove.cpp b/fdbserver/workloads/PhysicalShardMove.cpp index e23c9e7c642..b5fb9d3a296 100644 --- a/fdbserver/workloads/PhysicalShardMove.cpp +++ b/fdbserver/workloads/PhysicalShardMove.cpp @@ -25,7 +25,7 @@ #include "fdbserver/core/Knobs.h" #include "fdbserver/core/MoveKeys.h" #include "fdbserver/core/QuietDatabase.h" -#include "fdbserver/core/ServerCheckpoint.h" +#include "fdbserver/checkpoint/Checkpoint.h" #include "fdbserver/tester/workloads.h" #include "flow/Error.h" #include "flow/IRandom.h" diff --git a/fdbserver/workloads/StalePeerTest.cpp b/fdbserver/workloads/StalePeerTest.cpp new file mode 100644 index 00000000000..ae455504ae1 --- /dev/null +++ b/fdbserver/workloads/StalePeerTest.cpp @@ -0,0 +1,610 @@ +/* + * StalePeerTest.actor.cpp + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +// Test that verifies stale peer references are cleaned up after a process +// running a specific role is killed. Configurable via dstKillRole parameter. + +#include "fdbserver/tester/workloads.h" +#include "fdbserver/core/ServerDBInfo.h" +#include "fdbserver/core/QuietDatabase.h" +#include "fdbserver/core/FDBSimulatorProcessInfo.h" +#include "fdbrpc/SimulatorProcessInfo.h" +#include "fdbrpc/simulator.h" +#include "fdbrpc/FlowTransport.h" +#include "fdbclient/CoordinationInterface.h" +#include "fdbclient/ManagementAPI.h" +#include +#include "flow/CoroUtils.h" + +struct StalePeerTestWorkload : TestWorkload { + static constexpr auto NAME = "StalePeerTest"; + + double waitAfterKill; + std::string dstKillRole; + // Which source roles we inspect for stale peer references to the killed + // destination. "any" = every live process (server roles included). + // "tester_client" = only customer-client processes (ProcessClass::TesterClass, + // the tester processes running the workload). Server roles use the client + // library internally so their peer refs are affected by the client-side + // fixes too, but the contract we care about is that a pure customer client + // holds no stale ref to a killed client-facing role. + std::string srcCheckRole; + bool testPassed = false; + bool skippedNoTargets = false; + bool skippedClusterUnhealthy = false; + // Set when the chosen target was not referenced by any inspected source at + // kill time, so a post-kill Delta==0 would prove nothing (inconclusive). We + // skip rather than count such a run as a meaningful pass. + bool skippedVacuous = false; + // How many inspected source processes held a reference (tracked interface + // copy or live peer ref) to the killed address just before the kill. Logged + // for audit; the post-kill check is gated on this being > 0. + int preKillRefSources = 0; + + std::vector oldKillAddresses; + // Tracker role string for the killed interface, used purely for + // diagnostic per-role Delta reporting in the failure event. The pass + // criterion is the raw peer->peerReferences count, not the per-role + // Delta. Strings: + // "TLog" (tlog or log_router), "SS" (ss), + // "CP" (commit_proxy), "GP" (grv_proxy), + // "MS" (master), "RV" (resolver), + // "DD" (dd), "RK" (rk), "CC" (cluster_controller). + // Empty for coordinator (no service interface registered with the + // tracker -- diagnostic Delta dump is skipped). + std::string trackedDstRole; + + explicit StalePeerTestWorkload(WorkloadContext const& wcx) : TestWorkload(wcx), testPassed(true) { + waitAfterKill = getOption(options, "waitAfterKill"_sr, 60.0); + dstKillRole = getOption(options, "dstKillRole"_sr, "clientFacing"_sr).toString(); + srcCheckRole = getOption(options, "srcCheckRole"_sr, "any"_sr).toString(); + // dstKillRole="clientFacing" resolves (per-run, deterministically) to one + // of the client-facing destination roles -- the roles a client + // (DatabaseContext / its peer connections) directly addresses: coordinator, + // cluster controller, commit proxy, grv proxy, storage server. Same + // seed/config picks the same role, so per-run determinism is preserved + // while a single bulk ensemble covers all client-facing dst kills. + if (dstKillRole == "clientFacing") { + static const std::vector clientDstChoices = { + "coordinator", "cluster_controller", "commit_proxy", "grv_proxy", "ss" + }; + dstKillRole = clientDstChoices[deterministicRandom()->randomInt(0, clientDstChoices.size())]; + TraceEvent("StalePeerTestPickedClientDstRole").detail("DstKillRole", dstKillRole); + } + static const std::set validRoles = { + "tlog", "ss", "commit_proxy", "grv_proxy", "master", "resolver", "dd", + "rk", "coordinator", "log_router", "cluster_controller" + }; + if (!validRoles.contains(dstKillRole)) { + TraceEvent(SevError, "StalePeerTestInvalidDstKillRole") + .detail("DstKillRole", dstKillRole) + .detail("ValidOptions", + "clientFacing, tlog, ss, commit_proxy, grv_proxy, master, resolver, dd, rk, coordinator, " + "log_router, cluster_controller"); + ASSERT(false); + } + static const std::set validSrcChecks = { "any", "tester_client" }; + if (!validSrcChecks.contains(srcCheckRole)) { + TraceEvent(SevError, "StalePeerTestInvalidSrcCheckRole") + .detail("SrcCheckRole", srcCheckRole) + .detail("ValidOptions", "any, tester_client"); + ASSERT(false); + } + } + + Future setup(Database const& cx) override { return Void(); } + + Future start(Database const& cx) override { + if (clientId != 0) + return Void(); + return _start(this, cx); + } + + void disableFailureInjectionWorkloads(std::set& out) const override { out.insert("all"); } + + // Push addr into `out` only if it's valid and not protected by the simulator. + // sim2 silently refuses to kill addresses in protectedAddresses (coordinator + // majority, HTTP servers, etc.), and a silent no-op kill would trip the + // stuck-peer-ref check on an alive process. + void addIfKillable(std::vector& out, NetworkAddress addr) { + if (addr.isValid() && !g_simulator->isProtectedAddress(addr)) { + out.push_back(addr); + } + } + + // InterfaceTracker role string for the killed dst, used for the per-role + // Delta check and pre-kill reference snapshot. Deterministic from + // dstKillRole. Empty for coordinator: its endpoints use well-known tokens + // (ClientLeaderRegInterface) that are not registered with the tracker. + std::string computeTrackerRole() const { + if (dstKillRole == "tlog" || dstKillRole == "log_router") + return "TLog"; + if (dstKillRole == "ss") + return "SS"; + if (dstKillRole == "commit_proxy") + return "CP"; + if (dstKillRole == "grv_proxy") + return "GP"; + if (dstKillRole == "master") + return "MS"; + if (dstKillRole == "resolver") + return "RV"; + if (dstKillRole == "dd") + return "DD"; + if (dstKillRole == "rk") + return "RK"; + if (dstKillRole == "cluster_controller") + return "CC"; + return ""; // coordinator + } + + // Should this source process be inspected for stale refs? Mirrors the + // srcCheckRole filter used by both the pre-kill snapshot and the post-kill + // check so the two are always over the same population. "tester_client" + // keeps only customer-client (TesterClass) processes, excluding the + // simulator's internal TestSystem driver (IP 1.1.1.1), which is TesterClass + // but makes no workload transactions. + bool isInspectedSource(ISimulator::ProcessInfo* proc) const { + if (proc->failed || proc->rebooting) + return false; + if (srcCheckRole == "tester_client" && (getSimulatorProcessClass(proc) != ProcessClass::TesterClass || + proc->address.ip == IPAddress(0x01010101))) { + return false; + } + return proc->global(INetwork::enFlowTransport) != nullptr; + } + + // Count inspected source processes that currently hold a reference to `addr`: + // either a live tracked interface copy (per-role Delta > 0) or a live peer + // reference. Used pre-kill to establish non-vacuity (the target is actually + // referenced) and to pick the most-referenced storage server to kill. + int countSourcesReferencing(const NetworkAddress& addr, const std::string& trackerRole) const { + int n = 0; + for (auto* proc : g_simulator->getAllProcesses()) { + if (!isInspectedSource(proc)) + continue; + auto* transport = static_cast((void*)proc->global(INetwork::enFlowTransport)); + bool hasRef = !trackerRole.empty() && transport->interfaceTracker.getDelta(addr, trackerRole) > 0; + if (!hasRef) { + const auto& allPeers = transport->getAllPeers(); + auto it = allPeers.find(addr); + hasRef = (it != allPeers.end() && it->second->peerReferences > 0); + } + if (hasRef) + ++n; + } + return n; + } + + // Find a process address for the given role. + std::vector findAddressesForRole(Database const& cx) { + std::vector result; + const auto& info = dbInfo->get(); + if (dstKillRole == "tlog") { + for (const auto& tlogset : info.logSystemConfig.tLogs) { + if (!tlogset.isLocal) + continue; + for (const auto& log : tlogset.tLogs) { + if (log.present()) + addIfKillable(result, log.interf().address()); + } + } + } else if (dstKillRole == "log_router") { + // Log routers live on remote DCs in fearless configs. In a + // single-region cluster there are none, in which case the + // empty-target skip path below will treat this as a no-op. + for (const auto& tlogset : info.logSystemConfig.tLogs) { + for (const auto& lr : tlogset.logRouters) { + if (lr.present()) + addIfKillable(result, lr.interf().address()); + } + } + } else if (dstKillRole == "ss") { + // SS targets are resolved in _start from the authoritative recruited + // set via getStorageServers(cx) (see the ss path there), not here. + // Scanning simulator processes by StorageClass/UnsetClass could pick a + // process that hosts no recruited SS, in which case no SS interface is + // ever tracked at that address and the per-role Delta check passes + // vacuously. Leaving this empty; _start handles ss before calling. + } else if (dstKillRole == "commit_proxy") { + for (const auto& cp : info.client.commitProxies) { + addIfKillable(result, cp.address()); + } + trackedDstRole = "CP"; + } else if (dstKillRole == "grv_proxy") { + for (const auto& gp : info.client.grvProxies) { + addIfKillable(result, gp.address()); + } + trackedDstRole = "GP"; + } else if (dstKillRole == "master") { + addIfKillable(result, info.master.address()); + } else if (dstKillRole == "resolver") { + for (const auto& rv : info.resolvers) { + addIfKillable(result, rv.address()); + } + } else if (dstKillRole == "dd") { + if (info.distributor.present()) + addIfKillable(result, info.distributor.get().address()); + } else if (dstKillRole == "rk") { + if (info.ratekeeper.present()) + addIfKillable(result, info.ratekeeper.get().address()); + } else if (dstKillRole == "cluster_controller") { + addIfKillable(result, info.clusterInterface.address()); + } else if (dstKillRole == "coordinator") { + // sim2 keeps a majority of coordinators in protectedAddresses. Any + // un-protected coordinator is a valid kill target. With + // coordinators=3 (set in StalePeerTest.toml), 2 are protected and + // the third is killable. Resolve hostnames too -- simulation + // connection strings use hostnames (e.g. fakeCoordinatorDC0M0:1) + // rather than direct NetworkAddresses, so cs.coords is typically + // empty and the candidates live in cs.hostnames. + auto connRecord = cx->getConnectionRecord(); + if (connRecord) { + auto cs = connRecord->getConnectionString(); + for (const auto& addr : cs.coords) { + addIfKillable(result, addr); + } + for (const auto& hn : cs.hostnames) { + Optional resolved = hn.resolveBlocking(); + if (resolved.present()) { + addIfKillable(result, resolved.get()); + } + } + } + } + return result; + } + + static Future _start(StalePeerTestWorkload* self, Database cx) { + // Wait for cluster to stabilize + co_await delay(10.0); + while (self->dbInfo->get().recoveryState < RecoveryState::FULLY_RECOVERED) { + co_await self->dbInfo->onChange(); + } + + // Tracker role for the killed dst (deterministic from dstKillRole), used + // for both the pre-kill reference snapshot and the post-kill Delta check. + std::string trackerRole = self->computeTrackerRole(); + + // Resolve kill targets. + std::vector targetAddresses; + if (self->dstKillRole == "ss") { + // #1 non-vacuity: target an ACTUAL recruited storage server from the + // cluster's authoritative serverList, not just any StorageClass + // process. Killing a process that hosts no recruited SS would leave no + // SS interface tracked at that address, so the per-role Delta check + // would pass without ever exercising the client-side eviction path. + std::vector ssis = co_await getStorageServers(cx); + for (const auto& ssi : ssis) { + self->addIfKillable(targetAddresses, ssi.address()); + } + } else { + targetAddresses = self->findAddressesForRole(cx); + } + + if (targetAddresses.empty()) { + // No killable addresses for this role (e.g. every coordinator / + // proxy is protected, or log_routers don't exist in a + // single-region cluster). Skip the test -- there's nothing to + // leak refs for. + TraceEvent("StalePeerTestNoTargets") + .detail("Role", self->dstKillRole) + .detail("Note", "No killable addresses found; treating as skip"); + self->skippedNoTargets = true; + co_return; + } + + TraceEvent("StalePeerTestStarting") + .detail("Role", self->dstKillRole) + .detail("TargetsFound", targetAddresses.size()); + + // Pick the target to kill. For SS, choose the recruited server that the + // most inspected sources currently reference (cached interface / live + // peer ref) so the kill actually exercises client-side eviction, polling + // briefly (bounded) to let caches warm under the ReadWrite workload. For + // other roles the candidates are interchangeable, so take the first. In + // all cases snapshot how many inspected sources reference the chosen + // target right before the kill -- that count is the non-vacuity signal. + NetworkAddress oldAddr; + if (self->dstKillRole == "ss") { + double warmDeadline = now() + 30.0; + for (;;) { + NetworkAddress best; + int bestN = -1; + for (const auto& a : targetAddresses) { + int n = self->countSourcesReferencing(a, trackerRole); + if (n > bestN) { + bestN = n; + best = a; + } + } + if (bestN > 0 || now() >= warmDeadline) { + oldAddr = best; + self->preKillRefSources = bestN; + break; + } + co_await delay(2.0); + } + } else { + oldAddr = targetAddresses[0]; + self->preKillRefSources = self->countSourcesReferencing(oldAddr, trackerRole); + } + + // Non-vacuity gate: if no inspected source referenced the target at kill + // time, a post-kill Delta==0 would prove nothing. Mark the run + // inconclusive (skip) rather than recording it as a meaningful pass. + // Coordinator is exempt -- its endpoints use well-known tokens that are + // not tracked here, and its meaningful signal is connection-string + // removal (handled below), not a peer-ref drain. + if (self->dstKillRole != "coordinator" && self->preKillRefSources <= 0) { + TraceEvent(SevWarnAlways, "StalePeerTestVacuousNoPreKillRef") + .detail("Role", self->dstKillRole) + .detail("Address", oldAddr) + .detail("TrackerRole", trackerRole) + .detail("Note", "No inspected source referenced the target at kill time; skipping as inconclusive"); + self->skippedVacuous = true; + co_return; + } + + ISimulator::ProcessInfo* proc = g_simulator->getProcessByAddress(oldAddr); + if (!proc || proc->failed) { + TraceEvent(SevError, "StalePeerTestProcessNotFound").detail("Address", oldAddr); + self->testPassed = false; + co_return; + } + + self->oldKillAddresses.push_back(oldAddr); + + TraceEvent("StalePeerTestKilling") + .detail("Role", self->dstKillRole) + .detail("Address", oldAddr) + .detail("PreKillRefSources", self->preKillRefSources) + .detail("ProcessClass", getSimulatorProcessClass(proc).toString()) + .detail("Zone", proc->locality.zoneId()); + + g_simulator->killProcess(proc, ISimulator::KillType::KillInstantly); + TraceEvent("StalePeerTestKillDone") + .detail("Address", oldAddr) + .detail("Role", self->dstKillRole) + .detail("ProcFailedFlag", proc->failed); + + // Guard: sim2 silently refuses to kill protected addresses. If our + // candidate filter missed one, fail fast with a clear error rather + // than letting the later stale-peer-ref check attribute the refs on + // an alive process to a real leak. + if (!proc->failed) { + TraceEvent(SevError, "StalePeerTestKillIneffective") + .detail("Address", oldAddr) + .detail("Protected", g_simulator->isProtectedAddress(oldAddr)) + .detail("Rebooting", proc->rebooting); + self->testPassed = false; + co_return; + } + + // Wait for the kill's recovery + cleanup. Right after killProcess the + // broadcast dbInfo still carries the pre-kill FULLY_RECOVERED for a few + // seconds, so a bare wait-for-FULLY_RECOVERED races straight through + // without actually waiting for the kill's recovery -- then the later + // peer-ref check lands mid-recovery and skips as "cluster unhealthy". + // First wait (bounded) for the kill to drop recoveryState below + // FULLY_RECOVERED (recovery triggered); roles whose loss is handled + // without a master recovery (ss/dd/rk) simply hit this timeout. Then + // wait (bounded) for the recovery to complete. Both bounds fall through + // to the post-waitAfterKill cluster-health guard if exceeded, so this + // can never hang. + TraceEvent("StalePeerTestWaitingForRecovery"); + double recoveryTriggerDeadline = now() + 30.0; + while (self->dbInfo->get().recoveryState >= RecoveryState::FULLY_RECOVERED && now() < recoveryTriggerDeadline) { + co_await race(self->dbInfo->onChange(), delay(1.0)); + } + double recoveryCompleteDeadline = now() + 180.0; + while (self->dbInfo->get().recoveryState < RecoveryState::FULLY_RECOVERED && now() < recoveryCompleteDeadline) { + co_await race(self->dbInfo->onChange(), delay(5.0)); + } + TraceEvent("StalePeerTestRecovered").detail("RecoveryState", (int)self->dbInfo->get().recoveryState); + + // Coordinator-only: the dead coordinator stays in the cluster's + // connection string indefinitely, so every live process keeps a + // long-term LeaderMonitor connection to its address (PeerRef = 1 + // per process). That isn't a stale-ref leak -- it's the design + // pattern for cluster identity. Trigger an auto quorum change to + // swap the dead coordinator for a healthy candidate, mirroring + // what `coordinators auto` does in fdbcli. Once the dead address + // is no longer in the connection string the LeaderMonitor refs + // drop and the strict peerRefs == 0 check is meaningful. + if (self->dstKillRole == "coordinator") { + TraceEvent("StalePeerTestCoordinatorAutoChange"); + int autoChangeAttempts = 0; + for (;;) { + ++autoChangeAttempts; + CoordinatorsResult res = co_await changeQuorum(cx, autoQuorumChange()); + TraceEvent("StalePeerTestCoordinatorAutoChangeResult") + .detail("Attempt", autoChangeAttempts) + .detail("Result", (int)res); + if (res == CoordinatorsResult::SUCCESS || res == CoordinatorsResult::SAME_NETWORK_ADDRESSES) { + break; + } + if (autoChangeAttempts >= 20) { + TraceEvent(SevWarn, "StalePeerTestCoordinatorAutoChangeGivingUp").detail("LastResult", (int)res); + break; + } + co_await delay(1.0); + } + } + + TraceEvent("StalePeerTestWaiting").detail("WaitSeconds", self->waitAfterKill); + co_await delay(self->waitAfterKill); + + // Cluster-health guard: if the cluster never recovered to + // FULLY_RECOVERED within the wait window (e.g. perpetual_storage_wiggle + // + ssd-sharded-rocksdb configs that produce RkSSListFetchTimeout + // before the kill, or a degenerate recovery loop after the kill), the + // peer-ref check is meaningless -- the leak signal will be dominated + // by the cluster meltdown rather than the kill we're testing. Skip + // rather than fail: the contract this test verifies is "kill of role X + // drains its peer refs in a healthy cluster", not "every cluster + // configuration recovers within 240s". + if (self->dbInfo->get().recoveryState < RecoveryState::FULLY_RECOVERED) { + TraceEvent("StalePeerTestClusterUnhealthy") + .detail("RecoveryState", (int)self->dbInfo->get().recoveryState) + .detail("KillRole", self->dstKillRole) + .detail("Note", "Cluster did not return to FULLY_RECOVERED; skipping peer-ref check"); + self->skippedClusterUnhealthy = true; + co_return; + } + + // Pass criterion: per-role InterfaceTracker Delta == 0 at the killed + // address. Delta counts leaked COPIES of the killed role's interface + // RequestStreams still pinned somewhere -- i.e. the actual stale-interface + // leak our client-side fixes target. We deliberately do NOT use raw + // peer->peerReferences: a tester client legitimately retains + // connection-level / well-known-token refs to the killed address (the + // draining TCP connection, ping/leader-monitor endpoints, and refs to a + // co-resident role the process also hosted -- sim co-locates many roles + // per process). Those are real, expected peers (peerReferences > 0 with + // Delta == 0), not the staleness bug. peerReferences is logged as a + // diagnostic. Delta requires the InterfaceTracker, so this mode needs + // stale_peer_observability = true (deterministic: toggling it does not + // change simulator ordering). + // + // Coordinator is special -- its endpoints use well-known tokens + // (ClientLeaderRegInterface) and aren't tracked, and every process keeps + // a long-term LeaderMonitor connection to each coordinator address. The + // meaningful signal for a coordinator kill is whether the cluster removed + // the dead coord from the connection string; that is the autoQuorumChange + // result above, so no peer-ref check is done for coord. + // + // KNOWN COVERAGE GAP (intentional): a coordinator kill therefore asserts + // only that the dead coordinator was swapped out of the connection string + // (the autoQuorumChange succeeded) -- it does NOT assert that every source + // drained its peer ref to the dead address. The old refs do get dropped, but + // asynchronously: they live in clientLeaderServers (a state vector in + // monitorProxiesOneGeneration) and only go away when that generation rolls over + // to the new connection string. We check right after autoQuorumChange, which + // doesn't guarantee that rollover has happened yet, so a lingering ref could be present + // becaus of timing. + // Since dstKillRole="clientFacing" resolves uniformly across {coordinator, + // cluster_controller, commit_proxy, grv_proxy, ss}, roughly one in five + // clientFacing runs lands on coordinator and exercises only this weaker + // connection-string check. The per-role stale-interface drain is covered + // by the other four roles. (Tightening the coordinator path to a strict + // peer-ref assertion -- by waiting for the generation rollover -- is + // a possible follow-up.) + if (self->dstKillRole == "coordinator") { + TraceEvent("StalePeerTestChecking").detail("Mode", "coordinator/skip-peer-refs"); + co_return; + } + TraceEvent("StalePeerTestChecking"); + + // trackerRole was computed once at the top of _start (computeTrackerRole) + // and used for the pre-kill snapshot; reuse it here. It is non-empty for + // every role that reaches this point (coordinator returned above). + ASSERT(!trackerRole.empty()); + + auto allProcesses = g_simulator->getAllProcesses(); + int clientSourcesChecked = 0; // source processes inspected (post-filter) + int clientSourcesWithLeak = 0; // ...of which had a per-role Delta > 0 (a leak) + for (auto* proc : allProcesses) { + // Same source population as the pre-kill snapshot (see + // isInspectedSource): drops failed/rebooting and, in "tester_client" + // mode, keeps only customer-client (TesterClass) processes while + // excluding the simulator's internal TestSystem driver (IP 1.1.1.1). + if (!self->isInspectedSource(proc)) + continue; + auto* transport = static_cast((void*)proc->global(INetwork::enFlowTransport)); + ++clientSourcesChecked; + for (const auto& oldKillAddr : self->oldKillAddresses) { + auto& allPeers = transport->getAllPeers(); + auto it = allPeers.find(oldKillAddr); + const int peerRefs = (it != allPeers.end()) ? it->second->peerReferences : 0; + const int64_t delta = transport->interfaceTracker.getDelta(oldKillAddr, trackerRole); + if (delta > 0) { + ++clientSourcesWithLeak; + } + if (delta > 0) { + transport->interfaceTracker.prettyPrint(proc->address, self->oldKillAddresses); + transport->interfaceTracker.prettyPrintLeakedReceivers(proc->address, self->oldKillAddresses); + transport->interfaceTracker.prettyPrintLeakedRefs(proc->address, self->oldKillAddresses); + TraceEvent(SevError, "StalePeerTestFailed") + .detail("CheckedProcess", proc->address) + .detail("CheckedProcessClass", getSimulatorProcessClass(proc).toString()) + .detail("SrcCheckRole", self->srcCheckRole) + .detail("OldAddress", oldKillAddr) + .detail("KillRole", self->dstKillRole) + .detail("TrackerRole", trackerRole) + .detail("TrackerDelta", delta) + .detail("PeerReferences", peerRefs); + self->testPassed = false; + } + } + } + // Coverage visibility: how many source processes we examined, and how + // many showed a per-role interface leak (Delta > 0). This is logged for + // every run (pass or fail) so the aggregate leak signal is visible even + // when the run passes. PreKillRefSources records how many sources held a + // reference to the target before the drain -- it is > 0 here by + // construction (the non-vacuity gate above skips the run otherwise). + TraceEvent("StalePeerTestCheckCoverage") + .detail("SrcCheckRole", self->srcCheckRole) + .detail("KillRole", self->dstKillRole) + .detail("PreKillRefSources", self->preKillRefSources) + .detail("ClientSourcesChecked", clientSourcesChecked) + .detail("ClientSourcesWithLeak", clientSourcesWithLeak); + + // Non-vacuity assertion: a real check must have inspected at least one + // source. Reaching here means we killed a target that was referenced + // pre-kill (preKillRefSources > 0), so the inspected population cannot be + // empty. If it somehow is, the run validated nothing -- fail loudly + // rather than report a hollow pass. + if (clientSourcesChecked == 0) { + TraceEvent(SevError, "StalePeerTestNoSourcesInspected") + .detail("KillRole", self->dstKillRole) + .detail("SrcCheckRole", self->srcCheckRole) + .detail("PreKillRefSources", self->preKillRefSources); + self->testPassed = false; + } + + co_return; + } + + Future check(Database const& cx) override { + if (clientId != 0) { + return true; + } + if (oldKillAddresses.empty() && !skippedNoTargets && !skippedVacuous) { + TraceEvent(SevError, "StalePeerTestNoProcessKilled").detail("KillRole", dstKillRole); + testPassed = false; + } + TraceEvent("StalePeerTestResult") + .detail("Passed", testPassed) + .detail("KillRole", dstKillRole) + .detail("SrcCheckRole", srcCheckRole) + .detail("SkippedNoTargets", skippedNoTargets) + .detail("SkippedClusterUnhealthy", skippedClusterUnhealthy) + .detail("SkippedVacuous", skippedVacuous) + .detail("PreKillRefSources", preKillRefSources) + .detail("ProcessesKilled", oldKillAddresses.size()); + return testPassed; + } + + void getMetrics(std::vector& m) override {} +}; + +WorkloadFactory StalePeerTestWorkloadFactory; diff --git a/fdbserver/workloads/StorageServerCheckpointRestoreTest.cpp b/fdbserver/workloads/StorageServerCheckpointRestoreTest.cpp index 965ff6fe4eb..a904b460047 100644 --- a/fdbserver/workloads/StorageServerCheckpointRestoreTest.cpp +++ b/fdbserver/workloads/StorageServerCheckpointRestoreTest.cpp @@ -23,7 +23,7 @@ #include "fdbrpc/simulator.h" #include "fdbserver/kvstore/IKeyValueStore.h" #include "fdbserver/core/Knobs.h" -#include "fdbserver/core/ServerCheckpoint.h" +#include "fdbserver/checkpoint/Checkpoint.h" #include "fdbserver/core/MoveKeys.h" #include "fdbserver/core/QuietDatabase.h" #include "fdbserver/tester/workloads.h" diff --git a/fdbserver/workloads/UDPWorkload.cpp b/fdbserver/workloads/UDPWorkload.cpp index 28b99593a0b..bb5b9cb19e8 100644 --- a/fdbserver/workloads/UDPWorkload.cpp +++ b/fdbserver/workloads/UDPWorkload.cpp @@ -50,7 +50,7 @@ struct UDPWorkload : TestWorkload { std::unordered_map sent, received, acked, successes; PromiseStream toAck; - explicit(false) UDPWorkload(WorkloadContext const& wcx) : TestWorkload(wcx) { + explicit UDPWorkload(WorkloadContext const& wcx) : TestWorkload(wcx) { keyPrefix = getOption(options, "keyPrefix"_sr, "/udp/"_sr); runFor = getOption(options, "runFor"_sr, 60.0); minPort = getOption(options, "minPort"_sr, 5000); diff --git a/fdbserver/workloads/UnitTests.cpp b/fdbserver/workloads/UnitTests.cpp index 1b3a90ee4fa..968d7f11149 100644 --- a/fdbserver/workloads/UnitTests.cpp +++ b/fdbserver/workloads/UnitTests.cpp @@ -53,6 +53,8 @@ void forceLinkIPagerTests(); void forceLinkMockS3ServerTests(); void forceLinkAuditUtilsTests(); void forceLinkShardsAffectedByTeamFailureTests(); +void forceLinkNativeCdcRetagCleanupTests(); +void forceLinkNativeCdcMetadataTests(); void forceLinkClusterHealthMonitorTests(); void forceLinkGrvQueueDelayTests(); void forceLinkGrvProxyStarvationTests(); @@ -132,6 +134,8 @@ struct UnitTestWorkload : TestWorkload { forceLinkMockS3ServerTests(); forceLinkAuditUtilsTests(); forceLinkShardsAffectedByTeamFailureTests(); + forceLinkNativeCdcRetagCleanupTests(); + forceLinkNativeCdcMetadataTests(); forceLinkClusterHealthMonitorTests(); forceLinkGrvQueueDelayTests(); forceLinkGrvProxyStarvationTests(); diff --git a/flow/Histogram.cpp b/flow/Histogram.cpp index f9304796019..4e03d0ecc07 100644 --- a/flow/Histogram.cpp +++ b/flow/Histogram.cpp @@ -56,7 +56,7 @@ void HistogramRegistry::unregisterHistogram(Histogram* h) { TraceEvent(SevError, "HistogramNotRegistered").detail("group", h->group).detail("op", h->op); } int count = histograms.erase(name); - ASSERT(count == 1); + ASSERT_EQ(count, 1); } Histogram* HistogramRegistry::lookupHistogram(std::string const& name) { diff --git a/flow/IndexedSet.cpp b/flow/IndexedSet.cpp index 5582a3d1ac1..e5072f39a2f 100644 --- a/flow/IndexedSet.cpp +++ b/flow/IndexedSet.cpp @@ -372,11 +372,13 @@ TEST_CASE("performance/flow/IndexedSet/strings") { printf("%0.1f Map.KfindStr/sec\n", count / 1000.0 / (end - start)); + tt = 0; start = timer(); for (size_t i = 0; i < count; i++) { - aMap.find(hello); + tt += aMap.find(hello)->second; } end = timer(); + ASSERT(tt == count); printf("%0.1f std::map.KfindStr/sec\n", count / 1000.0 / (end - start)); return Void(); diff --git a/flow/Knobs.cpp b/flow/Knobs.cpp index 9932b696f04..377d542b955 100644 --- a/flow/Knobs.cpp +++ b/flow/Knobs.cpp @@ -56,7 +56,9 @@ void FlowKnobs::initialize(Randomize randomize, IsSimulated isSimulated) { init( ENABLE_COORDINATOR_DNS_CACHE, false ); if( randomize && buggify() ) ENABLE_COORDINATOR_DNS_CACHE = true; init( COORDINATOR_DNS_CACHE_REFRESH_INTERVAL, 3.0 ); init( COORDINATOR_DNS_CACHE_TTL, 30.0 ); + init( STALE_PEER_OBSERVABILITY, false ); init( CACHE_REFRESH_INTERVAL_WHEN_ALL_ALTERNATIVES_FAILED, 1.0 ); + init( PERSISTENT_CONNECT_FAILED_COUNT_TTL, isSimulated ? 120.0 : 600.0 ); init( DELAY_JITTER_OFFSET, 0.9 ); init( DELAY_JITTER_RANGE, 0.2 ); diff --git a/flow/MemoryTracker.cpp b/flow/MemoryTracker.cpp index 7b4027f5331..26b99c33e9b 100644 --- a/flow/MemoryTracker.cpp +++ b/flow/MemoryTracker.cpp @@ -149,17 +149,20 @@ __attribute__((no_instrument_function, noinline)) int captureFramesFP(void** out initStackBoundsForThread(); } void** fp = static_cast(__builtin_frame_address(0)); - // Fallback for threads where pthread_getattr_np failed: ±8 MB around - // the initial frame. - // Caveat: this may need to be constrained more tightly to deal with - // smaller stacks. - uintptr_t lo = gStackLow ? gStackLow : reinterpret_cast(fp); - uintptr_t hi = gStackHigh ? gStackHigh : reinterpret_cast(fp) + (8u << 20); + uintptr_t base = reinterpret_cast(fp); + // gStackLow spans all of RLIMIT_STACK, mostly unmapped for the main thread; the walk only ascends. + uintptr_t lo = std::max(gStackLow, base); + // Fallback for threads where pthread_getattr_np failed: 8 MB above the initial frame. + uintptr_t hi = gStackHigh ? gStackHigh : base + (8u << 20); + if (hi < 16) { + return 0; + } int n = 0; while (fp && n < max) { uintptr_t a = reinterpret_cast(fp); - // Reject out-of-stack or misaligned fp before dereferencing. - if (a < lo || a + 16 > hi) { + // Reject out-of-stack or misaligned fp before dereferencing. Subtraction, not + // `a + 16 > hi`: that addition wraps for an fp near the top of the address space. + if (a < lo || a > hi - 16) { break; } if (a & (sizeof(void*) - 1)) { diff --git a/flow/MkCertCli.cpp b/flow/MkCertCli.cpp index 02b3f615910..193eaeaadee 100644 --- a/flow/MkCertCli.cpp +++ b/flow/MkCertCli.cpp @@ -229,7 +229,10 @@ int main(int argc, char** argv) { case OPT_SERVER_CHAIN_LEN: try { serverArgs.length = std::stoul(args.OptionArg()); - assert(serverArgs.length > 0); + if (serverArgs.length == 0) { + fmt::print(stderr, "ERROR: Certificate chain length must be positive\n"); + return FDB_EXIT_ERROR; + } } catch (std::exception const& ex) { fmt::print(stderr, "ERROR: Invalid chain length ({})\n", ex.what()); return FDB_EXIT_ERROR; @@ -238,7 +241,6 @@ int main(int argc, char** argv) { case OPT_CLIENT_CHAIN_LEN: try { clientArgs.length = std::stoul(args.OptionArg()); - assert(clientArgs.length > 0); } catch (std::exception const& ex) { fmt::print(stderr, "ERROR: Invalid chain length ({})\n", ex.what()); return FDB_EXIT_ERROR; diff --git a/flow/Net2.cpp b/flow/Net2.cpp index 50d96a5d98d..4a6cccf8668 100644 --- a/flow/Net2.cpp +++ b/flow/Net2.cpp @@ -218,7 +218,7 @@ class Net2 final : public INetwork, public INetworkConnections { flowGlobalType global(int id) const override { return (globals.size() > id) ? globals[id] : nullptr; } void setGlobal(size_t id, flowGlobalType v) override { - ASSERT(id < globals.size()); + ASSERT_LT(id, globals.size()); globals[id] = v; } @@ -525,7 +525,7 @@ class Connection final : public IConnection, ReferenceCounted { if (err) { // Since there was an error, sent's value can't be used to infer that the buffer has data and the limit is // positive so check explicitly. - ASSERT(limit > 0); + ASSERT_GT(limit, 0); bool notEmpty = false; for (auto p = data; p; p = p->next) { if (p->bytes_written - p->bytes_sent > 0) { @@ -1230,7 +1230,7 @@ class SSLConnection final : public IConnection, ReferenceCounted if (err) { // Since there was an error, sent's value can't be used to infer that the buffer has data and the limit is // positive so check explicitly. - ASSERT(limit > 0); + ASSERT_GT(limit, 0); bool notEmpty = false; for (auto p = data; p; p = p->next) { if (p->bytes_written - p->bytes_sent > 0) { @@ -2252,7 +2252,7 @@ void ASIOReactor::wake() { } // namespace N2 SendBufferIterator::SendBufferIterator(SendBuffer const* p, int limit) : p(p), limit(limit) { - ASSERT(limit > 0); + ASSERT_GT(limit, 0); } void SendBufferIterator::operator++() { diff --git a/flow/Net2Packet.cpp b/flow/Net2Packet.cpp index efa77e115d5..f73c11893cf 100644 --- a/flow/Net2Packet.cpp +++ b/flow/Net2Packet.cpp @@ -147,14 +147,14 @@ void UnsentPacketQueue::sent(int bytes) { if (b->bytes_sent + bytes <= b->bytes_written && (b->bytes_sent + bytes != b->bytes_written || (!b->next && b->bytes_unwritten()))) { b->bytes_sent += bytes; - ASSERT(b->bytes_sent <= b->size()); + ASSERT_LE(b->bytes_sent, b->size()); break; } // We've sent an entire buffer bytes -= b->bytes_written - b->bytes_sent; b->bytes_sent = b->bytes_written; - ASSERT(b->bytes_written <= b->size()); + ASSERT_LE(b->bytes_written, b->size()); double queue_time = now() - b->enqueue_time; sendQueueLatencyHistogram->sampleSeconds(queue_time); unsent_first = b->nextPacketBuffer(); diff --git a/flow/Platform.cpp b/flow/Platform.cpp index 2e8fcd3a235..58236e4489a 100644 --- a/flow/Platform.cpp +++ b/flow/Platform.cpp @@ -2913,7 +2913,10 @@ THREAD_HANDLE startThread(void* (*func)(void*), void* arg, int stackSize, const pthread_t t; pthread_attr_t attr; - pthread_attr_init(&attr); + int attrError = pthread_attr_init(&attr); + if (attrError != 0) { + criticalError(FDB_EXIT_ERROR, "ThreadAttributesError", strerror(attrError)); + } if (stackSize != 0) { if (pthread_attr_setstacksize(&attr, stackSize) != 0) { // If setting the stack size fails the default stack size will be used, so failure to set @@ -2926,8 +2929,13 @@ THREAD_HANDLE startThread(void* (*func)(void*), void* arg, int stackSize, const } auto* args = new ThreadCreateArgs(func, arg); - pthread_create(&t, &attr, &runFunc, args); + int createError = pthread_create(&t, &attr, &runFunc, args); pthread_attr_destroy(&attr); + if (createError != 0) { + delete args; + // Callers transfer thread-owned state without a failed-launch recovery path. + criticalError(FDB_EXIT_ERROR, "ThreadCreationError", strerror(createError)); + } #if defined(__linux__) if (name != nullptr) { @@ -3713,6 +3721,57 @@ void registerCrashHandlerCallback(void (*f)()) { g_crashHandlerCallbacks.push_back(f); } +#ifdef __linux__ +// Sampled at registration time, not in the handler: reading /proc/self/maps needs +// malloc, and the handler can run with the allocator's locks held. +uintptr_t g_reportedStackLow = 0; +uintptr_t g_reportedStackHigh = 0; +uintptr_t g_mainStackMappedLow = 0; +uintptr_t g_mainStackMappedHigh = 0; +uint64_t g_stackRlimit = 0; + +// Global because crashHandler keeps its plain sa_handler signature. +uintptr_t g_faultAddress = 0; +bool g_faultAddressValid = false; + +void sampleMainStackExtent() { + struct rlimit rl; + if (getrlimit(RLIMIT_STACK, &rl) == 0) { + g_stackRlimit = rl.rlim_cur; + } + + // Must match initStackBoundsForThread in flow/MemoryTracker.cpp. + pthread_attr_t attr; + if (pthread_getattr_np(pthread_self(), &attr) == 0) { + void* base = nullptr; + size_t size = 0; + if (pthread_attr_getstack(&attr, &base, &size) == 0) { + g_reportedStackLow = reinterpret_cast(base); + g_reportedStackHigh = g_reportedStackLow + size; + } + pthread_attr_destroy(&attr); + } + + FILE* maps = fopen("/proc/self/maps", "r"); + if (maps == nullptr) { + return; + } + char line[512]; + while (fgets(line, sizeof(line), maps) != nullptr) { + if (strstr(line, "[stack]") == nullptr) { + continue; + } + unsigned long low = 0, high = 0; + if (sscanf(line, "%lx-%lx", &low, &high) == 2) { + g_mainStackMappedLow = low; + g_mainStackMappedHigh = high; + } + break; + } + fclose(maps); +} +#endif + // The crashHandler function is registered to handle signals before the process terminates. // Basic information about the crash is printed/traced, and stdout and trace events are flushed. void crashHandler(int sig) { @@ -3732,6 +3791,9 @@ void crashHandler(int sig) { fprintf(error ? stderr : stdout, "SIGNAL: %s (%d)\n", strsignal(sig), sig); if (error) { + if (g_faultAddressValid) { + fprintf(stderr, "FaultAddress: 0x%zx\n", (size_t)g_faultAddress); + } fprintf(stderr, "Trace: %s\n", backtrace.c_str()); } @@ -3739,6 +3801,18 @@ void crashHandler(int sig) { { TraceEvent te(error ? SevError : SevInfo, error ? "Crash" : "ProcessTerminated"); te.detail("Signal", sig).detail("Name", strsignal(sig)).detail("Trace", backtrace); + if (g_faultAddressValid) { + // FaultInReportedStack separates a wild pointer from stack bounds that reach + // below what is mapped. + te.detail("FaultAddress", format("0x%zx", (size_t)g_faultAddress)) + .detail("ReportedStackLow", format("0x%zx", (size_t)g_reportedStackLow)) + .detail("ReportedStackHigh", format("0x%zx", (size_t)g_reportedStackHigh)) + .detail("MainStackMappedLow", format("0x%zx", (size_t)g_mainStackMappedLow)) + .detail("MainStackMappedHigh", format("0x%zx", (size_t)g_mainStackMappedHigh)) + .detail("StackRlimit", g_stackRlimit) + .detail("FaultInReportedStack", + g_faultAddress >= g_reportedStackLow && g_faultAddress < g_reportedStackHigh); + } if (error) { te.setErrorKind(ErrorKind::BugDetected); } @@ -3773,14 +3847,26 @@ void crashHandler(int sig) { #endif } +#ifdef __linux__ +void crashHandlerSigInfo(int sig, siginfo_t* info, void*) { + if (info != nullptr && (sig == SIGSEGV || sig == SIGBUS)) { + g_faultAddress = reinterpret_cast(info->si_addr); + g_faultAddressValid = true; + } + crashHandler(sig); +} +#endif + void registerCrashHandler() { #ifdef __linux__ + sampleMainStackExtent(); + // For these otherwise fatal errors, attempt to log a trace of // what was happening and then exit struct sigaction action; - action.sa_handler = crashHandler; + action.sa_sigaction = crashHandlerSigInfo; sigfillset(&action.sa_mask); - action.sa_flags = 0; + action.sa_flags = SA_SIGINFO; // deliver siginfo_t so SIGSEGV reports its fault address sigaction(SIGILL, &action, nullptr); sigaction(SIGFPE, &action, nullptr); diff --git a/flow/SystemMonitor.cpp b/flow/SystemMonitor.cpp index a8761b829d0..945abd096b8 100644 --- a/flow/SystemMonitor.cpp +++ b/flow/SystemMonitor.cpp @@ -262,19 +262,7 @@ SystemStatistics customSystemMonitor(std::string const& eventName, StatisticsSta total_memory += FastAllocator<8192>::getTotalMemory(); total_memory += FastAllocator<16384>::getTotalMemory(); - uint64_t unused_memory = 0; - unused_memory += FastAllocator<16>::getApproximateMemoryUnused(); - unused_memory += FastAllocator<32>::getApproximateMemoryUnused(); - unused_memory += FastAllocator<64>::getApproximateMemoryUnused(); - unused_memory += FastAllocator<96>::getApproximateMemoryUnused(); - unused_memory += FastAllocator<128>::getApproximateMemoryUnused(); - unused_memory += FastAllocator<256>::getApproximateMemoryUnused(); - unused_memory += FastAllocator<512>::getApproximateMemoryUnused(); - unused_memory += FastAllocator<1024>::getApproximateMemoryUnused(); - unused_memory += FastAllocator<2048>::getApproximateMemoryUnused(); - unused_memory += FastAllocator<4096>::getApproximateMemoryUnused(); - unused_memory += FastAllocator<8192>::getApproximateMemoryUnused(); - unused_memory += FastAllocator<16384>::getApproximateMemoryUnused(); + const uint64_t unused_memory = getTotalUnusedAllocatedMemory(); if (total_memory > 0) { TraceEvent("FastAllocMemoryUsage") diff --git a/flow/TLSConfig.cpp b/flow/TLSConfig.cpp index 0ab43b54341..b904a4a8008 100644 --- a/flow/TLSConfig.cpp +++ b/flow/TLSConfig.cpp @@ -180,40 +180,31 @@ void ConfigureSSLStream(Reference policy, } } -std::string TLSConfig::getCertificatePathSync() const { - if (!tlsCertPath.empty()) { - return tlsCertPath; +static std::string resolveTLSMaterialPath(const std::string& configuredPath, + const char* envName, + const char* defaultFileName) { + if (!configuredPath.empty()) { + return configuredPath; } - std::string envCertPath; - if (platform::getEnvironmentVar("FDB_TLS_CERTIFICATE_FILE", envCertPath)) { - return envCertPath; + std::string envPath; + if (platform::getEnvironmentVar(envName, envPath)) { + return envPath; } - const char* defaultCertFileName = "cert.pem"; - if (fileExists(joinPath(platform::getDefaultConfigPath(), defaultCertFileName))) { - return joinPath(platform::getDefaultConfigPath(), defaultCertFileName); + if (fileExists(joinPath(platform::getDefaultConfigPath(), defaultFileName))) { + return joinPath(platform::getDefaultConfigPath(), defaultFileName); } return std::string(); } -std::string TLSConfig::getKeyPathSync() const { - if (!tlsKeyPath.empty()) { - return tlsKeyPath; - } - - std::string envKeyPath; - if (platform::getEnvironmentVar("FDB_TLS_KEY_FILE", envKeyPath)) { - return envKeyPath; - } - - const char* defaultKeyFileName = "key.pem"; - if (fileExists(joinPath(platform::getDefaultConfigPath(), defaultKeyFileName))) { - return joinPath(platform::getDefaultConfigPath(), defaultKeyFileName); - } +std::string TLSConfig::getCertificatePathSync() const { + return resolveTLSMaterialPath(tlsCertPath, "FDB_TLS_CERTIFICATE_FILE", "cert.pem"); +} - return std::string(); +std::string TLSConfig::getKeyPathSync() const { + return resolveTLSMaterialPath(tlsKeyPath, "FDB_TLS_KEY_FILE", "key.pem"); } std::string TLSConfig::getCAPathSync() const { @@ -737,7 +728,7 @@ std::string getX509Name(const X509_NAME* name) { X509_NAME_print_ex(out.get(), name, /* indent= */ 0, /* flags */ XN_FLAG_ONELINE); unsigned char* rawName = nullptr; long length = BIO_get_mem_data(out.get(), &rawName); - ASSERT(length > 0); + ASSERT_GT(length, 0); std::string result((const char*)rawName, length); return result; } @@ -1016,8 +1007,7 @@ bool TLSPolicy::verify_peer(bool preverified, X509_STORE_CTX* store_ctx, const N .detail("Rule", rule.toString()); } } else { - TraceEvent(SevInfo, "TLSPolicySuccess") - .suppressFor(1.0) + TraceEvent(SevDebug, "TLSPolicySuccess") .detail("PeerAddress", peerAddress) .detail("Reason", verifier.getSuccessReason()); } diff --git a/flow/bench/BenchActorPatterns.cpp b/flow/bench/BenchActorPatterns.cpp new file mode 100644 index 00000000000..78ee601af0f --- /dev/null +++ b/flow/bench/BenchActorPatterns.cpp @@ -0,0 +1,373 @@ +/* + * BenchActorPatterns.cpp + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "benchmark/benchmark.h" + +#include "flow/ThreadHelper.h" +#include "flow/genericactors.h" + +#include + +namespace { + +Future emptyUncancellable(Uncancellable = {}) { + co_return; +} + +Future readyFuture() { + return Void(); +} + +Future waitUncancellable(Future input, Uncancellable = {}) { + co_await input; +} + +Future waitOne(Future input) { + co_await input; +} + +Future waitEither(Future first, Future second) { + co_await race(first, second); +} + +Future g_input; + +Future waitSnapshot() { + Future input = g_input; + co_await input; +} + +Future sumStream(FutureStream input) { + int sum = 0; + try { + while (true) { + sum += co_await input; + } + } catch (Error& e) { + if (e.code() != error_code_end_of_stream) { + throw; + } + } + co_return sum; +} + +enum class ScalarPattern { + EmptyUncancellable, + ReadyFuture, + WaitUncancellableReady, + WaitReady, + WaitCancel, + WaitAfter, + RaceBothReady, + RaceFirstReady, + RaceSecondReady, + RaceCancel, + RaceAfter, + WaitRaceAfter, + FanoutRaceAfter, + Quorum +}; + +template +Future runScalar(benchmark::State* state) { + Promise pending; + Future never = pending.getFuture(); + Future ready = Void(); + std::vector> promises; + std::vector> inputs; + if constexpr (pattern == ScalarPattern::Quorum) { + promises.resize(3); + inputs.resize(3); + } + + // Destruction, including cancellation of unresolved actors, is part of each operation. + for (auto _ : *state) { + if constexpr (pattern == ScalarPattern::EmptyUncancellable) { + Future f = emptyUncancellable(); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::ReadyFuture) { + Future f = readyFuture(); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::WaitUncancellableReady) { + Future f = waitUncancellable(ready); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::WaitReady) { + Future f = waitOne(ready); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::WaitCancel) { + Future f = waitOne(never); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::RaceBothReady) { + Future f = waitEither(ready, ready); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::RaceFirstReady) { + Future f = waitEither(ready, never); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::RaceSecondReady) { + Future f = waitEither(never, ready); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::RaceCancel) { + Future f = waitEither(never, never); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::Quorum) { + promises.clear(); + promises.resize(3); + for (int i = 0; i < 3; ++i) { + inputs[i] = promises[i].getFuture(); + } + Future f = quorum(inputs, 2); + for (auto& promise : promises) { + promise.send(Void()); + } + benchmark::DoNotOptimize(f); + } else { + Promise signal; + if constexpr (pattern == ScalarPattern::WaitAfter) { + Future f = waitOne(signal.getFuture()); + signal.send(Void()); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::RaceAfter) { + Future f = waitEither(signal.getFuture(), never); + signal.send(Void()); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::WaitRaceAfter) { + Future f = waitOne(waitEither(signal.getFuture(), never)); + signal.send(Void()); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::FanoutRaceAfter) { + Future f = waitEither(signal.getFuture(), never); + Future first = waitOne(f); + Future second = waitOne(f); + signal.send(Void()); + benchmark::DoNotOptimize(first); + benchmark::DoNotOptimize(second); + } + } + benchmark::ClobberMemory(); + } + ASSERT_EQ(pending.getFutureReferenceCount(), 1); + state->SetItemsProcessed(state->iterations()); + co_return; +} + +template +void benchScalar(benchmark::State& state) { + onMainThread([&state] { return runScalar(&state); }).getBlocking(); +} + +enum class BatchPattern { + WaitFifo, + WaitLifo, + Race, + RaceSameInput, + NestedRace, + WaitRace, + FanoutRace, + FanoutWaitRace, + FanoutSnapshotRace +}; + +template +Future runBatch(benchmark::State* state) { + const int count = state->range(0); + constexpr bool fanout = pattern == BatchPattern::FanoutRace || pattern == BatchPattern::FanoutWaitRace || + pattern == BatchPattern::FanoutSnapshotRace; + Promise pending; + Future never = pending.getFuture(); + for (auto _ : *state) { + state->PauseTiming(); + { + std::vector> signals(count); + std::vector> first(count); + std::vector> second(fanout ? count : 0); + // Measure construction and completion with all frames live together; exclude input setup and final cleanup. + state->ResumeTiming(); + for (int i = 0; i < count; ++i) { + if constexpr (pattern == BatchPattern::WaitFifo || pattern == BatchPattern::WaitLifo) { + first[i] = waitOne(signals[i].getFuture()); + } else if constexpr (pattern == BatchPattern::Race) { + first[i] = waitEither(signals[i].getFuture(), never); + } else if constexpr (pattern == BatchPattern::RaceSameInput) { + first[i] = waitEither(signals[i].getFuture(), signals[i].getFuture()); + } else if constexpr (pattern == BatchPattern::NestedRace) { + first[i] = waitEither(waitEither(signals[i].getFuture(), never), never); + } else if constexpr (pattern == BatchPattern::WaitRace) { + first[i] = waitOne(waitEither(signals[i].getFuture(), never)); + } else if constexpr (pattern == BatchPattern::FanoutRace) { + Future f = waitEither(signals[i].getFuture(), never); + first[i] = waitOne(f); + second[i] = waitOne(f); + } else if constexpr (pattern == BatchPattern::FanoutWaitRace) { + Future f = waitEither(waitOne(signals[i].getFuture()), never); + first[i] = waitOne(f); + second[i] = waitOne(f); + } else if constexpr (pattern == BatchPattern::FanoutSnapshotRace) { + g_input = signals[i].getFuture(); + Future f = waitEither(waitSnapshot(), never); + g_input = f; + first[i] = waitSnapshot(); + second[i] = waitSnapshot(); + } + } + benchmark::DoNotOptimize(first.data()); + benchmark::DoNotOptimize(second.data()); + for (int i = 0; i < count; ++i) { + const int index = pattern == BatchPattern::WaitLifo ? count - 1 - i : i; + signals[index].send(Void()); + } + benchmark::ClobberMemory(); + state->PauseTiming(); + for (auto const& f : first) { + ASSERT(f.isReady() && !f.isError()); + } + for (auto const& f : second) { + ASSERT(f.isReady() && !f.isError()); + } + if constexpr (pattern == BatchPattern::FanoutSnapshotRace) { + g_input = Future(); + } + } + ASSERT_EQ(pending.getFutureReferenceCount(), 1); + state->ResumeTiming(); + } + state->SetItemsProcessed(state->iterations() * count); + co_return; +} + +template +void benchBatch(benchmark::State& state) { + onMainThread([&state] { return runBatch(&state); }).getBlocking(); +} + +Future runStreamSum(benchmark::State* state) { + const int count = state->range(0); + for (auto _ : *state) { + PromiseStream stream; + Future sum = sumStream(stream.getFuture()); + for (int i = 0; i < count; ++i) { + stream.send(1); + } + stream.sendError(end_of_stream()); + benchmark::DoNotOptimize(sum); + ASSERT_EQ(sum.get(), count); + } + state->SetItemsProcessed(state->iterations() * count); + co_return; +} + +void benchStreamSum(benchmark::State& state) { + onMainThread([&state] { return runStreamSum(&state); }).getBlocking(); +} + +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::EmptyUncancellable) + ->Name("actor_patterns/empty_uncancellable") + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::ReadyFuture)->Name("actor_patterns/ready_future")->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::WaitUncancellableReady) + ->Name("actor_patterns/wait_uncancellable_ready") + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::WaitReady)->Name("actor_patterns/wait_ready")->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::WaitCancel)->Name("actor_patterns/wait_construct_cancel")->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::WaitAfter)->Name("actor_patterns/wait_after")->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::RaceBothReady)->Name("actor_patterns/race_both_ready")->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::RaceFirstReady)->Name("actor_patterns/race_first_ready")->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::RaceSecondReady) + ->Name("actor_patterns/race_second_ready") + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::RaceCancel)->Name("actor_patterns/race_construct_cancel")->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::RaceAfter)->Name("actor_patterns/race_after")->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::WaitRaceAfter)->Name("actor_patterns/wait_race_after")->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::FanoutRaceAfter) + ->Name("actor_patterns/fanout_race_after") + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::Quorum)->Name("actor_patterns/quorum_2_of_3")->UseRealTime(); + +BENCHMARK_TEMPLATE(benchBatch, BatchPattern::WaitFifo) + ->Name("actor_patterns/batch_wait_fifo") + ->Arg(64) + ->Arg(4096) + ->Arg(65536) + ->Arg(1000000) + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchBatch, BatchPattern::WaitLifo) + ->Name("actor_patterns/batch_wait_lifo") + ->Arg(64) + ->Arg(4096) + ->Arg(65536) + ->Arg(1000000) + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchBatch, BatchPattern::Race) + ->Name("actor_patterns/batch_race") + ->Arg(64) + ->Arg(4096) + ->Arg(65536) + ->Arg(1000000) + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchBatch, BatchPattern::RaceSameInput) + ->Name("actor_patterns/batch_race_same_input") + ->Arg(64) + ->Arg(4096) + ->Arg(65536) + ->Arg(1000000) + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchBatch, BatchPattern::NestedRace) + ->Name("actor_patterns/batch_nested_race") + ->Arg(64) + ->Arg(4096) + ->Arg(65536) + ->Arg(1000000) + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchBatch, BatchPattern::WaitRace) + ->Name("actor_patterns/batch_wait_race") + ->Arg(64) + ->Arg(4096) + ->Arg(65536) + ->Arg(1000000) + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchBatch, BatchPattern::FanoutRace) + ->Name("actor_patterns/batch_fanout_race") + ->Arg(64) + ->Arg(4096) + ->Arg(65536) + ->Arg(1000000) + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchBatch, BatchPattern::FanoutWaitRace) + ->Name("actor_patterns/batch_fanout_wait_race") + ->Arg(64) + ->Arg(4096) + ->Arg(65536) + ->Arg(1000000) + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchBatch, BatchPattern::FanoutSnapshotRace) + ->Name("actor_patterns/batch_fanout_snapshot_race") + ->Arg(64) + ->Arg(4096) + ->Arg(65536) + ->Arg(1000000) + ->UseRealTime(); +BENCHMARK(benchStreamSum) + ->Name("actor_patterns/stream_sum") + ->Arg(64) + ->Arg(4096) + ->Arg(65536) + ->Arg(1000000) + ->UseRealTime(); + +} // namespace diff --git a/flow/bench/README.md b/flow/bench/README.md index 348ec9f6cfd..67967ff3926 100644 --- a/flow/bench/README.md +++ b/flow/bench/README.md @@ -55,8 +55,26 @@ Existing Benchmarks - `bench_stream` measures the performance of writing to and reading from a `PromiseStream` - `bench_random` measures the performance of `DeterministicRandom`. - `bench_timer` measures the performance of FoundationDB timers. +- `actor_patterns` measures ready futures, coroutine waits and cancellation, races, nested and shared-input graphs, stream consumption, and quorum completion. - `Memcpy` compares `rte_memcpy_noinline` and `memcpy` across aligned/unaligned and cached/uncached copy cases. +Actor patterns +============== + +Run the actor-pattern suite with `bin/flow_bench --benchmark_filter='^actor_patterns/'`. +Use `--benchmark_repetitions=5 --benchmark_out=actor-patterns.json --benchmark_out_format=json` +to retain repeated measurements. Batch cases cover 64, 4096, 65536, and 1000000 inputs; +for a shorter run, select one size, for example `--benchmark_filter='^actor_patterns/batch_.*/4096/'`. + +These benchmarks report wall-clock time and items per second. One item is a complete +actor graph, including both outputs for fan-out cases, or one consumed value for +`stream_sum`. Scalar cases include construction, completion or cancellation, and +destruction. Batch cases measure graph construction and FIFO/LIFO completion with +all graphs outstanding together; input-promise/vector setup, result checks, and +final cleanup are excluded. Stream cases include consumer startup and end-of-stream +handling. `ready_future` returns an already-ready `Future` without a coroutine; +`empty_uncancellable` executes an uncancellable coroutine. + Future use cases ================ diff --git a/flow/flow.cpp b/flow/flow.cpp index 8debd77524a..605691bbef3 100644 --- a/flow/flow.cpp +++ b/flow/flow.cpp @@ -137,10 +137,10 @@ std::string UID::toString() const { } UID UID::fromString(std::string const& s) { - ASSERT(s.size() == 32); + ASSERT_EQ(s.size(), 32); uint64_t a = 0, b = 0; int r = sscanf(s.c_str(), "%16" SCNx64 "%16" SCNx64, &a, &b); - ASSERT(r == 2); + ASSERT_EQ(r, 2); return UID(a, b); } @@ -422,7 +422,7 @@ void bindDeterministicRandomToOpenssl() { } int nChooseK(int n, int k) { - assert(n >= k && k >= 0); + ASSERT(n >= k && k >= 0); if (k == 0) { return 1; } @@ -437,7 +437,7 @@ int nChooseK(int n, int k) { ret *= n - i + 1; ret /= i; } - ASSERT(ret <= INT_MAX); + ASSERT_LE(ret, INT_MAX); return ret; } diff --git a/flow/include/flow/Deque.h b/flow/include/flow/Deque.h index 2b664d2f68a..4f130816547 100644 --- a/flow/include/flow/Deque.h +++ b/flow/include/flow/Deque.h @@ -61,6 +61,9 @@ class Deque { } void operator=(Deque const& r) { + if (this == &r) { + return; + } cleanup(); arr = nullptr; @@ -92,6 +95,9 @@ class Deque { } void operator=(Deque&& r) noexcept { + if (this == &r) { + return; + } cleanup(); begin = r.begin; diff --git a/flow/include/flow/Error.h b/flow/include/flow/Error.h index 028a7826222..e13e222cca3 100644 --- a/flow/include/flow/Error.h +++ b/flow/include/flow/Error.h @@ -159,7 +159,7 @@ void assert_impl(char const* a_nm, bool (*compare)(T const&, U const&), char const* file, int line) { - if (!compare(a, b)) { + if (!(compare(a, b) || isAssertDisabled(line))) [[unlikely]] { throw internal_error_impl(a_nm, Traceable::toString(a), opName, b_nm, Traceable::toString(b), file, line); } } diff --git a/flow/include/flow/FastRef.h b/flow/include/flow/FastRef.h index 1c9fe802d8f..7276df47a14 100644 --- a/flow/include/flow/FastRef.h +++ b/flow/include/flow/FastRef.h @@ -127,6 +127,8 @@ class Reference { if (ptr) delref(ptr); } + // Comparing pointees also handles self-assignment. + // NOLINTNEXTLINE(bugprone-unhandled-self-assignment) Reference& operator=(const Reference& r) { P* oldPtr = ptr; P* newPtr = r.ptr; diff --git a/flow/include/flow/FlowThread.h b/flow/include/flow/FlowThread.h index 62c91f72777..3d038a71221 100644 --- a/flow/include/flow/FlowThread.h +++ b/flow/include/flow/FlowThread.h @@ -218,6 +218,9 @@ class ThreadFutureStream { } void operator=(const ThreadFutureStream& rhs) { + if (this == &rhs) { + return; + } rhs.queue->addFutureRef(); if (queue) queue->delFutureRef(); @@ -271,6 +274,8 @@ class ThreadReturnPromiseStream { return ThreadFutureStream(queue); } + // The incoming queue reference is acquired before the old one is released. + // NOLINTNEXTLINE(bugprone-unhandled-self-assignment) void operator=(const ThreadReturnPromiseStream& rhs) { rhs.queue->addPromiseRef(); if (queue) diff --git a/flow/include/flow/IndexedSet.h b/flow/include/flow/IndexedSet.h index 2dac816aa5c..5d603340b11 100644 --- a/flow/include/flow/IndexedSet.h +++ b/flow/include/flow/IndexedSet.h @@ -372,6 +372,8 @@ class MapPair { template MapPair(Key_&& key, Value_&& value) : key(std::forward(key)), value(std::forward(value)) {} + // Memberwise assignment has the same self-assignment behavior as a defaulted operator. + // NOLINTNEXTLINE(bugprone-unhandled-self-assignment) void operator=(MapPair const& rhs) { key = rhs.key; value = rhs.value; diff --git a/flow/include/flow/Knobs.h b/flow/include/flow/Knobs.h index c003d8905bc..bf56084d39f 100644 --- a/flow/include/flow/Knobs.h +++ b/flow/include/flow/Knobs.h @@ -108,6 +108,35 @@ class FlowKnobs : public KnobsImpl { double COORDINATOR_DNS_CACHE_TTL; double CACHE_REFRESH_INTERVAL_WHEN_ALL_ALTERNATIVES_FAILED; + // When true, enables tracing to debug stale peer issues (see StalePeerTest.toml). + // Not recommended for production due to potential performance overhead. + // + // Debugging workflow: if a simulation test fails with a dangling peer reference + // error, re-run the same seed with stale_peer_observability=true. This causes + // interfaceTracker to emit trace events recording creation and destruction at + // different layers of the RPC stack: interface, request stream, flow receiver. + // See InterfaceTracker-related traces in FlowTransport.cpp. These traces include + // backtraces that can be used to identify where the leak originated. From there, + // reason about what the code is doing, why it runs into the stale peer issue, + // and what can be done to prune the stale peers. The fix will vary case by case. + bool STALE_PEER_OBSERVABILITY; + + // Used for two purposes off the same value: (1) eviction age -- an address is + // pruned from TransportData::persistentConnectFailedCount once it has had no + // connect failure for this many seconds; and (2) sweep cadence -- the prune + // scan runs at most once per this interval (and only when triggered by a + // connect failure). Bounds that map and the locationCachePeerEvictor + // snapshot/streak maps it feeds. 0 disables pruning. Coupling both onto one + // knob is a deliberate simplification: sweep about as often as the eviction + // horizon. + // + // NOTE: this value should be set to a multiple of LOCATION_CACHE_PEER_EVICTOR_DELAY + // so that a dead address is not pruned from this map before the evictor has a + // chance to observe its failure delta and evict the corresponding location cache entries. + // It should also be a multiple of the reconnect interval, so that an address that is + // still being targeted by RPCs re-fails and refreshes its timestamp before the TTL, + // and is never pruned while actively dead. + double PERSISTENT_CONNECT_FAILED_COUNT_TTL; double DELAY_JITTER_OFFSET; double DELAY_JITTER_RANGE; double BUSY_WAIT_THRESHOLD; diff --git a/flow/include/flow/TDMetric.h b/flow/include/flow/TDMetric.h index a32a6da6985..8725d35003f 100644 --- a/flow/include/flow/TDMetric.h +++ b/flow/include/flow/TDMetric.h @@ -173,7 +173,7 @@ struct MetricBatch { MetricBatch() = default; explicit MetricBatch(FDBScope* in) { - assert(in != nullptr); + ASSERT(in != nullptr); scope.inserts = std::move(in->inserts); scope.appends = std::move(in->appends); scope.updates = std::move(in->updates); diff --git a/flow/include/flow/ThreadHelper.h b/flow/include/flow/ThreadHelper.h index 762977c0598..1e0e91a12fd 100644 --- a/flow/include/flow/ThreadHelper.h +++ b/flow/include/flow/ThreadHelper.h @@ -615,6 +615,8 @@ class ThreadFuture { if (sav) sav->delref(); } + // Acquiring the incoming reference first keeps the shared state alive during self-assignment. + // NOLINTNEXTLINE(bugprone-unhandled-self-assignment) void operator=(const ThreadFuture& rhs) { if (rhs.sav) rhs.sav->addref(); diff --git a/flow/include/flow/flow.h b/flow/include/flow/flow.h index 86926a3c56d..e199deca5d3 100644 --- a/flow/include/flow/flow.h +++ b/flow/include/flow/flow.h @@ -937,7 +937,7 @@ class void set(const void* _Nonnull pointerToContinuationInstance, Future f, const void* _Nonnull thisPointer) { // Verify Swift did not make a copy of the `self` value for this method // call. - assert(this == thisPointer); + ASSERT_ABORT(this == thisPointer); // FIXME: Propagate `SwiftCC` to Swift using forward // interop, without relying on passing it via a `void *` @@ -1012,6 +1012,8 @@ SWIFT_CONFORMS_TO_PROTOCOL(flow_swift.FlowFutureOps) if (sav) sav->delFutureRef(); } + // Acquiring the incoming reference first keeps the shared state alive during self-assignment. + // NOLINTNEXTLINE(bugprone-unhandled-self-assignment) void operator=(const Future& rhs) { if (rhs.sav) rhs.sav->addFutureRef(); @@ -1117,6 +1119,8 @@ class SWIFT_SENDABLE Promise final { sav->delPromiseRef(); } + // Acquiring the incoming reference first keeps the shared state alive during self-assignment. + // NOLINTNEXTLINE(bugprone-unhandled-self-assignment) void operator=(const Promise& rhs) { if (rhs.sav) rhs.sav->addPromiseRef(); @@ -1281,6 +1285,20 @@ struct NotifiedQueue : private SingleCallback bool shouldFireImmediately() { return SingleCallback::next != this; } }; +// Global callback for futureRef tracking release, set by FlowTransport at init +inline void (*g_futureRefReleasedCallback)(int64_t id) = nullptr; + +// Global callback for futureRef tracking copy, set by FlowTransport at init. +// Given the source FutureStream's tracking id, registers a brand-new tracked +// ref (cloning the source's addr/token) and returns its id. A copied +// FutureStream is a distinct ref (it bumps the queue's future ref count), so it +// needs its own tracking id rather than sharing the source's -- otherwise +// copies would go untracked (the move ctor propagates the id, but a copy can't +// steal it). The flow layer has no endpoint context of its own, so this defers +// to FlowTransport, which holds the source record's addr/token. Returns -1 when +// the source isn't tracked or observability is off. +inline int64_t (*g_futureRefCopiedCallback)(int64_t srcId) = nullptr; + template class SWIFT_SENDABLE FutureStream { public: @@ -1294,26 +1312,56 @@ class SWIFT_SENDABLE FutureStream { void addCallbackAndClear(SingleCallback* cb) { queue->addCallbackAndDelFutureRef(cb); queue = nullptr; + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && m_futureRefTrackingId >= 0 && g_futureRefReleasedCallback) { + g_futureRefReleasedCallback(m_futureRefTrackingId); + } + m_futureRefTrackingId = -1; + } + FutureStream() : queue(nullptr), m_futureRefTrackingId(-1) {} + FutureStream(const FutureStream& rhs) : queue(rhs.queue), m_futureRefTrackingId(-1) { + queue->addFutureRef(); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && rhs.m_futureRefTrackingId >= 0 && g_futureRefCopiedCallback) { + m_futureRefTrackingId = g_futureRefCopiedCallback(rhs.m_futureRefTrackingId); + } + } + FutureStream(FutureStream&& rhs) noexcept : queue(rhs.queue), m_futureRefTrackingId(rhs.m_futureRefTrackingId) { + rhs.queue = 0; + rhs.m_futureRefTrackingId = -1; } - FutureStream() : queue(nullptr) {} - FutureStream(const FutureStream& rhs) : queue(rhs.queue) { queue->addFutureRef(); } - FutureStream(FutureStream&& rhs) noexcept : queue(rhs.queue) { rhs.queue = 0; } ~FutureStream() { if (queue) queue->delFutureRef(); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && m_futureRefTrackingId >= 0 && g_futureRefReleasedCallback) { + g_futureRefReleasedCallback(m_futureRefTrackingId); + } } void operator=(const FutureStream& rhs) { + if (this == &rhs) { + return; + } rhs.queue->addFutureRef(); if (queue) queue->delFutureRef(); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && m_futureRefTrackingId >= 0 && g_futureRefReleasedCallback) { + g_futureRefReleasedCallback(m_futureRefTrackingId); + } queue = rhs.queue; + m_futureRefTrackingId = -1; + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && rhs.m_futureRefTrackingId >= 0 && g_futureRefCopiedCallback) { + m_futureRefTrackingId = g_futureRefCopiedCallback(rhs.m_futureRefTrackingId); + } } void operator=(FutureStream&& rhs) noexcept { if (rhs.queue != queue) { if (queue) queue->delFutureRef(); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && m_futureRefTrackingId >= 0 && g_futureRefReleasedCallback) { + g_futureRefReleasedCallback(m_futureRefTrackingId); + } queue = rhs.queue; + m_futureRefTrackingId = rhs.m_futureRefTrackingId; rhs.queue = nullptr; + rhs.m_futureRefTrackingId = -1; } } bool operator==(const FutureStream& rhs) { return rhs.queue == queue; } @@ -1326,10 +1374,14 @@ class SWIFT_SENDABLE FutureStream { return queue->error; } - explicit FutureStream(NotifiedQueue* queue) : queue(queue) {} + explicit FutureStream(NotifiedQueue* queue) : queue(queue), m_futureRefTrackingId(-1) {} + FutureStream(NotifiedQueue* queue, int64_t trackingId) : queue(queue), m_futureRefTrackingId(trackingId) {} + + int64_t getFutureRefTrackingId() const { return m_futureRefTrackingId; } private: NotifiedQueue* queue; + int64_t m_futureRefTrackingId; }; template @@ -1410,6 +1462,8 @@ class SWIFT_SENDABLE PromiseStream { PromiseStream() : queue(new NotifiedQueue(0, 1)) {} PromiseStream(const PromiseStream& rhs) : queue(rhs.queue) { queue->addPromiseRef(); } PromiseStream(PromiseStream&& rhs) noexcept : queue(rhs.queue) { rhs.queue = 0; } + // Acquiring the incoming reference first keeps the shared state alive during self-assignment. + // NOLINTNEXTLINE(bugprone-unhandled-self-assignment) void operator=(const PromiseStream& rhs) { rhs.queue->addPromiseRef(); if (queue) diff --git a/flow/include/flow/genericactors.h b/flow/include/flow/genericactors.h index 443babad286..fe143e86454 100644 --- a/flow/include/flow/genericactors.h +++ b/flow/include/flow/genericactors.h @@ -2027,7 +2027,7 @@ struct FlowLock : NonCopyable, public ReferenceCounted { FlowLock* lock; int64_t remaining; Releaser() : lock(0), remaining(0) {} - explicit(false) Releaser(FlowLock& lock, int64_t amount = 1) : lock(&lock), remaining(amount) {} + explicit Releaser(FlowLock& lock, int64_t amount = 1) : lock(&lock), remaining(amount) {} Releaser(Releaser&& r) noexcept : lock(r.lock), remaining(r.remaining) { r.remaining = 0; } void operator=(Releaser&& r) { if (remaining) @@ -2402,9 +2402,9 @@ class AndFuture { AndFuture& operator=(AndFuture const& f) = default; AndFuture& operator=(AndFuture&& f) noexcept = default; - explicit(false) AndFuture(Future const& f) : futureCount(1), futures{ f } {} + explicit AndFuture(Future const& f) : futureCount(1), futures{ f } {} - explicit(false) AndFuture(Error const& e) : futureCount(1), futures{ Future(e) } {} + explicit AndFuture(Error const& e) : futureCount(1), futures{ Future(e) } {} operator Future() { return getFuture(); } diff --git a/flow/include/flow/network.h b/flow/include/flow/network.h index 4b712a13315..72c68421acc 100644 --- a/flow/include/flow/network.h +++ b/flow/include/flow/network.h @@ -87,6 +87,8 @@ struct NetworkMetrics { } // Since networkBusyness is atomic we need to redefine copy assignment operator + // All fields support self-assignment; the array is copied element by element. + // NOLINTNEXTLINE(bugprone-unhandled-self-assignment) NetworkMetrics& operator=(const NetworkMetrics& rhs) { for (int i = 0; i < SLOW_EVENT_BINS; i++) { countSlowEvents[i] = rhs.countSlowEvents[i]; diff --git a/flow/include/flow/swift_stream_support.h b/flow/include/flow/swift_stream_support.h index 6919a121bb7..ee594d998e6 100644 --- a/flow/include/flow/swift_stream_support.h +++ b/flow/include/flow/swift_stream_support.h @@ -21,6 +21,8 @@ #ifndef SWIFT_STREAM_SUPPORT_H #define SWIFT_STREAM_SUPPORT_H +#ifdef WITH_SWIFT + #include "swift.h" #include "flow.h" #include "unsafe_swift_compat.h" @@ -46,7 +48,7 @@ class FlowSingleCallbackForSwiftContinuation : SingleCallback { void set(const void* _Nonnull pointerToContinuationInstance, FutureStream fs, const void* _Nonnull thisPointer) { // Verify Swift did not make a copy of the `self` value for this method // call. - assert(this == thisPointer); + ASSERT_ABORT(this == thisPointer); // FIXME: Propagate `SwiftCC` to Swift using forward // interop, without relying on passing it via a `void *` @@ -153,4 +155,6 @@ struct UNSAFE_SWIFT_CXX_IMMORTAL_REF SwiftContinuationSingleCallbackCInt : Singl } }; +#endif /* WITH_SWIFT */ + #endif diff --git a/packaging/multiversion/clients/postinst b/packaging/multiversion/clients/postinst index 7b2e9936fa7..f5997392005 100644 --- a/packaging/multiversion/clients/postinst +++ b/packaging/multiversion/clients/postinst @@ -9,7 +9,7 @@ then fi if [ ! -d "${PKG_CONFIG_DIR}" ] then - mkdir ${PKG_CONFIG_DIR} + mkdir "${PKG_CONFIG_DIR}" fi update-alternatives --install /usr/bin/fdbcli fdbclients /usr/lib/foundationdb-@FDB_VERSION@@FDB_BUILDTIME_STRING@/bin/fdbcli @ALTERNATIVES_PRIORITY@ \ @@ -19,6 +19,7 @@ update-alternatives --install /usr/bin/fdbcli fdbclients /usr/lib/foundationdb-@ --slave /usr/bin/fdbdr fdbdr /usr/lib/foundationdb-@FDB_VERSION@@FDB_BUILDTIME_STRING@/bin/fdbbackup \ --slave /usr/bin/backup_agent backup_agent /usr/lib/foundationdb-@FDB_VERSION@@FDB_BUILDTIME_STRING@/bin/fdbbackup \ --slave /usr/@LIB_DIR@/libfdb_c.so libfdb_c /usr/lib/foundationdb-@FDB_VERSION@@FDB_BUILDTIME_STRING@/lib/libfdb_c.so \ + --slave /usr/@LIB_DIR@/libfdb_c_shim.so libfdb_c_shim /usr/lib/foundationdb-@FDB_VERSION@@FDB_BUILDTIME_STRING@/lib/libfdb_c_shim.so \ --slave /usr/@LIB_DIR@/pkgconfig/foundationdb-client.pc foundationdb-client.pc /usr/lib/foundationdb-@FDB_VERSION@@FDB_BUILDTIME_STRING@/lib/pkgconfig/foundationdb-client.pc \ --slave /usr/@LIB_DIR@/cmake/FoundationDB-Client FoundationDB-ClientConfig /usr/lib/foundationdb-@FDB_VERSION@@FDB_BUILDTIME_STRING@/lib/cmake/FoundationDB-Client \ --slave /usr/include/foundationdb fdb-client-headers /usr/lib/foundationdb-@FDB_VERSION@@FDB_BUILDTIME_STRING@/include/foundationdb diff --git a/packaging/multiversion/clients/prerm b/packaging/multiversion/clients/prerm index 0a87f237544..bc09201eca9 100644 --- a/packaging/multiversion/clients/prerm +++ b/packaging/multiversion/clients/prerm @@ -1,3 +1,7 @@ #!/usr/bin/env bash -update-alternatives --remove fdbclients /usr/lib/foundationdb-@FDB_VERSION@@FDB_BUILDTIME_STRING@/bin/fdbcli +case "${1:-remove}" in + 0|remove) + update-alternatives --remove fdbclients /usr/lib/foundationdb-@FDB_VERSION@@FDB_BUILDTIME_STRING@/bin/fdbcli + ;; +esac diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index bf6b1785cac..f2006935a98 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -21,6 +21,7 @@ set_tests_properties(test_venv_setup PROPERTIES RESOURCE_LOCK TEST_VENV_SETUP) # We need some variables to configure the test setup set(ENABLE_BUGGIFY ON CACHE BOOL "Enable buggify for tests") set(ENABLE_SIMULATION_TESTS OFF CACHE BOOL "Enable simulation tests (useful if you can't run Joshua)") +set(ENABLE_LONG_RUNNING_TESTS OFF CACHE BOOL "Add a long running tests package") set(RUN_IGNORED_TESTS OFF CACHE BOOL "Run tests that are marked for ignore") set(TEST_KEEP_LOGS "FAILED" CACHE STRING "Which logs to keep (NONE, FAILED, ALL)") set(TEST_KEEP_SIMDIR "NONE" CACHE STRING "Which simfdb directories to keep (NONE, FAILED, ALL)") @@ -176,7 +177,6 @@ if(WITH_PYTHON) add_fdb_test(TEST_FILES slow/BackupS3BlobBulkLoadRestore.toml) add_fdb_test(TEST_FILES slow/BackupS3BlobBulkLoadRestoreWithChaos.toml) add_fdb_test(TEST_FILES slow/BackupS3BlobBulkLoadRestoreMultiRange.toml) - add_fdb_test(TEST_FILES slow/BackupS3BlobBulkLoadRestoreNarrowFleet.toml) add_fdb_test(TEST_FILES slow/BackupS3BlobBulkLoadRestoreJobIncomplete.toml) add_fdb_test(TEST_FILES fast/BulkLoading.toml) add_fdb_test(TEST_FILES fast/BulkLoadingDestTeamFailure.toml) @@ -225,11 +225,17 @@ if(WITH_PYTHON) add_fdb_test(TEST_FILES fast/RandomUnitTests.toml) add_fdb_test(TEST_FILES fast/RangeLocking.toml) add_fdb_test(TEST_FILES fast/NativeCdcEndToEnd.toml) + add_fdb_test(TEST_FILES fast/NativeCdcBuggify.toml) add_fdb_test(TEST_FILES fast/NativeCdcAssignmentPublication.toml) + add_fdb_test(TEST_FILES fast/NativeCdcProxyRebalance.toml) + add_fdb_test(TEST_FILES fast/NativeCdcProxyRebalanceAutomatic.toml) add_fdb_test(TEST_FILES fast/NativeCdcSharedTag.toml) add_fdb_test(TEST_FILES fast/NativeCdcRetiredSharedTagSnapshot.toml) add_fdb_test(TEST_FILES fast/NativeCdcMemoryBound.toml) + add_fdb_test(TEST_FILES fast/NativeCdcRetagCompatibility.toml) + add_fdb_test(TEST_FILES fast/NativeCdcRetaggingMemoryBound.toml) add_fdb_test(TEST_FILES fast/NativeCdcReplyChunking.toml) + add_fdb_test(TEST_FILES fast/NativeCdcMultipleRanges.toml) add_fdb_test(TEST_FILES fast/NativeCdcLargeVersionBatching.toml) add_fdb_test(TEST_FILES fast/NativeCdcOversizedPeek.toml) add_fdb_test(TEST_FILES fast/NativeCdcDurableAckScan.toml) @@ -239,12 +245,16 @@ if(WITH_PYTHON) add_fdb_test( TEST_FILES fast/NativeCdcDisableRestart-1.toml fast/NativeCdcDisableRestart-2.toml) + add_fdb_test( + TEST_FILES fast/NativeCdcRetagDisableRestart-1.toml + fast/NativeCdcRetagDisableRestart-2.toml) add_fdb_test(TEST_FILES fast/RangeLockCycle.toml) add_fdb_test(TEST_FILES fast/ReadHotDetectionCorrectness.toml IGNORE) # TODO re-enable once read hot detection is enabled. add_fdb_test(TEST_FILES fast/ReportConflictingKeys.toml) add_fdb_test(TEST_FILES fast/RESTUnit.toml IGNORE) add_fdb_test(TEST_FILES fast/SelectorCorrectness.toml) add_fdb_test(TEST_FILES fast/ShardedRocksNondeterministicTest.toml) + add_fdb_test(TEST_FILES fast/StalePeerTest.toml) add_fdb_test(TEST_FILES fast/Sideband.toml) add_fdb_test(TEST_FILES fast/SidebandSingle.toml) add_fdb_test(TEST_FILES fast/SidebandWithStatus.toml) diff --git a/tests/fast/NativeCdcBuggify.toml b/tests/fast/NativeCdcBuggify.toml new file mode 100644 index 00000000000..cfdac454ca5 --- /dev/null +++ b/tests/fast/NativeCdcBuggify.toml @@ -0,0 +1,43 @@ +[configuration] +buggify = true +faultInjection = true + +[[knobs]] +enable_native_cdc = true +# Keep injected lost initialization replies within the workload's operation deadline. +cc_recovery_init_req_max_timeout = 30.0 +# More active streams than tags exercise retention shared with a lagging consumer. +native_cdc_tag_count = 2 +# Small replies exercise shared-tag buffering within the buggified 10 KB proxy buffer. +maximum_peek_bytes = 1000 + +[[test]] +testTitle = 'NativeCdcBuggify' +useDB = true +waitForQuiescenceEnd = false +timeout = 600 +# Keep failures bounded by Attrition while BUGGIFY varies internal timing and limits. +connectionFailuresDisableDuration = 1000000 +runFailureWorkloads = false + + [[test.workload]] + testName = 'NativeCdcEndToEnd' + initialStreamCount = 6 + minStreamCount = 3 + maxStreamCount = 8 + keyCount = 16 + writesPerRound = 5 + rounds = 20 + drainProbability = 0.25 + delayBetweenRounds = 0.5 + operationTimeout = 500.0 + + [[test.workload]] + testName = 'Attrition' + machinesToKill = 1 + machinesToLeave = 3 + reboot = true + testDuration = 20.0 + waitForVersion = true + allowFaultInjection = false + killDc = false diff --git a/tests/fast/NativeCdcMultipleRanges.toml b/tests/fast/NativeCdcMultipleRanges.toml new file mode 100644 index 00000000000..e8407bc345b --- /dev/null +++ b/tests/fast/NativeCdcMultipleRanges.toml @@ -0,0 +1,21 @@ +[configuration] +config = 'single' +singleRegion = true +buggify = false +faultInjection = false + +[[knobs]] +enable_native_cdc = true +native_cdc_tag_count = 1 + +[[test]] +testTitle = 'NativeCdcMultipleRanges' +useDB = true +waitForQuiescenceEnd = false +connectionFailuresDisableDuration = 1000000 +runFailureWorkloads = false + + [[test.workload]] + testName = 'NativeCdcEndToEnd' + testMultipleRanges = true + operationTimeout = 180.0 diff --git a/tests/fast/NativeCdcProxyRebalance.toml b/tests/fast/NativeCdcProxyRebalance.toml new file mode 100644 index 00000000000..74b0b019f80 --- /dev/null +++ b/tests/fast/NativeCdcProxyRebalance.toml @@ -0,0 +1,33 @@ +[configuration] +config = 'single commit_proxies=1 grv_proxies=2' +singleRegion = true +datacenters = 1 +machineCount = 12 +statelessProcessClassesPerDC = 3 +buggify = false +faultInjection = false + +[[knobs]] +enable_native_cdc = true +native_cdc_tag_count = 2 +# Invoke the production policy from the workload after all three registrations are published. +cdc_proxy_rebalance_enabled = false + +[[test]] +testTitle = 'NativeCdcProxyRebalance' +useDB = true +waitForQuiescenceEnd = false +runFailureWorkloads = false +connectionFailuresDisableDuration = 1000000 +timeout = 180 + + [[test.workload]] + testName = 'NativeCdcEndToEnd' + initialStreamCount = 3 + minStreamCount = 3 + maxStreamCount = 3 + keyCount = 4 + writesPerRound = 1 + rounds = 0 + testProxyRebalance = true + operationTimeout = 90.0 diff --git a/tests/fast/NativeCdcProxyRebalanceAutomatic.toml b/tests/fast/NativeCdcProxyRebalanceAutomatic.toml new file mode 100644 index 00000000000..b2162522ad8 --- /dev/null +++ b/tests/fast/NativeCdcProxyRebalanceAutomatic.toml @@ -0,0 +1,34 @@ +[configuration] +config = 'single commit_proxies=1 grv_proxies=2' +singleRegion = true +datacenters = 1 +machineCount = 12 +statelessProcessClassesPerDC = 3 +buggify = false +faultInjection = false + +[[knobs]] +enable_native_cdc = true +native_cdc_tag_count = 2 +cdc_proxy_rebalance_enabled = true +# Leave time for the three registrations to establish one shared-tag group before the controller scans. +cdc_proxy_rebalance_interval = 20.0 + +[[test]] +testTitle = 'NativeCdcProxyRebalanceAutomatic' +useDB = true +waitForQuiescenceEnd = false +runFailureWorkloads = false +connectionFailuresDisableDuration = 1000000 +timeout = 240 + + [[test.workload]] + testName = 'NativeCdcEndToEnd' + initialStreamCount = 3 + minStreamCount = 3 + maxStreamCount = 3 + keyCount = 4 + writesPerRound = 1 + rounds = 0 + testProxyRebalanceAutomatic = true + operationTimeout = 150.0 diff --git a/tests/fast/NativeCdcRetagCompatibility.toml b/tests/fast/NativeCdcRetagCompatibility.toml new file mode 100644 index 00000000000..54335c23a02 --- /dev/null +++ b/tests/fast/NativeCdcRetagCompatibility.toml @@ -0,0 +1,38 @@ +[configuration] +config = 'double commit_proxies=1 grv_proxies=1' +singleRegion = true +datacenters = 1 +machineCount = 16 +statelessProcessClassesPerDC = 4 +buggify = false +faultInjection = false + +[[knobs]] +enable_native_cdc = true +native_cdc_tag_count = 2 +native_cdc_retag_cleanup_interval = 0.25 +cdc_proxy_failure_coalesce_delay = 0.1 +cdc_proxy_consume_poll_timeout = 0.25 +cdc_proxy_buffer_bytes = 33554432 +maximum_peek_bytes = 65536 +cdc_proxy_consume_reply_bytes = 65536 + +[[test]] +testTitle = 'NativeCdcRetagCompatibility' +useDB = true +runFailureWorkloads = false +runConsistencyCheck = false +waitForQuiescenceEnd = false +connectionFailuresDisableDuration = 1000000 +timeout = 600 + + [[test.workload]] + testName = 'NativeCdcEndToEnd' + initialStreamCount = 4 + minStreamCount = 4 + keyCount = 4 + writesPerRound = 1 + testRetagCompatibility = true + testRetagTransactionRetries = true + memoryTestValueBytes = 512 + operationTimeout = 180.0 diff --git a/tests/fast/NativeCdcRetagDisableRestart-1.toml b/tests/fast/NativeCdcRetagDisableRestart-1.toml new file mode 100644 index 00000000000..d06cb1c1890 --- /dev/null +++ b/tests/fast/NativeCdcRetagDisableRestart-1.toml @@ -0,0 +1,33 @@ +[configuration] +config = 'single' +singleRegion = true +storageEngineExcludeTypes = [3, 4, 5] +buggify = false +faultInjection = false + +[[knobs]] +enable_native_cdc = true +native_cdc_tag_count = 2 +native_cdc_retag_cleanup_interval = 0.25 + +[[test]] +testTitle = 'NativeCdcRetagDisableRestart' +clearAfterTest = false +runFailureWorkloads = false +runConsistencyCheck = false +waitForQuiescenceEnd = false + + [[test.workload]] + testName = 'NativeCdcEndToEnd' + initialStreamCount = 1 + minStreamCount = 1 + keyCount = 4 + writesPerRound = 1 + prepareRestartDrain = true + testRetaggedRestart = true + testRetagTransactionRetries = true + + [[test.workload]] + testName = 'SaveAndKill' + restartInfoLocation = 'simfdb/restartInfo.ini' + testDuration = 20.0 diff --git a/tests/fast/NativeCdcRetagDisableRestart-2.toml b/tests/fast/NativeCdcRetagDisableRestart-2.toml new file mode 100644 index 00000000000..47ce7ab3329 --- /dev/null +++ b/tests/fast/NativeCdcRetagDisableRestart-2.toml @@ -0,0 +1,28 @@ +[configuration] +storageEngineExcludeTypes = [4, 5] +buggify = false +faultInjection = false + +[[knobs]] +enable_native_cdc = false +native_cdc_tag_count = 2 +native_cdc_retag_cleanup_interval = 0.25 + +[[test]] +testTitle = 'NativeCdcRetagDisableRestart' +runSetup = false +runFailureWorkloads = false +runConsistencyCheck = false +waitForQuiescenceBegin = false +waitForQuiescenceEnd = false +connectionFailuresDisableDuration = 1000000 + + [[test.workload]] + testName = 'NativeCdcEndToEnd' + initialStreamCount = 1 + minStreamCount = 1 + keyCount = 4 + writesPerRound = 1 + drainAfterRestart = true + testRetaggedRestart = true + operationTimeout = 180.0 diff --git a/tests/fast/NativeCdcRetaggingMemoryBound.toml b/tests/fast/NativeCdcRetaggingMemoryBound.toml new file mode 100644 index 00000000000..db22b44485b --- /dev/null +++ b/tests/fast/NativeCdcRetaggingMemoryBound.toml @@ -0,0 +1,38 @@ +[configuration] +config = 'single logs=1 commit_proxies=1 grv_proxies=1' +singleRegion = true +datacenters = 1 +machineCount = 10 +statelessProcessClassesPerDC = 4 +buggify = false +faultInjection = false + +[[knobs]] +enable_native_cdc = true +native_cdc_tag_count = 2 +native_cdc_retag_cleanup_interval = 0.25 +cdc_proxy_buffer_bytes = 4608 +maximum_peek_bytes = 1152 +cdc_proxy_consume_reply_bytes = 4096 +# Both readers must deliver before acknowledgement, well before a consume lease can expire. +cdc_proxy_consume_poll_timeout = 600.0 +cdc_proxy_pop_scan_interval = 600.0 + +[[test]] +testTitle = 'NativeCdcRetaggingMemoryBound' +useDB = true +runFailureWorkloads = false +runConsistencyCheck = false +waitForQuiescenceEnd = false +connectionFailuresDisableDuration = 1000000 +timeout = 600 + + [[test.workload]] + testName = 'NativeCdcEndToEnd' + initialStreamCount = 2 + minStreamCount = 2 + keyCount = 2 + writesPerRound = 1 + testRetaggingMemoryBound = true + retentionValidationDelay = 1.0 + operationTimeout = 20.0 diff --git a/tests/fast/StalePeerTest.toml b/tests/fast/StalePeerTest.toml new file mode 100644 index 00000000000..c3e0d60e648 --- /dev/null +++ b/tests/fast/StalePeerTest.toml @@ -0,0 +1,95 @@ +[configuration] +buggify = false +machineCount = 10 +desiredTLogCount = 6 +generateFearless = false +config = 'triple' + +# Exclude ShardedRocksDB (enum 5): release-7.4 weights simulation storage-engine +# selection heavily toward RocksDB-family engines (PROBABILITY_FACTOR_SHARDED_ROCKSDB_ +# ENGINE_SELECTED_SIM=100, vs no such knob on 7.3), and ShardedRocksDB's slower post-kill +# shard relocation occasionally exceeds this test's fixed waitAfterKill window, which is +# unrelated to the stale-peer eviction logic under test. Same exclusion precedent as +# tests/fast/DDPipelineSaturation.toml (excluded there for a different reason, OOM). +# Plain RocksDB (enum 4) is intentionally kept in scope. +storageEngineExcludeTypes = [5] +tenantModes = ['disabled'] +encryptModes = ['disabled'] +disableTss = true + +# Force 3 coordinators so dstKillRole=coordinator always has a non-protected target. +# With sim2's default coordinator count and protectedAddresses (a majority is protected), small +# clusters can end up with all coordinators in protectedAddresses, in which case +# the test trivially skips and we get no signal. +coordinators = 3 + +[[knobs]] + +# Tune knobs to aggressively evict potentially dead storage servers from NativeAPI's LocationCache +location_cache_peer_evictor_enabled = true +location_cache_peer_evictor_failed_threshold = 0 +location_cache_peer_evictor_delay = 5.0 +location_cache_peer_evictor_scan_chunk = 2 + +# Ensure stale commit/grv proxies are also cleaned up +shrink_proxy_list_clear_cache_below_threshold = true +dbcontext_eager_proxy_update = true + +# The pass criterion is the per-role InterfaceTracker delta (leaked copies of +# the killed role's interface RequestStreams), which requires this bookkeeping, +# so it must be ON. Raw peerReferences is NOT the criterion (a client keeps +# expected connection-level refs to the killed address). +stale_peer_observability = true + +[[test]] +testTitle = 'StalePeerTest' + +# The purpose of this test is purely measuring stale peer references +# and ensuring there are none for a given (srcCheckRole, dstKillRole). +# These checks in general are run outside this test with stale fix features +# turned on e.g. location_cache_peer_evictor_enabled. +runConsistencyCheck = false +waitForQuiescenceBegin = false +waitForQuiescenceEnd = false +clearAfterTest = false + + [[test.workload]] + testName = 'StalePeerTest' + # dstKillRole: which role's process to kill (the "destination" whose stale + # peer refs we hunt). Options: 'tlog', 'ss', 'commit_proxy', 'grv_proxy', + # 'master', 'resolver', 'dd', 'rk', 'coordinator', 'log_router', + # 'cluster_controller', and the special value 'clientFacing'. + # 'clientFacing' resolves per-run (deterministically by seed) to one of the + # roles a client directly addresses: coordinator, cluster_controller, + # commit_proxy, grv_proxy, ss. Use it for bulk validation that all + # client-facing dst-role kills are stale-ref clean in a single ensemble. + dstKillRole = 'clientFacing' + # srcCheckRole: which source processes we inspect for a lingering peer ref + # to the killed dst. 'any' = every live process (server roles included). + # 'tester_client' = only customer-client processes (ProcessClass::TesterClass, + # the testers running the workload). The client-side fixes target what a + # pure customer client sees; server roles drive the client lib internally + # but are out of scope, so we check only tester clients here. + srcCheckRole = 'tester_client' + # Wait after the kill before checking. For the client contract the drain is + # fast (clientInfo rotation for proxies/CC; location-cache eviction for SS), + # so a short wait suffices. + waitAfterKill = 120.0 + + # Drive real client->dst connections so the tester_client check is not + # vacuous: without sustained reads/writes a tester client may never open a + # connection to the killed dst (especially an SS). ReadWrite has each client + # populate + read/write a keyspace, so clients hold genuine peer refs to + # commit/grv proxies and storage servers and we observe them drop the dead + # dst while keeping the live ones. + [[test.workload]] + testName = 'ReadWrite' + testDuration = 180.0 + transactionsPerSecond = 100 + nodeCount = 1000 + readsPerTransactionA = 10 + writesPerTransactionA = 1 + readsPerTransactionB = 0 + writesPerTransactionB = 0 + alpha = 0.0 + setup = true diff --git a/tests/rare/ClogRemoteTLog.toml b/tests/rare/ClogRemoteTLog.toml index 0d8dfc91b47..89760125b04 100644 --- a/tests/rare/ClogRemoteTLog.toml +++ b/tests/rare/ClogRemoteTLog.toml @@ -12,6 +12,8 @@ worker_health_monitor_interval = 10 cc_enable_worker_health_monitor = true cc_health_trigger_recovery = true cc_enable_remote_tlog_degradation_monitoring = true +# Clogged links can be reported as disconnected instead of degraded. +cc_enable_remote_tlog_disconnect_monitoring = true gray_failure_allow_remote_ss_to_complain = true cc_worker_health_checking_interval = 45 cc_min_degradation_interval = 30 diff --git a/tests/slow/BackupS3BlobBulkLoadRestoreNarrowFleet.toml b/tests/slow/BackupS3BlobBulkLoadRestoreNarrowFleet.toml deleted file mode 100644 index adf1bce339e..00000000000 --- a/tests/slow/BackupS3BlobBulkLoadRestoreNarrowFleet.toml +++ /dev/null @@ -1,163 +0,0 @@ -# BulkLoad restore onto a fleet with barely room for a disjoint destination team -# -# Variant of BackupS3BlobBulkLoadRestore.toml that cuts the extra storage machines its sibling tests rely -# on down to the minimum that still permits a legal destination team, so bulk-load task placement -# genuinely fails and the restore has to recover by narrowing task ranges rather than by having spare -# capacity. See the comment on the configuration below for where that minimum comes from. -# -# Reproduces the condition measured on a 100M-key validation cluster of ~40 storage servers: every -# candidate destination team is rejected for overlapping the source, so the team selector reports -# ValidTeamSize 0 with no unhealthy or ineligible teams involved. On that cluster the recovery path ran -# end to end - GetTeamFailedToFindValidTeam, then a relocation declared stuck, then a task split - and the -# restore completed with the consistency check passing. -# -# A run that exercises the fix logs GetTeamFailedToFindValidTeam followed by BulkLoad task splits, and no -# SplitDeclined. Exact counts depend on the dataset knobs below, so they are deliberately not asserted -# here; a fleet large enough to place a disjoint team makes the test vacuous rather than failing it. -# -# The rest of this file is inherited from BackupS3BlobBulkLoadRestore.toml. -# -# BulkLoad Validation Test -# Tests that BulkLoad restore produces identical results to traditional restore -# -# This test validates BulkLoad produces the same data as traditional restore: -# - Backup creates BOTH range files AND SST files (snapshotMode=2) -# - Range files are used by traditional restore -# - SST files are used by BulkLoad restore -# - Validation: compare BulkLoad restore vs traditional restore -# 1. Restore with --add-prefix to system keyspace using TRADITIONAL (rangefile) mode -# This creates a "known good" baseline -# 2. Clear normalKeys (original data) -# 3. Restore to normalKeys using BULKLOAD mode (reads SST files) -# 4. Run audit_storage validate_restore to compare: -# - BulkLoad-restored data (in normalKeys) -# - Traditional-restored data (in system key prefix) -# 5. Clean up validation prefix data -# - This validates BulkLoad produces identical results to traditional restore -# -# Configuration aligned with working tests/slow/BulkDumpingS3.toml - -testClass = "Backup" - -[configuration] -storageEngineExcludeTypes = ["ssd-sharded-rocksdb"] # FIXME: remove after allowing bulkloading with fetchKey and shardedrocksdb -disableTss = true # TODO(BulkLoad): support TSS - -# HA (multi-region) configuration - randomly enabled for test coverage -generateFearless = false -simpleConfig = false -minimumRegions = 1 -# singleRegion not set - allows random HA configuration - -# Ensure enough storage servers for non-overlapping BulkLoad teams -extraMachineCountDC = 3 -# -# getTeamForBulkLoad rejects any candidate team sharing a *server* with the source, and source is the union -# of the owners of every shard the task range spans, so a wide enough range has no legal destination. -# Provoking that is the point of this test; escaping it means narrowing the range, which is the path under -# test. Narrowing bottoms out at one manifest per task, and one manifest can still span several destination -# shards, so the cluster must field StorageTeamSize servers on distinct machines owning none of the range. -# -# That headroom has to come from more storage *machines*, not more servers per machine. Both add servers, -# but co-tenants share the machine's simulated disk, and getTeamForBulkLoad also rejects teams whose -# servers are low on disk -- its own comment warns that random low disk space can strand this test. -# Measured at processesPerMachine 2: DiskNearCapacity 1175 vs 149, GetTeamReturnEmpty 534 vs 12, DDExiting -# 58 vs 3, and the restore froze at 35 of 45 tasks for 2200s until the no-progress timeout fired twice. -# Placement itself was fixed -- zero GetTeamFailedToFindValidTeam -- but disk starvation replaced it. So -# pin one server per machine and buy headroom with extraStorageMachineCountPerDC, whose machines are -# storage-class with a disk each. -# -# machineCount is not the lever either: the run this was written for had 22 machines and only 6 holding -# data, because assignClasses starves storage independently of fleet size. -# -# Reproduction: seed 8008 is a local regression pair. At extraStorageMachineCountPerDC 0 it fails the way -# this test used to -- 4 machines holding data, 750 placement failures, 8 SplitDeclined "single manifest", -# restore_error. At 5 the same seed still provokes the condition (100 placement failures, 2 splits) and now -# recovers, completing with the dataset verified. That pair is the check to run when changing anything here: -# the fix must keep the provocation, not remove it. -# -# Provocation is seed-dependent and uncommon -- roughly 1 seed in 8 locally -- so a clean sweep proves -# little on its own. Grep GetTeamFailedToFindValidTeam and DDBulkLoadTaskSplit to tell a passing run that -# exercised the path from one that never reached it. -processesPerMachine = 1 -extraStorageMachineCountPerDC = 5 -# The knobs below still matter for keeping the test honest: manifests are written one per shard -# (FileBackupAgent.cpp), so min_shard_bytes controls how many exist and -# manifest_count_max_per_bulkload_task how many a task groups. Left to their defaults this test degrades -# into a single task holding a single manifest over the whole key space, which is not worth asserting on. - -# Explicit simple config - single replication, single region, explicit process counts -config = "triple usable_regions=1 storage_engine=ssd-2 perpetual_storage_wiggle=0 commit_proxies=3 grv_proxies=3 resolvers=3 logs=3" - -# Disable buggify and fault injection to avoid interference with MockS3/BulkLoad -buggify = false -faultInjection = false - -# Required knobs for BulkLoad functionality (from BulkDumpingS3.toml) -[[knobs]] -manifest_count_max_per_bulkload_task = 10 -min_shard_bytes = 10000 -bulkload_sim_failure_injection = false -shard_encode_location_metadata = true -enable_read_lock_on_range = true -enable_version_vector = false -enable_version_vector_tlog_unicast = false -enable_version_vector_reply_recovery = false -min_byte_sampling_probability = 0.5 -cc_enforce_use_unfit_dd_in_sim = true -disable_audit_storage_final_replica_check_in_sim = true -max_trace_lines = 5000000 -# Allow more time for BulkDump job to complete (Linux runs 4x slower than macOS) -bulkdump_job_timeout = 1200 -bulkload_job_timeout = 1200 - -# Disable buggified delays -[[flow_knobs]] -MAX_BUGGIFIED_DELAY = 0.0 - -# S3/Blobstore settings for stability/determinism -blobstore_max_connection_life = 300 -blobstore_request_timeout_min = 300 -blobstore_request_tries = 5 -blobstore_connect_tries = 5 -blobstore_connect_timeout = 30 -http_send_size = 1024 -http_read_size = 1024 -connection_monitor_loop_time = 0.1 -connection_monitor_timeout = 1.0 -connection_monitor_idle_timeout = 60.0 -dd_team_zero_server_left_log_delay = 0 -dd_rebalance_parallelism = 1 - -[[test]] -testTitle = 'BackupS3BlobBulkLoadRestoreNarrowFleet' -useDB = true -clearAfterTest = false -simBackupAgents = 'BackupToFile' -waitForQuiescence = false -connectionFailuresDisableDuration = 1000000 -runFailureWorkloads = false -timeout = 3600 - - [[test.workload]] - testName = 'Cycle' - nodeCount = 2000 - transactionsPerSecond = 100.0 - testDuration = 30.0 - - [[test.workload]] - testName = 'BackupS3BlobCorrectness' - backupAfter = 10.0 - restoreAfter = 600.0 - abortAndRestartAfter = 0.0 - stopDifferentialAfter = 0.0 - performRestore = true - backupRangesCount = -1 - skipDirtyRestore = false - backupURL = 'blobstore://mocks3:mocksecret:mocktoken@127.0.0.1:8080/backup_container?bucket=backup_bucket®ion=us-east-1&secure_connection=0&cwpf=1&cu=1' - # BulkDump/BulkLoad integration options - snapshotMode = 2 # 2 = BOTH (creates range files AND SST files for comparison) - useRangeFileRestore = false # false = use BulkLoad for restore - # Validation: Compare BulkLoad-restored vs traditional-restored using audit_storage validate_restore - # Compares BulkLoad-restored (normalKeys) vs traditional-restored (prefix) - performValidation = true