From d6ade37ddd6b7cb565a6baabdec9d72e1ce6cb38 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Mon, 17 Aug 2026 17:51:59 -0700 Subject: [PATCH 001/170] Retry lost old log router initialization replies --- fdbserver/logsystem/LogSystem.cpp | 44 +++- .../logsystem/LogSystemRecoveryTests.cpp | 195 ++++++++++++++++++ fdbserver/worker/worker.cpp | 180 +++++++++++++++- 3 files changed, 405 insertions(+), 14 deletions(-) diff --git a/fdbserver/logsystem/LogSystem.cpp b/fdbserver/logsystem/LogSystem.cpp index 43ae3e14122..68f14757374 100644 --- a/fdbserver/logsystem/LogSystem.cpp +++ b/fdbserver/logsystem/LogSystem.cpp @@ -2096,6 +2096,40 @@ Future LogSystem::epochEnd(Reference>> outLo } } +namespace { + +Future initializeOldLogRouter(RequestStream worker, + InitializeLogRouterRequest request, + bool forRemote) { + Future firstReply = + transformErrors(throwErrorOr(worker.getReplyUnlessFailedFor( + request, SERVER_KNOBS->TLOG_TIMEOUT, SERVER_KNOBS->MASTER_FAILURE_SLOPE_DURING_RECOVERY)), + cluster_recovery_failed()); + if (forRemote) { + co_return co_await firstReply; + } + + auto first = co_await race(firstReply, delay(SERVER_KNOBS->CC_RERECRUIT_LOG_ROUTER_TIMEOUT)); + if (first.index() == 0) { + co_return std::get<0>(std::move(first)); + } + + // Reuse the live role if its initialization reply was lost. Keep the first reply valid, and let the + // transaction-system recovery monitor bound the wait after this single retry. + TraceEvent(SevWarn, "OldLogRouterInitializationRetry", request.reqId) + .detail("Locality", request.locality) + .detail("Tag", request.routerTag); + request.reply.reset(); + Future retryReply = + transformErrors(throwErrorOr(worker.getReplyUnlessFailedFor( + request, SERVER_KNOBS->TLOG_TIMEOUT, SERVER_KNOBS->MASTER_FAILURE_SLOPE_DURING_RECOVERY)), + cluster_recovery_failed()); + auto result = co_await race(firstReply, retryReply); + co_return result.index() == 0 ? std::get<0>(std::move(result)) : std::get<1>(std::move(result)); +} + +} // namespace + Future LogSystem::recruitOldLogRouters(std::vector workers, LogEpoch recoveryCount, int8_t locality, @@ -2156,10 +2190,7 @@ Future LogSystem::recruitOldLogRouters(std::vector worker req.knownLockedTLogIds = knownLockedTLogIds; req.allowDropInSim = SERVER_KNOBS->CC_RECOVERY_INIT_REQ_ALLOW_DROP_IN_SIM && !forRemote; req.isReplacement = false; - auto reply = transformErrors( - throwErrorOr(workers[nextRouter].logRouter.getReplyUnlessFailedFor( - req, SERVER_KNOBS->TLOG_TIMEOUT, SERVER_KNOBS->MASTER_FAILURE_SLOPE_DURING_RECOVERY)), - cluster_recovery_failed()); + auto reply = initializeOldLogRouter(workers[nextRouter].logRouter, req, forRemote); logRouterInitializationReplies.back().push_back(reply); allReplies.push_back(reply); nextRouter = (nextRouter + 1) % workers.size(); @@ -2212,10 +2243,7 @@ Future LogSystem::recruitOldLogRouters(std::vector worker req.recoverAt = old.recoverAt; req.allowDropInSim = SERVER_KNOBS->CC_RECOVERY_INIT_REQ_ALLOW_DROP_IN_SIM && !forRemote; req.isReplacement = false; - auto reply = transformErrors( - throwErrorOr(workers[nextRouter].logRouter.getReplyUnlessFailedFor( - req, SERVER_KNOBS->TLOG_TIMEOUT, SERVER_KNOBS->MASTER_FAILURE_SLOPE_DURING_RECOVERY)), - cluster_recovery_failed()); + auto reply = initializeOldLogRouter(workers[nextRouter].logRouter, req, forRemote); logRouterInitializationReplies.back().push_back(reply); allReplies.push_back(reply); nextRouter = (nextRouter + 1) % workers.size(); diff --git a/fdbserver/logsystem/LogSystemRecoveryTests.cpp b/fdbserver/logsystem/LogSystemRecoveryTests.cpp index 70baf43d283..9e63c717da3 100644 --- a/fdbserver/logsystem/LogSystemRecoveryTests.cpp +++ b/fdbserver/logsystem/LogSystemRecoveryTests.cpp @@ -20,8 +20,13 @@ #include "fdbserver/logsystem/LogSystem.h" #include "fdbserver/logsystem/LogSystemConsumer.h" +#include "fdbserver/core/Knobs.h" +#include "fdbrpc/FailureMonitor.h" +#include "flow/ScopeExit.h" #include "flow/UnitTest.h" +#include + namespace { Reference makeSingleLogSet(const std::vector& tlogs, bool isLocal = true) { @@ -54,10 +59,200 @@ std::tuple, bool> makeLogGroupResults( return std::make_tuple(replicationFactor, std::move(lockResults), nonAvailableTLogsCompletePolicy); } +class ScopedHealthyTestEndpoints { + IFailureMonitor& monitor = IFailureMonitor::failureMonitor(); + std::map savedStatus; + +public: + void add(Endpoint const& endpoint) { + const NetworkAddress address = endpoint.getPrimaryAddress(); + if (savedStatus.emplace(address, monitor.getState(address)).second) { + monitor.setStatus(address, FailureStatus(false)); + } + assertAvailable(endpoint); + } + void assertAvailable(Endpoint const& endpoint) const { ASSERT(monitor.getState(endpoint).isAvailable()); } + + ~ScopedHealthyTestEndpoints() { + for (const auto& [address, status] : savedStatus) { + monitor.setStatus(address, status); + } + } +}; + +class OldLogRouterRecruitmentFixture { + ScopedHealthyTestEndpoints healthyEndpoints; + Reference logSystem; + Reference logSet; + WorkerInterface worker; + TLogInterface router; + const bool forRemote; + Future recruitment; + +public: + static constexpr double retryDelay = 0.01; + + explicit OldLogRouterRecruitmentFixture(bool forRemote) + : logSystem(makeReference(UID(1, 1), LocalityData(), LogEpoch(1))), logSet(makeReference()), + router(UID(2, 2), UID(2, 2), LocalityData()), forRemote(forRemote) { + worker.logRouter.getEndpoint(TaskPriority::Worker); + router.initEndpoints(); + // These in-process RPCs have no connection that would mark their address healthy. + healthyEndpoints.add(worker.logRouter.getEndpoint()); + healthyEndpoints.add(router.waitFailure.getEndpoint()); + logSet->locality = 1; + logSet->startVersion = 100; + logSystem->logRouterTags = 1; + logSystem->recoverAt = 200; + logSystem->knownLockedTLogIds[1] = { 2, 4 }; + if (forRemote) { + OldLogData old; + old.logRouterTags = 1; + old.recoverAt = 200; + old.tLogs.push_back(logSet); + logSystem->oldLogData.push_back(std::move(old)); + } else { + logSystem->tLogs.push_back(logSet); + } + } + + ~OldLogRouterRecruitmentFixture() { recruitment.cancel(); } + + Future start() { + ASSERT(!recruitment.isValid()); + const double savedTimeout = SERVER_KNOBS->CC_RERECRUIT_LOG_ROUTER_TIMEOUT; + ScopeExit restoreTimeout( + [savedTimeout] { setServerKnob("cc_rerecruit_log_router_timeout", KnobValueRef::create(savedTimeout)); }); + setServerKnob("cc_rerecruit_log_router_timeout", KnobValueRef::create(retryDelay)); + // The eager coroutine captures its retry deadline before this synchronous scope restores the knob. + recruitment = logSystem->recruitOldLogRouters( + { worker }, LogEpoch(2), 1, 100, { LocalityData() }, makeReference(), forRemote); + return recruitment; + } + + FutureStream requests() const { return worker.logRouter.getFuture(); } + Future nextRequest() const { + return timeoutError(waitOrError(requests(), recruitment), 1.0); + } + FutureStream> failureRequests() const { return router.waitFailure.getFuture(); } + Future onConfigChange() { return logSystem->onLogSystemConfigChange(); } + void reply(InitializeLogRouterRequest const& request) const { request.reply.send(router); } + bool hasRouter() const { + return logSet->logRouters.size() == 1 && logSet->logRouters.front()->get().id() == router.id(); + } + bool hasNoRouters() const { return logSet->logRouters.empty(); } + void assertPending() const { + healthyEndpoints.assertAvailable(worker.logRouter.getEndpoint()); + healthyEndpoints.assertAvailable(router.waitFailure.getEndpoint()); + if (recruitment.isError()) { + throw recruitment.getError(); + } + ASSERT(!recruitment.isReady()); + } + void cancel() { recruitment.cancel(); } +}; + +void assertSameLogRouterRequest(InitializeLogRouterRequest const& first, InitializeLogRouterRequest const& retry) { + ASSERT(first.reqId == retry.reqId); + ASSERT_EQ(first.recoveryCount, retry.recoveryCount); + ASSERT(first.routerTag == retry.routerTag); + ASSERT_EQ(first.startVersion, retry.startVersion); + ASSERT(first.tLogLocalities == retry.tLogLocalities); + ASSERT(first.tLogPolicy == retry.tLogPolicy); + ASSERT_EQ(first.locality, retry.locality); + ASSERT(first.recoverAt == retry.recoverAt); + ASSERT(first.knownLockedTLogIds == retry.knownLockedTLogIds); + ASSERT_EQ(first.allowDropInSim, retry.allowDropInSim); + ASSERT_EQ(first.isReplacement, retry.isReplacement); + ASSERT(!(first.reply.getFuture() == retry.reply.getFuture())); +} + } // namespace void forceLinkLogSystemRecoveryTests() {} +TEST_CASE("/LogSystem/RecruitOldLogRouters/LocalRetryPreservesRequest") { + OldLogRouterRecruitmentFixture fixture(false); + Future changed = fixture.onConfigChange(); + Future recruitment = fixture.start(); + InitializeLogRouterRequest request = co_await fixture.nextRequest(); + InitializeLogRouterRequest retry = co_await fixture.nextRequest(); + assertSameLogRouterRequest(request, retry); + fixture.assertPending(); + ASSERT(fixture.hasNoRouters()); + fixture.reply(retry); + co_await timeoutError(changed, 1.0); + ASSERT(fixture.hasRouter()); + fixture.assertPending(); + fixture.cancel(); +} + +TEST_CASE("/LogSystem/RecruitOldLogRouters/LocalLateOriginalReply") { + OldLogRouterRecruitmentFixture fixture(false); + Future changed = fixture.onConfigChange(); + Future recruitment = fixture.start(); + InitializeLogRouterRequest request = co_await fixture.nextRequest(); + InitializeLogRouterRequest retry = co_await fixture.nextRequest(); + assertSameLogRouterRequest(request, retry); + fixture.reply(request); + co_await timeoutError(changed, 1.0); + ASSERT(fixture.hasRouter()); + fixture.assertPending(); + fixture.cancel(); +} + +TEST_CASE("/LogSystem/RecruitOldLogRouters/LocalReplySuccess") { + OldLogRouterRecruitmentFixture fixture(false); + Future changed = fixture.onConfigChange(); + Future recruitment = fixture.start(); + InitializeLogRouterRequest request = co_await fixture.nextRequest(); + fixture.reply(request); + co_await timeoutError(changed, 1.0); + ASSERT(fixture.hasRouter()); + ReplyPromise failureRequest = co_await fixture.failureRequests(); + + co_await delay(2 * OldLogRouterRecruitmentFixture::retryDelay); + fixture.assertPending(); + ASSERT(!fixture.requests().isReady()); + fixture.cancel(); + ASSERT(recruitment.isError()); + ASSERT_EQ(recruitment.getError().code(), error_code_actor_cancelled); + ASSERT_EQ(failureRequest.getFutureReferenceCount(), 0); +} + +TEST_CASE("/LogSystem/RecruitOldLogRouters/LocalRetryWaitAndCancellation") { + OldLogRouterRecruitmentFixture fixture(false); + Future recruitment = fixture.start(); + InitializeLogRouterRequest request = co_await fixture.nextRequest(); + InitializeLogRouterRequest retry = co_await fixture.nextRequest(); + assertSameLogRouterRequest(request, retry); + co_await delay(2 * OldLogRouterRecruitmentFixture::retryDelay); + fixture.assertPending(); + ASSERT(!fixture.requests().isReady()); + ASSERT(request.reply.getFutureReferenceCount() > 0); + ASSERT(retry.reply.getFutureReferenceCount() > 0); + fixture.cancel(); + ASSERT(recruitment.isError()); + ASSERT_EQ(recruitment.getError().code(), error_code_actor_cancelled); + ASSERT(fixture.hasNoRouters()); + ASSERT_EQ(request.reply.getFutureReferenceCount(), 0); + ASSERT_EQ(retry.reply.getFutureReferenceCount(), 0); +} + +TEST_CASE("/LogSystem/RecruitOldLogRouters/RemoteReplyWait") { + OldLogRouterRecruitmentFixture fixture(true); + Future recruitment = fixture.start(); + InitializeLogRouterRequest request = co_await fixture.nextRequest(); + ASSERT(!request.allowDropInSim); + co_await delay(2 * OldLogRouterRecruitmentFixture::retryDelay); + fixture.assertPending(); + ASSERT(!fixture.requests().isReady()); + ASSERT(fixture.hasNoRouters()); + fixture.reply(request); + co_await timeoutError(recruitment, 1.0); + ASSERT(fixture.hasRouter()); +} + TEST_CASE("/LogSystem/GetPseudoPopTag/LogRouterWithoutMappedLocality") { LocalityData locality; auto logSystem = makeReference(UID(), locality, LogEpoch(1)); diff --git a/fdbserver/worker/worker.cpp b/fdbserver/worker/worker.cpp index 99d7aac4c40..ff1edea1932 100644 --- a/fdbserver/worker/worker.cpp +++ b/fdbserver/worker/worker.cpp @@ -19,6 +19,7 @@ */ #include +#include #include #include #include @@ -28,6 +29,7 @@ #include "flow/Buggify.h" #include "flow/CodeProbe.h" #include "flow/IAsyncFile.h" +#include "fdbrpc/FailureMonitor.h" #include "fdbrpc/Locality.h" #include "fdbclient/GlobalConfig.h" #include "fdbclient/ProcessInterface.h" @@ -44,6 +46,7 @@ #include "flow/ObjectSerializer.h" #include "flow/Platform.h" #include "flow/ProtocolVersion.h" +#include "flow/ScopeExit.h" #include "flow/SystemMonitor.h" #include "flow/TDMetric.h" #include "fdbrpc/simulator.h" @@ -58,6 +61,7 @@ #include "fdbserver/datadistributor/DataDistributor.h" #include "fdbserver/grvproxy/GrvProxyServer.h" #include "fdbserver/logrouter/LogRouter.h" +#include "fdbserver/logsystem/LogSystem.h" #include "fdbserver/core/BackupInterface.h" #include "RoleLineage.h" #include "fdbserver/core/WorkerInterface.h" @@ -1969,6 +1973,171 @@ bool skipInitRspInSim(const UID workerInterfID, const bool allowDropInSim) { return skip; } +Promise cacheLogRouterInitialization(WorkerCache& cache, + InitializeLogRouterRequest const& request) { + Promise ready; + cache.set(request.reqId, ready.getFuture()); + return ready; +} + +bool replyToCachedLogRouter(WorkerCache& cache, InitializeLogRouterRequest const& request) { + if (!cache.exists(request.reqId)) { + return false; + } + forwardPromise(Uncancellable{}, request.reply, cache.get(request.reqId)); + return true; +} + +TEST_CASE("/fdbserver/worker/logRouterInitialization/cachedReplySurvivesDroppedResponse") { + WorkerCache cache; + InitializeLogRouterRequest first{}; + first.reqId = UID(1, 1); + Future firstReply = first.reply.getFuture(); + ASSERT(!replyToCachedLogRouter(cache, first)); + + TLogInterface router(UID(2, 2), UID(2, 2), LocalityData()); + Promise roleLifetime; + Promise initialized = cacheLogRouterInitialization(cache, first); + Future role = cache.removeOnReady(first.reqId, roleLifetime.getFuture()); + initialized.send(router); + ASSERT(!role.isReady()); + ASSERT(!firstReply.isReady()); + + InitializeLogRouterRequest duplicate = first; + duplicate.reply.reset(); + Future duplicateReply = duplicate.reply.getFuture(); + ASSERT(replyToCachedLogRouter(cache, duplicate)); + TLogInterface cachedRouter = co_await timeoutError(duplicateReply, 1.0); + ASSERT_EQ(cachedRouter.id(), router.id()); + ASSERT(!firstReply.isReady()); + ASSERT(cache.exists(first.reqId)); + ASSERT(!role.isReady()); + + roleLifetime.send(Void()); + co_await role; + ASSERT(!cache.exists(first.reqId)); + ASSERT(!replyToCachedLogRouter(cache, duplicate)); +} + +TEST_CASE("/fdbserver/worker/logRouterInitialization/recruitmentRetriesDroppedResponse") { + constexpr double retryDelay = 0.01; + IFailureMonitor& failureMonitor = IFailureMonitor::failureMonitor(); + std::map savedStatus; + ScopeExit restoreStatus([&] { + for (const auto& [address, status] : savedStatus) { + failureMonitor.setStatus(address, status); + } + }); + WorkerInterface worker; + worker.logRouter.getEndpoint(TaskPriority::Worker); + TLogInterface router(UID(2, 2), UID(2, 2), LocalityData()); + router.initEndpoints(); + const Endpoint endpoints[] = { worker.logRouter.getEndpoint(), router.waitFailure.getEndpoint() }; + for (const auto& endpoint : endpoints) { + const NetworkAddress address = endpoint.getPrimaryAddress(); + if (savedStatus.emplace(address, failureMonitor.getState(address)).second) { + failureMonitor.setStatus(address, FailureStatus(false)); + } + ASSERT(failureMonitor.getState(endpoint).isAvailable()); + } + + auto logSystem = makeReference(UID(1, 1), LocalityData(), LogEpoch(1)); + auto logSet = makeReference(); + logSet->locality = 1; + logSet->startVersion = 100; + logSystem->logRouterTags = 1; + logSystem->recoverAt = 200; + logSystem->knownLockedTLogIds[1] = { 2, 4 }; + logSystem->tLogs.push_back(logSet); + Future changed = logSystem->onLogSystemConfigChange(); + WorkerCache cache; + Promise roleLifetime; + Promise initialized; + InitializeLogRouterRequest first{}; + InitializeLogRouterRequest retry{}; + ReplyPromise failureRequest; + Future firstReply; + Future retryReply; + Future role; + Future recruitment; + ScopeExit cancelActors([&] { + recruitment.cancel(); + role.cancel(); + }); + { + const double savedTimeout = SERVER_KNOBS->CC_RERECRUIT_LOG_ROUTER_TIMEOUT; + ScopeExit restoreTimeout( + [savedTimeout] { setServerKnob("cc_rerecruit_log_router_timeout", KnobValueRef::create(savedTimeout)); }); + setServerKnob("cc_rerecruit_log_router_timeout", KnobValueRef::create(retryDelay)); + recruitment = logSystem->recruitOldLogRouters( + { worker }, LogEpoch(2), 1, 100, { LocalityData() }, makeReference(), false); + } + + const char* stage = "first request"; + try { + first = co_await timeoutError(waitOrError(worker.logRouter.getFuture(), recruitment), 1.0); + firstReply = first.reply.getFuture(); + ASSERT(!replyToCachedLogRouter(cache, first)); + initialized = cacheLogRouterInitialization(cache, first); + role = cache.removeOnReady(first.reqId, roleLifetime.getFuture()); + initialized.send(router); + ASSERT(!firstReply.isReady()); + + stage = "retry request"; + retry = co_await timeoutError(waitOrError(worker.logRouter.getFuture(), recruitment), 1.0); + ASSERT(retry.reqId == first.reqId); + ASSERT_EQ(retry.recoveryCount, first.recoveryCount); + ASSERT(retry.routerTag == first.routerTag); + ASSERT_EQ(retry.startVersion, first.startVersion); + ASSERT(retry.tLogLocalities == first.tLogLocalities); + ASSERT(retry.tLogPolicy == first.tLogPolicy); + ASSERT_EQ(retry.locality, first.locality); + ASSERT(retry.recoverAt == first.recoverAt); + ASSERT(retry.knownLockedTLogIds == first.knownLockedTLogIds); + ASSERT_EQ(retry.allowDropInSim, first.allowDropInSim); + ASSERT_EQ(retry.isReplacement, first.isReplacement); + ASSERT(!(first.reply.getFuture() == retry.reply.getFuture())); + retryReply = retry.reply.getFuture(); + ASSERT(replyToCachedLogRouter(cache, retry)); + stage = "cached reply"; + TLogInterface cachedRouter = co_await timeoutError(retryReply, 1.0); + ASSERT_EQ(cachedRouter.id(), router.id()); + stage = "router publication"; + co_await timeoutError(waitOrError(changed, recruitment), 1.0); + ASSERT_EQ(logSet->logRouters.size(), 1); + ASSERT_EQ(logSet->logRouters.front()->get().id(), router.id()); + ASSERT(!firstReply.isReady()); + ASSERT(cache.exists(first.reqId)); + ASSERT(!role.isReady()); + ASSERT(!recruitment.isReady()); + stage = "failure monitoring"; + failureRequest = co_await timeoutError(waitAndForward(router.waitFailure.getFuture()), 1.0); + co_await delay(2 * retryDelay); + ASSERT(!worker.logRouter.getFuture().isReady()); + ASSERT(!recruitment.isReady()); + + recruitment.cancel(); + ASSERT(recruitment.isError()); + ASSERT_EQ(recruitment.getError().code(), error_code_actor_cancelled); + ASSERT_EQ(failureRequest.getFutureReferenceCount(), 0); + firstReply = Future(); + ASSERT_EQ(first.reply.getFutureReferenceCount(), 0); + ASSERT(cache.exists(first.reqId)); + roleLifetime.send(Void()); + co_await role; + ASSERT(!cache.exists(first.reqId)); + } catch (Error& e) { + fprintf(stderr, + "Log router recruitment test failed at %s (cacheReady=%d, firstReplyReady=%d, configChanged=%d): %s\n", + stage, + cache.exists(first.reqId) && cache.get(first.reqId).isReady(), + firstReply.isValid() && firstReply.isReady(), + changed.isReady(), + e.what()); + throw; + } +} + #ifdef FLOW_GRPC_ENABLED Future registerWorkerGrpcServices(UID id, Reference ccr) { if (GrpcServer::instance() == nullptr) { @@ -2635,7 +2804,7 @@ class WorkerServerCore { while (true) { InitializeLogRouterRequest req = co_await interf.logRouter.getFuture(); - if (!logRouterCache.exists(req.reqId)) { + if (!replyToCachedLogRouter(logRouterCache, req)) { LocalLineage _; getCurrentLineage()->modify(&RoleLineage::role) = recruitment::LogRouter; TLogInterface recruited(locality); @@ -2657,19 +2826,18 @@ class WorkerServerCore { DUMPTOKEN(recruited.enablePopRequest); DUMPTOKEN(recruited.snapRequest); - ReplyPromise logRouterReady = req.reply; - logRouterCache.set(req.reqId, logRouterReady.getFuture()); + Promise logRouterReady = cacheLogRouterInitialization(logRouterCache, req); Future logRouterProcess = logRouter(recruited, req, dbInfo); logRouterProcess = logRouterCache.removeOnReady(req.reqId, logRouterProcess); errorForwarders.add( zombie(recruited, forwardError(errors, Role::LOG_ROUTER, recruited.id(), logRouterProcess))); TraceEvent("LogRouterInitRequest", req.reqId).detail("LogRouterId", recruited.id()); + // A lost response must not leave duplicate requests waiting on that response's promise. + logRouterReady.send(recruited); if (!skipInitRspInSim(interf.id(), req.allowDropInSim)) { - logRouterReady.send(recruited); + req.reply.send(recruited); } - } else { - forwardPromise(Uncancellable{}, req.reply, logRouterCache.get(req.reqId)); } } } From f38807d9320a79bae6285ac6ae4f4756220b8ecf Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Mon, 17 Aug 2026 17:58:45 -0700 Subject: [PATCH 002/170] Retire old TLog roles after terminal recovery --- .../clustercontroller/ClusterController.cpp | 15 +- .../clustercontroller/ClusterController.h | 25 --- .../clustercontroller/ClusterRecovery.cpp | 5 +- fdbserver/logsystem/LogSystem.cpp | 26 ++- .../logsystem/LogSystemRecoveryTests.cpp | 67 +++++++ .../include/fdbserver/logsystem/LogSystem.h | 7 + fdbserver/workloads/OldTLogRetirement.cpp | 181 ++++++++++++++++++ tests/CMakeLists.txt | 1 + tests/fast/OldTLogRetirement.toml | 31 +++ 9 files changed, 323 insertions(+), 35 deletions(-) create mode 100644 fdbserver/workloads/OldTLogRetirement.cpp create mode 100644 tests/fast/OldTLogRetirement.toml diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index 629eb33192d..eda1111bcc4 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -1444,6 +1444,7 @@ void clusterRegisterMaster(ClusterControllerData* self, RegisterMasterRequest co } if (req.recoveryState == RecoveryState::FULLY_RECOVERED) { + ASSERT(req.logSystemConfig.oldTLogs.empty()); self->db.unfinishedRecoveries = 0; } @@ -3937,7 +3938,7 @@ TEST_CASE("/fdbserver/clustercontroller/deferBetterMasterRecoveryUntilInitialSto return Void(); } -TEST_CASE("/fdbserver/clustercontroller/recoverForExcludedOldTLogLocality") { +TEST_CASE("/fdbserver/clustercontroller/ignoreExcludedOldTLogLocality") { ClusterControllerData data(ClusterControllerFullInterface(), LocalityData(), ServerCoordinators(Reference( @@ -3969,12 +3970,24 @@ TEST_CASE("/fdbserver/clustercontroller/recoverForExcludedOldTLogLocality") { oldTLogSet.tLogs.push_back(OptionalInterface(oldTLog)); OldTLogConf oldTLogConf; oldTLogConf.tLogs.push_back(oldTLogSet); + LocalityData currentLocality; + currentLocality.set(LocalityData::keyProcessId, Standalone(std::string{ "current-tlog" })); + TLogInterface currentTLog(currentLocality); + TLogSet currentTLogSet; + currentTLogSet.tLogs.push_back(OptionalInterface(currentTLog)); ServerDBInfo dbInfo; dbInfo.master.locality = masterLocality; + dbInfo.logSystemConfig.tLogs.push_back(currentTLogSet); dbInfo.logSystemConfig.oldTLogs.push_back(oldTLogConf); dbInfo.recoveryState = RecoveryState::FULLY_RECOVERED; data.db.serverInfo->set(dbInfo); + // An unregistered current log stops unrelated placement comparisons after checking the old roles. + ASSERT(!data.betterMasterExists()); + + auto& currentWorker = data.id_worker[currentLocality.processId()]; + currentWorker.details.interf = WorkerInterface(currentLocality); + currentWorker.priorityInfo.isExcluded = true; ASSERT(data.betterMasterExists()); return Void(); } diff --git a/fdbserver/clustercontroller/ClusterController.h b/fdbserver/clustercontroller/ClusterController.h index 71bb046d8b9..3978d678a56 100644 --- a/fdbserver/clustercontroller/ClusterController.h +++ b/fdbserver/clustercontroller/ClusterController.h @@ -2458,31 +2458,6 @@ class ClusterControllerData { std::vector backup_workers; std::set backup_addresses; - if (dbi.recoveryState == RecoveryState::FULLY_RECOVERED) { - for (const auto& oldLog : dbi.logSystemConfig.oldTLogs) { - for (const auto& logSet : oldLog.tLogs) { - for (const auto& tlog : logSet.tLogs) { - if (!tlog.present()) { - continue; - } - - auto tlogWorker = std::find_if(id_worker.begin(), id_worker.end(), [&tlog](const auto& worker) { - return worker.second.details.interf.address() == tlog.interf().address(); - }); - const auto& locality = tlogWorker == id_worker.end() - ? tlog.interf().filteredLocality - : tlogWorker->second.details.interf.locality; - if (db.config.isExcludedServer(tlog.interf().addresses(), locality)) { - TraceEvent("BetterMasterExists", id) - .detail("Reason", "OldTLogExcluded") - .detail("ProcessID", locality.processId()); - return true; - } - } - } - } - } - for (auto& logSet : dbi.logSystemConfig.tLogs) { for (auto& it : logSet.tLogs) { auto tlogWorker = id_worker.find(it.interf().filteredLocality.processId()); diff --git a/fdbserver/clustercontroller/ClusterRecovery.cpp b/fdbserver/clustercontroller/ClusterRecovery.cpp index fc30414f4b7..b411d6ab21a 100644 --- a/fdbserver/clustercontroller/ClusterRecovery.cpp +++ b/fdbserver/clustercontroller/ClusterRecovery.cpp @@ -540,8 +540,6 @@ Future trackTlogRecovery(Reference self, configuration.expectedLogSets(!self->primaryDcId.empty() ? self->primaryDcId[0] : Optional())) .detail("RecoveryCount", newState.recoveryCount); co_await self->cstate.write(newState, finalUpdate); - // Keep oldLogData in memory even after the coordinated state drops old generations. ServerDBInfo uses - // it to keep old-generation TLogs serving in case this master has to run recovery again. if (self->cstateUpdated.canBeSet()) { self->cstateUpdated.send(Void()); } @@ -554,6 +552,8 @@ Future trackTlogRecovery(Reference self, } if (finalUpdate) { + oldLogSystems->get()->stopRejoins(); + self->logSystem->retireOldLogRoles(newState); self->recoveryState = RecoveryState::FULLY_RECOVERED; TraceEvent(getRecoveryEventName(ClusterRecoveryEventType::CLUSTER_RECOVERY_STATE_EVENT_NAME).c_str(), self->dbgid) @@ -584,7 +584,6 @@ Future trackTlogRecovery(Reference self, self->registrationTrigger.trigger(); if (finalUpdate) { - oldLogSystems->get()->stopRejoins(); rejoinRequests = rejoinRequestHandler(self); co_return; } diff --git a/fdbserver/logsystem/LogSystem.cpp b/fdbserver/logsystem/LogSystem.cpp index 43ae3e14122..49e06b913a1 100644 --- a/fdbserver/logsystem/LogSystem.cpp +++ b/fdbserver/logsystem/LogSystem.cpp @@ -607,6 +607,20 @@ void LogSystem::coreStateWritten(DBCoreState const& newState) { } } +void LogSystem::retireOldLogRoles(DBCoreState const& finalState) { + ASSERT(finalState.recoveryCount == epoch); + ASSERT(finalState.oldTLogData.empty()); + ASSERT(expectedLogSets > 0 && finalState.tLogs.size() == static_cast(expectedLogSets)); + ASSERT(recoveryCompleteWrittenToCoreState.get()); + if (!oldLogRolesRetired) { + oldLogRolesRetired = true; + TraceEvent("RetireOldTLogRoles", dbgid) + .detail("RecoveryCount", epoch) + .detail("OldGenerations", oldLogData.size()); + logSystemConfigChanged.trigger(); + } +} + Future LogSystem::onError() const { // Never returns normally, but throws an error if the subsystem stops working while (true) { @@ -1116,12 +1130,12 @@ LogSystemConfig LogSystem::getLogSystemConfig() const { } } - // ServerDBInfo uses oldTLogs to keep old-generation TLog roles from displacing - // themselves while this cluster controller is alive. Durable state/logsKey can - // drop recovered old generations earlier, but these roles may still be needed - // if this recovery has to run again. - for (const auto& oldData : oldLogData) { - logSystemConfig.oldTLogs.push_back(toOldTLogConf(oldData)); + // A storage-recovered state can still need old TLog locks for remote recruitment. + // Keep those roles alive until the terminal recovery state is durable. + if (!oldLogRolesRetired) { + for (const auto& oldData : oldLogData) { + logSystemConfig.oldTLogs.push_back(toOldTLogConf(oldData)); + } } return logSystemConfig; } diff --git a/fdbserver/logsystem/LogSystemRecoveryTests.cpp b/fdbserver/logsystem/LogSystemRecoveryTests.cpp index 70baf43d283..18069a42aca 100644 --- a/fdbserver/logsystem/LogSystemRecoveryTests.cpp +++ b/fdbserver/logsystem/LogSystemRecoveryTests.cpp @@ -58,6 +58,73 @@ std::tuple, bool> makeLogGroupResults( void forceLinkLogSystemRecoveryTests() {} +TEST_CASE("/LogSystem/RetireOldLogRoles/FinalCoreState") { + constexpr LogEpoch epoch = 2; + LocalityData locality; + auto logSystem = makeReference(UID(), locality, epoch); + logSystem->logSystemType = LogSystemType::tagPartitioned; + logSystem->expectedLogSets = 2; + logSystem->oldestBackupEpoch = epoch - 1; + logSystem->repopulateRegionAntiQuorum = 1; + logSystem->recoveryComplete = Void(); + logSystem->remoteRecovery = Never(); + logSystem->remoteRecoveryComplete = Never(); + logSystem->hasRemoteServers = true; + logSystem->tLogs.push_back(makeSingleLogSet({ TLogInterface(locality) })); + OldLogData old; + old.epoch = epoch - 1; + old.epochEnd = 100; + old.tLogs.push_back(makeSingleLogSet({ TLogInterface(locality) })); + logSystem->oldLogData.push_back(old); + const auto oldRoles = logSystem->getLogSystemConfig().oldTLogs; + + DBCoreState partialState; + logSystem->toCoreState(partialState); + partialState.recoveryCount = epoch; + ASSERT_EQ(partialState.oldTLogData.size(), 1); + logSystem->coreStateWritten(partialState); + ASSERT(!logSystem->recoveryCompleteWrittenToCoreState.get()); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + + // Local recovery and backups can finish before the expected remote log set exists. + logSystem->oldestBackupEpoch = epoch; + logSystem->toCoreState(partialState); + ASSERT(partialState.oldTLogData.empty()); + ASSERT_EQ(partialState.tLogs.size(), 1); + logSystem->coreStateWritten(partialState); + ASSERT(logSystem->recoveryCompleteWrittenToCoreState.get()); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + + logSystem->tLogs.push_back(makeSingleLogSet({ TLogInterface(locality) }, false)); + logSystem->remoteRecovery = Void(); + ASSERT(!logSystem->remoteRecoveryComplete.isReady()); + DBCoreState finalState; + logSystem->toCoreState(finalState); + finalState.recoveryCount = epoch; + ASSERT(finalState.oldTLogData.empty()); + ASSERT_EQ(finalState.tLogs.size(), logSystem->expectedLogSets); + logSystem->coreStateWritten(finalState); + LogSystemConfig expected = logSystem->getLogSystemConfig(); + ASSERT_EQ(expected.tLogs.size(), 2); + ASSERT(expected.oldTLogs == oldRoles); + expected.oldTLogs.clear(); + + Future configChanged = logSystem->onLogSystemConfigChange(); + ASSERT(!configChanged.isReady()); + logSystem->retireOldLogRoles(finalState); + ASSERT(configChanged.isReady() && !configChanged.isError()); + ASSERT(!logSystem->remoteRecoveryComplete.isReady()); + ASSERT(logSystem->getLogSystemConfig() == expected); + ASSERT_EQ(logSystem->oldLogData.size(), 1); + ASSERT(logSystem->oldLogData.front().tLogs.front() == old.tLogs.front()); + + Future unchanged = logSystem->onLogSystemConfigChange(); + logSystem->retireOldLogRoles(finalState); + ASSERT(!unchanged.isReady()); + ASSERT(logSystem->getLogSystemConfig() == expected); + return Void(); +} + TEST_CASE("/LogSystem/GetPseudoPopTag/LogRouterWithoutMappedLocality") { LocalityData locality; auto logSystem = makeReference(UID(), locality, LogEpoch(1)); diff --git a/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h b/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h index 01f443cdcf6..c4917187333 100644 --- a/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h +++ b/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h @@ -377,6 +377,10 @@ struct LogSystem : ReferenceCounted { void coreStateWritten(DBCoreState const& newState); + // Requires the successfully committed terminal recovery state with every expected current log set. + // Finalizing a partial state for a coordinator change is insufficient. + void retireOldLogRoles(DBCoreState const& finalState); + Future onError() const; static Future pushResetChecker(Reference self, NetworkAddress addr); @@ -547,6 +551,9 @@ struct LogSystem : ReferenceCounted { static Future lockTLog(UID myID, Reference>> tlog); template static std::vector getReadyNonError(std::vector> const& futures); + +private: + bool oldLogRolesRetired = false; }; // Recovery version calculation for version vector unicast diff --git a/fdbserver/workloads/OldTLogRetirement.cpp b/fdbserver/workloads/OldTLogRetirement.cpp new file mode 100644 index 00000000000..388626a0ec8 --- /dev/null +++ b/fdbserver/workloads/OldTLogRetirement.cpp @@ -0,0 +1,181 @@ +/* + * OldTLogRetirement.cpp + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "fdbclient/ManagementAPI.h" +#include "fdbrpc/simulator.h" +#include "fdbserver/core/Knobs.h" +#include "fdbserver/core/RecoveryState.h" +#include "fdbserver/core/ServerDBInfo.h" +#include "fdbserver/tester/workloads.h" +#include "flow/CodeProbe.h" +#include "flow/CoroUtils.h" +#include "flow/ScopeExit.h" + +#include +#include +#include +#include + +class OldTLogRetirementWorkload : public TestWorkload { +public: + static constexpr auto NAME = "OldTLogRetirement"; + explicit OldTLogRetirementWorkload(WorkloadContext const& wcx) + : TestWorkload(wcx), enabled(!clientId && g_network->isSimulated()), + operationTimeout(getOption(options, "operationTimeout"_sr, 120.0)) {} + + void disableFailureInjectionWorkloads(std::set& out) const override { out.insert("all"); } + Future start(Database const& cx) override { return enabled ? run(cx) : Void(); } + Future check(Database const& cx) override { return !enabled || completed; } + void getMetrics(std::vector&) override {} + +private: + const bool enabled; + const double operationTimeout; + bool completed = false; + static constexpr double cleanupTimeout = 30.0; + + static double stabilityDuration() { + return SERVER_KNOBS->SINGLETON_RECRUIT_BME_DELAY + SERVER_KNOBS->WAIT_FOR_GOOD_RECRUITMENT_DELAY + + 2 * SERVER_KNOBS->CHECK_OUTSTANDING_INTERVAL + 1.0; + } + + static std::array singletonMembers(const ServerDBInfo& info) { + return { info.distributor.present() ? info.distributor.get().id() : UID(), + info.ratekeeper.present() ? info.ratekeeper.get().id() : UID(), + info.consistencyScan.present() ? info.consistencyScan.get().id() : UID() }; + } + + static std::vector transactionSystemMembers(const ServerDBInfo& info) { + std::vector members{ info.master.id(), info.logSystemConfig.recruitmentID }; + for (const auto& logSet : info.logSystemConfig.tLogs) { + for (const auto& log : logSet.tLogs) { + members.push_back(log.id()); + } + } + for (const auto& proxy : info.client.commitProxies) { + members.push_back(proxy.id()); + } + for (const auto& proxy : info.client.grvProxies) { + members.push_back(proxy.id()); + } + for (const auto& resolver : info.resolvers) { + members.push_back(resolver.id()); + } + std::sort(members.begin(), members.end()); + return members; + } + + Future stableBaseline() { + while (true) { + while (dbInfo->get().recoveryState != RecoveryState::FULLY_RECOVERED) { + co_await dbInfo->onChange(); + } + const ServerDBInfo baseline = dbInfo->get(); + const auto members = transactionSystemMembers(baseline); + const auto singletons = singletonMembers(baseline); + Future stable = delay(stabilityDuration()); + while (dbInfo->get().recoveryState == RecoveryState::FULLY_RECOVERED && + dbInfo->get().recoveryCount == baseline.recoveryCount && + transactionSystemMembers(dbInfo->get()) == members && + singletonMembers(dbInfo->get()) == singletons) { + if (stable.isReady()) { + co_return dbInfo->get(); + } + co_await (dbInfo->onChange() || stable); + } + } + } + + Future excludeAndCheck(Database cx, AddressExclusion excluded, UID oldLogId, DBRecoveryCount previousCount) { + TraceEvent("OldTLogRetirementExclude").detail("Address", excluded).detail("RecoveryCount", previousCount); + co_await excludeServers(cx, { excluded }); + // A successful recovery writes the coordinated-state recovery count twice. + const DBRecoveryCount expectedCount = previousCount + 2; + while (true) { + ASSERT_LE(dbInfo->get().recoveryCount, expectedCount); + if (dbInfo->get().recoveryCount == expectedCount && + dbInfo->get().recoveryState == RecoveryState::FULLY_RECOVERED) { + break; + } + co_await dbInfo->onChange(); + } + const auto members = transactionSystemMembers(dbInfo->get()); + ASSERT(!dbInfo->get().logSystemConfig.hasTLog(oldLogId)); + ASSERT(dbInfo->get().logSystemConfig.oldTLogs.empty()); + for (const auto& log : dbInfo->get().logSystemConfig.allPresentLogs()) { + ASSERT(!excluded.excludes(log.address())); + } + TraceEvent("OldTLogRetirementFirstRecovery").detail("RecoveryCount", expectedCount); + auto singletons = singletonMembers(dbInfo->get()); + Future stable = delay(stabilityDuration()); + while (true) { + ASSERT_EQ(dbInfo->get().recoveryCount, expectedCount); + ASSERT(dbInfo->get().recoveryState == RecoveryState::FULLY_RECOVERED); + ASSERT(dbInfo->get().logSystemConfig.oldTLogs.empty()); + ASSERT(transactionSystemMembers(dbInfo->get()) == members); + const auto currentSingletons = singletonMembers(dbInfo->get()); + if (currentSingletons != singletons) { + // Singleton replacement postpones the controller's better-master check. + singletons = currentSingletons; + stable = delay(stabilityDuration()); + } + if (stable.isReady()) { + break; + } + co_await (dbInfo->onChange() || stable); + } + CODE_PROBE(true, "Excluded old TLog retires without a second recovery"); + TraceEvent("OldTLogRetirementStable").detail("RecoveryCount", expectedCount); + } + + Future run(Database cx) { + const ServerDBInfo baseline = co_await timeoutError(stableBaseline(), operationTimeout); + Optional selected; + for (const auto& log : baseline.logSystemConfig.allLocalLogs()) { + if (log.address() != baseline.clusterInterface.address() && log.address() != baseline.master.address() && + !g_simulator->isProtectedAddress(log.address())) { + selected = log; + break; + } + } + ASSERT(selected.present()); + const AddressExclusion excluded(selected.get().address().ip, selected.get().address().port); + bool cleanupStarted = false; + ScopeExit cleanupOnCancel([cx, excluded, &cleanupStarted] { + if (!cleanupStarted) { + uncancellable(timeoutError(includeServers(cx, { excluded }), cleanupTimeout)); + } + }); + ErrorOr result = co_await coro::errorOr( + timeoutError(excludeAndCheck(cx, excluded, selected.get().id(), baseline.recoveryCount), operationTimeout)); + cleanupStarted = true; + ErrorOr cleanup = + co_await coro::errorOr(uncancellable(timeoutError(includeServers(cx, { excluded }), cleanupTimeout))); + if (result.isError()) { + throw result.getError(); + } + if (cleanup.isError()) { + throw cleanup.getError(); + } + completed = true; + } +}; + +WorkloadFactory OldTLogRetirementWorkloadFactory; diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 72d5f3f5cd7..7f474ef8154 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -212,6 +212,7 @@ if(WITH_PYTHON) add_fdb_test(TEST_FILES fast/NativeCdcRetiredRecovery.toml) add_fdb_test(TEST_FILES fast/NativeCdcRetiredRecoveryMultiRegion.toml) add_fdb_test(TEST_FILES fast/NativeCdcSatellite.toml) + add_fdb_test(TEST_FILES fast/OldTLogRetirement.toml) add_fdb_test( TEST_FILES fast/NativeCdcDisableRestart-1.toml fast/NativeCdcDisableRestart-2.toml) diff --git a/tests/fast/OldTLogRetirement.toml b/tests/fast/OldTLogRetirement.toml new file mode 100644 index 00000000000..4f1c16674b1 --- /dev/null +++ b/tests/fast/OldTLogRetirement.toml @@ -0,0 +1,31 @@ +[configuration] +config = 'double' +singleRegion = true +datacenters = 1 +machineCount = 16 +processesPerMachine = 1 +coordinators = 1 +desiredTLogCount = 4 +commitProxyCount = 2 +grvProxyCount = 2 +resolverCount = 1 +statelessProcessClassesPerDC = 4 +disableTss = true +buggify = false +faultInjection = false + +[[test]] +testTitle = 'OldTLogRetirement' +runFailureWorkloads = false +connectionFailuresDisableDuration = 1000000 +timeout = 300 + + [[test.workload]] + testName = 'Cycle' + nodeCount = 1000 + transactionsPerSecond = 100.0 + testDuration = 60.0 + + [[test.workload]] + testName = 'OldTLogRetirement' + operationTimeout = 120.0 From 48c23072740ed9604db9a8796be6a1d6d0f7c0b3 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Mon, 17 Aug 2026 18:13:06 -0700 Subject: [PATCH 003/170] Remove standalone old TLog retirement workload --- fdbserver/workloads/OldTLogRetirement.cpp | 181 ---------------------- tests/CMakeLists.txt | 1 - tests/fast/OldTLogRetirement.toml | 31 ---- 3 files changed, 213 deletions(-) delete mode 100644 fdbserver/workloads/OldTLogRetirement.cpp delete mode 100644 tests/fast/OldTLogRetirement.toml diff --git a/fdbserver/workloads/OldTLogRetirement.cpp b/fdbserver/workloads/OldTLogRetirement.cpp deleted file mode 100644 index 388626a0ec8..00000000000 --- a/fdbserver/workloads/OldTLogRetirement.cpp +++ /dev/null @@ -1,181 +0,0 @@ -/* - * OldTLogRetirement.cpp - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -#include "fdbclient/ManagementAPI.h" -#include "fdbrpc/simulator.h" -#include "fdbserver/core/Knobs.h" -#include "fdbserver/core/RecoveryState.h" -#include "fdbserver/core/ServerDBInfo.h" -#include "fdbserver/tester/workloads.h" -#include "flow/CodeProbe.h" -#include "flow/CoroUtils.h" -#include "flow/ScopeExit.h" - -#include -#include -#include -#include - -class OldTLogRetirementWorkload : public TestWorkload { -public: - static constexpr auto NAME = "OldTLogRetirement"; - explicit OldTLogRetirementWorkload(WorkloadContext const& wcx) - : TestWorkload(wcx), enabled(!clientId && g_network->isSimulated()), - operationTimeout(getOption(options, "operationTimeout"_sr, 120.0)) {} - - void disableFailureInjectionWorkloads(std::set& out) const override { out.insert("all"); } - Future start(Database const& cx) override { return enabled ? run(cx) : Void(); } - Future check(Database const& cx) override { return !enabled || completed; } - void getMetrics(std::vector&) override {} - -private: - const bool enabled; - const double operationTimeout; - bool completed = false; - static constexpr double cleanupTimeout = 30.0; - - static double stabilityDuration() { - return SERVER_KNOBS->SINGLETON_RECRUIT_BME_DELAY + SERVER_KNOBS->WAIT_FOR_GOOD_RECRUITMENT_DELAY + - 2 * SERVER_KNOBS->CHECK_OUTSTANDING_INTERVAL + 1.0; - } - - static std::array singletonMembers(const ServerDBInfo& info) { - return { info.distributor.present() ? info.distributor.get().id() : UID(), - info.ratekeeper.present() ? info.ratekeeper.get().id() : UID(), - info.consistencyScan.present() ? info.consistencyScan.get().id() : UID() }; - } - - static std::vector transactionSystemMembers(const ServerDBInfo& info) { - std::vector members{ info.master.id(), info.logSystemConfig.recruitmentID }; - for (const auto& logSet : info.logSystemConfig.tLogs) { - for (const auto& log : logSet.tLogs) { - members.push_back(log.id()); - } - } - for (const auto& proxy : info.client.commitProxies) { - members.push_back(proxy.id()); - } - for (const auto& proxy : info.client.grvProxies) { - members.push_back(proxy.id()); - } - for (const auto& resolver : info.resolvers) { - members.push_back(resolver.id()); - } - std::sort(members.begin(), members.end()); - return members; - } - - Future stableBaseline() { - while (true) { - while (dbInfo->get().recoveryState != RecoveryState::FULLY_RECOVERED) { - co_await dbInfo->onChange(); - } - const ServerDBInfo baseline = dbInfo->get(); - const auto members = transactionSystemMembers(baseline); - const auto singletons = singletonMembers(baseline); - Future stable = delay(stabilityDuration()); - while (dbInfo->get().recoveryState == RecoveryState::FULLY_RECOVERED && - dbInfo->get().recoveryCount == baseline.recoveryCount && - transactionSystemMembers(dbInfo->get()) == members && - singletonMembers(dbInfo->get()) == singletons) { - if (stable.isReady()) { - co_return dbInfo->get(); - } - co_await (dbInfo->onChange() || stable); - } - } - } - - Future excludeAndCheck(Database cx, AddressExclusion excluded, UID oldLogId, DBRecoveryCount previousCount) { - TraceEvent("OldTLogRetirementExclude").detail("Address", excluded).detail("RecoveryCount", previousCount); - co_await excludeServers(cx, { excluded }); - // A successful recovery writes the coordinated-state recovery count twice. - const DBRecoveryCount expectedCount = previousCount + 2; - while (true) { - ASSERT_LE(dbInfo->get().recoveryCount, expectedCount); - if (dbInfo->get().recoveryCount == expectedCount && - dbInfo->get().recoveryState == RecoveryState::FULLY_RECOVERED) { - break; - } - co_await dbInfo->onChange(); - } - const auto members = transactionSystemMembers(dbInfo->get()); - ASSERT(!dbInfo->get().logSystemConfig.hasTLog(oldLogId)); - ASSERT(dbInfo->get().logSystemConfig.oldTLogs.empty()); - for (const auto& log : dbInfo->get().logSystemConfig.allPresentLogs()) { - ASSERT(!excluded.excludes(log.address())); - } - TraceEvent("OldTLogRetirementFirstRecovery").detail("RecoveryCount", expectedCount); - auto singletons = singletonMembers(dbInfo->get()); - Future stable = delay(stabilityDuration()); - while (true) { - ASSERT_EQ(dbInfo->get().recoveryCount, expectedCount); - ASSERT(dbInfo->get().recoveryState == RecoveryState::FULLY_RECOVERED); - ASSERT(dbInfo->get().logSystemConfig.oldTLogs.empty()); - ASSERT(transactionSystemMembers(dbInfo->get()) == members); - const auto currentSingletons = singletonMembers(dbInfo->get()); - if (currentSingletons != singletons) { - // Singleton replacement postpones the controller's better-master check. - singletons = currentSingletons; - stable = delay(stabilityDuration()); - } - if (stable.isReady()) { - break; - } - co_await (dbInfo->onChange() || stable); - } - CODE_PROBE(true, "Excluded old TLog retires without a second recovery"); - TraceEvent("OldTLogRetirementStable").detail("RecoveryCount", expectedCount); - } - - Future run(Database cx) { - const ServerDBInfo baseline = co_await timeoutError(stableBaseline(), operationTimeout); - Optional selected; - for (const auto& log : baseline.logSystemConfig.allLocalLogs()) { - if (log.address() != baseline.clusterInterface.address() && log.address() != baseline.master.address() && - !g_simulator->isProtectedAddress(log.address())) { - selected = log; - break; - } - } - ASSERT(selected.present()); - const AddressExclusion excluded(selected.get().address().ip, selected.get().address().port); - bool cleanupStarted = false; - ScopeExit cleanupOnCancel([cx, excluded, &cleanupStarted] { - if (!cleanupStarted) { - uncancellable(timeoutError(includeServers(cx, { excluded }), cleanupTimeout)); - } - }); - ErrorOr result = co_await coro::errorOr( - timeoutError(excludeAndCheck(cx, excluded, selected.get().id(), baseline.recoveryCount), operationTimeout)); - cleanupStarted = true; - ErrorOr cleanup = - co_await coro::errorOr(uncancellable(timeoutError(includeServers(cx, { excluded }), cleanupTimeout))); - if (result.isError()) { - throw result.getError(); - } - if (cleanup.isError()) { - throw cleanup.getError(); - } - completed = true; - } -}; - -WorkloadFactory OldTLogRetirementWorkloadFactory; diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 7f474ef8154..72d5f3f5cd7 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -212,7 +212,6 @@ if(WITH_PYTHON) add_fdb_test(TEST_FILES fast/NativeCdcRetiredRecovery.toml) add_fdb_test(TEST_FILES fast/NativeCdcRetiredRecoveryMultiRegion.toml) add_fdb_test(TEST_FILES fast/NativeCdcSatellite.toml) - add_fdb_test(TEST_FILES fast/OldTLogRetirement.toml) add_fdb_test( TEST_FILES fast/NativeCdcDisableRestart-1.toml fast/NativeCdcDisableRestart-2.toml) diff --git a/tests/fast/OldTLogRetirement.toml b/tests/fast/OldTLogRetirement.toml deleted file mode 100644 index 4f1c16674b1..00000000000 --- a/tests/fast/OldTLogRetirement.toml +++ /dev/null @@ -1,31 +0,0 @@ -[configuration] -config = 'double' -singleRegion = true -datacenters = 1 -machineCount = 16 -processesPerMachine = 1 -coordinators = 1 -desiredTLogCount = 4 -commitProxyCount = 2 -grvProxyCount = 2 -resolverCount = 1 -statelessProcessClassesPerDC = 4 -disableTss = true -buggify = false -faultInjection = false - -[[test]] -testTitle = 'OldTLogRetirement' -runFailureWorkloads = false -connectionFailuresDisableDuration = 1000000 -timeout = 300 - - [[test.workload]] - testName = 'Cycle' - nodeCount = 1000 - transactionsPerSecond = 100.0 - testDuration = 60.0 - - [[test.workload]] - testName = 'OldTLogRetirement' - operationTimeout = 120.0 From 4436a6005bd82771ed04b46ce4e31b9eaf26ca50 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Mon, 17 Aug 2026 20:02:31 -0700 Subject: [PATCH 004/170] Wait for remote log prefixes before retiring old logs With repopulate_anti_quorum=1, storage recovery can finish before current remote TLogs have copied the preceding log generations. Keep that history in coordinator state until every lagging remote TLog has durable progress at the current local start version. Preserve STORAGE_RECOVERED separately so a lost region can still be removed. Cover partial recruitment, the real old-to-current cursor handoff, interrupted waits, and stale rejoin replies with focused regression tests. --- .../clustercontroller/ClusterRecovery.cpp | 4 +- fdbserver/logsystem/LogSystem.cpp | 98 ++++- .../logsystem/LogSystemRecoveryTests.cpp | 348 +++++++++++++++++- .../include/fdbserver/logsystem/LogSystem.h | 6 + 4 files changed, 448 insertions(+), 8 deletions(-) diff --git a/fdbserver/clustercontroller/ClusterRecovery.cpp b/fdbserver/clustercontroller/ClusterRecovery.cpp index b411d6ab21a..07a5a3ba1fb 100644 --- a/fdbserver/clustercontroller/ClusterRecovery.cpp +++ b/fdbserver/clustercontroller/ClusterRecovery.cpp @@ -531,6 +531,8 @@ Future trackTlogRecovery(Reference self, bool allLogs = newState.tLogs.size() == configuration.expectedLogSets(!self->primaryDcId.empty() ? self->primaryDcId[0] : Optional()); + // Anti-quorum recovery must still permit removing a lost region while its old history remains durable. + bool storageRecovered = newState.oldTLogData.empty() || self->logSystem->storageRecovered(); bool finalUpdate = newState.oldTLogData.empty() && allLogs; TraceEvent("TrackTLogRecovery") .detail("FinalUpdate", finalUpdate) @@ -565,7 +567,7 @@ Future trackTlogRecovery(Reference self, self->dbgid) .detail("ActiveGenerations", 1) .trackLatest(self->clusterRecoveryGenerationsEventHolder->trackingKey); - } else if (newState.oldTLogData.empty() && self->recoveryState < RecoveryState::STORAGE_RECOVERED) { + } else if (storageRecovered && self->recoveryState < RecoveryState::STORAGE_RECOVERED) { self->recoveryState = RecoveryState::STORAGE_RECOVERED; TraceEvent(getRecoveryEventName(ClusterRecoveryEventType::CLUSTER_RECOVERY_STATE_EVENT_NAME).c_str(), self->dbgid) diff --git a/fdbserver/logsystem/LogSystem.cpp b/fdbserver/logsystem/LogSystem.cpp index 49e06b913a1..d4a358bfe4c 100644 --- a/fdbserver/logsystem/LogSystem.cpp +++ b/fdbserver/logsystem/LogSystem.cpp @@ -491,6 +491,11 @@ Reference LogSystem::fromOldLogSystemConfig(UID const& dbgid, } void LogSystem::purgeOldRecoveredGenerationsCoreState(DBCoreState& newState) { + // Anti-quorum recovery may still need these generations after a controller or router failure. + // Do not let incremental purging bypass the remote-prefix durability barrier. + if (!remoteLogPrefixRecovered()) { + return; + } Version oldestGenerationRecoverAtVersion = std::min(recoveredVersion->get(), remoteRecoveredVersion->get()); TraceEvent("ToCoreStateOldestGenerationRecoverAtVersion") .detail("RecoveredVersion", recoveredVersion->get()) @@ -537,6 +542,8 @@ void LogSystem::toCoreState(DBCoreState& newState) const { if (remoteRecoveryComplete.isValid() && remoteRecoveryComplete.isError()) throw remoteRecoveryComplete.getError(); + if (remoteLogPrefixComplete.isValid() && remoteLogPrefixComplete.isError()) + throw remoteLogPrefixComplete.getError(); newState.tLogs.clear(); newState.logRouterTags = logRouterTags; @@ -554,9 +561,7 @@ void LogSystem::toCoreState(DBCoreState& newState) const { } newState.oldTLogData.clear(); - if (!recoveryComplete.isValid() || !recoveryComplete.isReady() || - (repopulateRegionAntiQuorum == 0 && (!remoteRecoveryComplete.isValid() || !remoteRecoveryComplete.isReady())) || - epoch != oldestBackupEpoch) { + if (!storageRecovered() || !remoteLogPrefixRecovered()) { for (const auto& oldData : oldLogData) { newState.oldTLogData.push_back(toOldTLogCoreData(oldData)); TraceEvent("BWToCore") @@ -572,10 +577,23 @@ void LogSystem::toCoreState(DBCoreState& newState) const { newState.logSystemType = logSystemType; } +bool LogSystem::storageRecovered() const { + return recoveryComplete.isValid() && recoveryComplete.isReady() && !recoveryComplete.isError() && + (repopulateRegionAntiQuorum != 0 || (remoteStorageRecovered() && !remoteRecoveryComplete.isError())) && + epoch == oldestBackupEpoch; +} + bool LogSystem::remoteStorageRecovered() const { return remoteRecoveryComplete.isValid() && remoteRecoveryComplete.isReady(); } +bool LogSystem::remoteLogPrefixRecovered() const { + return repopulateRegionAntiQuorum == 0 || !hasRemoteServers || oldLogData.empty() || + (remoteStorageRecovered() && !remoteRecoveryComplete.isError()) || + (remoteLogPrefixComplete.isValid() && remoteLogPrefixComplete.isReady() && + !remoteLogPrefixComplete.isError()); +} + Future LogSystem::onCoreStateChanged() const { std::vector> changes; changes.push_back(Never()); @@ -588,6 +606,9 @@ Future LogSystem::onCoreStateChanged() const { if (remoteRecoveryComplete.isValid() && !remoteRecoveryComplete.isReady()) { changes.push_back(remoteRecoveryComplete); } + if (remoteLogPrefixComplete.isValid() && !remoteLogPrefixComplete.isReady()) { + changes.push_back(remoteLogPrefixComplete); + } changes.push_back(backupWorkerChanged.onTrigger()); // changes to oldestBackupEpoch changes.push_back(recoveredVersion->onChange()); changes.push_back(remoteRecoveredVersion->onChange()); @@ -596,6 +617,7 @@ Future LogSystem::onCoreStateChanged() const { void LogSystem::coreStateWritten(DBCoreState const& newState) { if (newState.oldTLogData.empty()) { + ASSERT(remoteLogPrefixRecovered()); recoveryCompleteWrittenToCoreState.set(true); } for (auto& t : newState.tLogs) { @@ -607,11 +629,78 @@ void LogSystem::coreStateWritten(DBCoreState const& newState) { } } +namespace { + +Future waitForRemoteTLogPrefixDurable(UID dbgid, + Reference>> tlog, + Version handoffVersion) { + TraceEvent("WaitForRemoteTLogPrefixDurable", dbgid) + .detail("TLogID", tlog->get().id()) + .detail("HandoffVersion", handoffVersion); + while (true) { + Future changed = tlog->onChange(); + if (!tlog->get().present()) { + co_await changed; + continue; + } + + auto result = co_await race(tlog->get().interf().getQueuingMetrics.getReplyUnlessFailedFor( + TLogQueuingMetricsRequest(), + SERVER_KNOBS->TLOG_TIMEOUT, + SERVER_KNOBS->MASTER_FAILURE_SLOPE_DURING_RECOVERY), + changed); + if (result.index() != 0 || changed.isReady()) { + continue; + } + ErrorOr reply = std::get<0>(result); + if (reply.isError()) { + if (reply.getError().code() == error_code_actor_cancelled) { + throw reply.getError(); + } + throw tlog_failed(); + } + + // Old-router pops advance from this remote log's durable acknowledgements. The current router's initial + // pop becomes visible only after the old prefix is consumed. Since the metric is durable known-committed + // progress and router pop positions are exclusive, require the handoff rather than its preceding version. + if (reply.get().v >= handoffVersion) { + TraceEvent("RemoteTLogPrefixDurable", dbgid) + .detail("TLogID", tlog->get().id()) + .detail("HandoffVersion", handoffVersion) + .detail("DurableKnownCommittedVersion", reply.get().v); + co_return; + } + co_await (delay(SERVER_KNOBS->METRIC_UPDATE_RATE) || changed); + } +} + +} // namespace + +Future LogSystem::onRemoteLogPrefixDurable() { + ASSERT(expectedLogSets > 0 && tLogs.size() == static_cast(expectedLogSets)); + if (!remoteLogPrefixComplete.isValid()) { + const Version handoffVersion = getMaxLocalStartVersion(tLogs); + std::vector> remotePrefixes; + if (!remoteLogPrefixRecovered()) { + for (const auto& logSet : tLogs) { + if (!logSet->isLocal && logSet->startVersion < handoffVersion) { + for (const auto& tlog : logSet->logServers) { + remotePrefixes.push_back(waitForRemoteTLogPrefixDurable(dbgid, tlog, handoffVersion)); + } + } + } + } + remoteLogPrefixComplete = waitForAll(remotePrefixes); + } + return remoteLogPrefixComplete; +} + void LogSystem::retireOldLogRoles(DBCoreState const& finalState) { ASSERT(finalState.recoveryCount == epoch); ASSERT(finalState.oldTLogData.empty()); ASSERT(expectedLogSets > 0 && finalState.tLogs.size() == static_cast(expectedLogSets)); ASSERT(recoveryCompleteWrittenToCoreState.get()); + ASSERT(remoteLogPrefixRecovered()); if (!oldLogRolesRetired) { oldLogRolesRetired = true; TraceEvent("RetireOldTLogRoles", dbgid) @@ -1130,7 +1219,7 @@ LogSystemConfig LogSystem::getLogSystemConfig() const { } } - // A storage-recovered state can still need old TLog locks for remote recruitment. + // A storage-recovered state can still need old TLog locks and history for remote recruitment/catch-up. // Keep those roles alive until the terminal recovery state is durable. if (!oldLogRolesRetired) { for (const auto& oldData : oldLogData) { @@ -2507,6 +2596,7 @@ Future LogSystem::newRemoteEpoch(Reference oldLogSystem, remoteRecoveryComplete = waitForAll(recoveryComplete); remoteTrackTLogRecovery = LogSystem::trackTLogRecoveryActor(allRemoteTLogServers, remoteRecoveredVersion); tLogs.push_back(logSet); + onRemoteLogPrefixDurable(); TraceEvent("RemoteLogRecruitment_CompletingRecovery").log(); } diff --git a/fdbserver/logsystem/LogSystemRecoveryTests.cpp b/fdbserver/logsystem/LogSystemRecoveryTests.cpp index 18069a42aca..a45c5d085a0 100644 --- a/fdbserver/logsystem/LogSystemRecoveryTests.cpp +++ b/fdbserver/logsystem/LogSystemRecoveryTests.cpp @@ -20,6 +20,8 @@ #include "fdbserver/logsystem/LogSystem.h" #include "fdbserver/logsystem/LogSystemConsumer.h" +#include "fdbserver/logsystem/LogSystemFactory.h" +#include "flow/CoroUtils.h" #include "flow/UnitTest.h" namespace { @@ -34,6 +36,159 @@ Reference makeSingleLogSet(const std::vector& tlogs, bool return logSet; } +Reference makeRetirementLogSet(const std::vector& tlogs, + bool isLocal, + int8_t locality, + Version startVersion) { + auto logSet = makeSingleLogSet(tlogs, isLocal); + logSet->locality = locality; + logSet->startVersion = startVersion; + logSet->tLogVersion = TLogVersion::V6; + logSet->tLogReplicationFactor = 1; + logSet->tLogPolicy = makeReference(); + for (const auto& tlog : tlogs) { + logSet->tLogLocalities.push_back(tlog.filteredLocality); + } + return logSet; +} + +Reference makeLaggingRetirementLogSystem(const std::vector& remoteLogs, + const TLogInterface& oldRouter, + const TLogInterface& currentRouter) { + constexpr LogEpoch epoch = 2; + LocalityData locality; + auto logSystem = makeReference(UID(), locality, epoch); + logSystem->logSystemType = LogSystemType::tagPartitioned; + logSystem->expectedLogSets = 2; + logSystem->oldestBackupEpoch = epoch; + logSystem->repopulateRegionAntiQuorum = 1; + logSystem->recoveryComplete = Void(); + logSystem->remoteRecovery = Void(); + logSystem->remoteRecoveryComplete = Never(); + logSystem->hasRemoteServers = true; + logSystem->logRouterTags = 1; + logSystem->tLogs.push_back(makeRetirementLogSet({ TLogInterface(locality) }, true, 0, 100)); + auto remote = makeRetirementLogSet(remoteLogs, false, 1, 60); + remote->logRouters.push_back( + makeReference>>(OptionalInterface(currentRouter))); + logSystem->tLogs.push_back(remote); + + OldLogData old; + old.epoch = epoch - 1; + old.epochBegin = 50; + old.epochEnd = 100; + old.recoverAt = 109; + old.logRouterTags = 1; + old.tLogs.push_back(makeRetirementLogSet({ TLogInterface(locality) }, true, 0, 50)); + auto oldRemote = makeRetirementLogSet({ TLogInterface(locality) }, false, 1, 60); + oldRemote->logRouters.push_back( + makeReference>>(OptionalInterface(oldRouter))); + old.tLogs.push_back(oldRemote); + logSystem->oldLogData.push_back(old); + // Make the ordinary generation-purge criteria eligible so the prefix barrier is the only retention gate. + logSystem->recoveredVersion->set(old.recoverAt + 1); + logSystem->remoteRecoveredVersion->set(old.recoverAt + 1); + return logSystem; +} + +DBCoreState makePendingRetirementCoreState(const Reference& logSystem) { + DBCoreState pendingState; + logSystem->toCoreState(pendingState); + pendingState.recoveryCount = logSystem->epoch; + ASSERT(!pendingState.oldTLogData.empty()); + ASSERT_EQ(pendingState.oldTLogData.size(), logSystem->oldLogData.size()); + ASSERT_EQ(pendingState.tLogs.size(), logSystem->expectedLogSets); + const auto oldGenerations = pendingState.oldTLogData; + logSystem->purgeOldRecoveredGenerationsCoreState(pendingState); + ASSERT(pendingState.oldTLogData == oldGenerations); + logSystem->coreStateWritten(pendingState); + ASSERT(logSystem->remoteLogsWrittenToCoreState); + ASSERT(!logSystem->recoveryCompleteWrittenToCoreState.get()); + return pendingState; +} + +DBCoreState makeRetirementCoreState(const Reference& logSystem) { + DBCoreState finalState; + logSystem->toCoreState(finalState); + finalState.recoveryCount = logSystem->epoch; + ASSERT(finalState.oldTLogData.empty()); + ASSERT_EQ(finalState.tLogs.size(), logSystem->expectedLogSets); + logSystem->coreStateWritten(finalState); + return finalState; +} + +void assertRetirementCoreStateError(const Reference& logSystem, int errorCode) { + Optional error; + try { + DBCoreState state; + logSystem->toCoreState(state); + } catch (Error& e) { + error = e; + } + ASSERT(error.present()); + ASSERT_EQ(error.get().code(), errorCode); +} + +TLogQueuingMetricsReply makeRetirementMetricsReply(Version version) { + TLogQueuingMetricsReply reply{}; + reply.localTime = now(); + reply.instanceID = 1; + reply.v = version; + return reply; +} + +Future serveRetirementMetrics(TLogInterface tlog, + Reference> version, + PromiseStream reports) { + while (true) { + TLogQueuingMetricsRequest req = co_await tlog.getQueuingMetrics.getFuture(); + const Version reported = version->get(); + req.reply.send(makeRetirementMetricsReply(reported)); + reports.send(reported); + } +} + +TLogPeekReply makeRetirementPeekReply(Version begin, Optional requestedEnd, Version end) { + TLogPeekReply reply; + reply.begin = begin; + reply.end = std::min(end, requestedEnd.orDefault(end)); + ASSERT_LT(begin, reply.end); + reply.popped = begin; + reply.maxKnownVersion = reply.end - 1; + reply.minKnownCommittedVersion = reply.end - 1; + return reply; +} + +Future serveRetirementRouter(TLogInterface router, Version begin, Version end, PromiseStream requests) { + while (true) { + co_await Choose() + .When(router.peekMessages.getFuture(), + [&](const TLogPeekRequest& req) { + ASSERT_GE(req.begin, begin); + req.reply.send(makeRetirementPeekReply(req.begin, req.end, end)); + requests.send(req.begin); + }) + .When(router.peekStreamMessages.getFuture(), + [&](const TLogPeekStreamRequest& req) { + ASSERT_GE(req.begin, begin); + req.reply.setByteLimit(req.limitBytes); + Future ready = req.reply.onReady(); + ASSERT(ready.isReady() && !ready.isError()); + req.reply.send(TLogPeekStreamReply(makeRetirementPeekReply(req.begin, req.end, end))); + req.reply.sendError(end_of_stream()); + requests.send(req.begin); + }) + .run(); + } +} + +Future advanceRetirementCursorTo(Reference cursor, Version end) { + while (cursor->version().version < end) { + co_await cursor->getMore(); + } + co_return; +} + std::tuple, bool> makeLogGroupResults( int replicationFactor, const std::vector>& perTLogUCV, @@ -71,6 +226,7 @@ TEST_CASE("/LogSystem/RetireOldLogRoles/FinalCoreState") { logSystem->remoteRecoveryComplete = Never(); logSystem->hasRemoteServers = true; logSystem->tLogs.push_back(makeSingleLogSet({ TLogInterface(locality) })); + logSystem->tLogs.back()->startVersion = 100; OldLogData old; old.epoch = epoch - 1; old.epochEnd = 100; @@ -89,15 +245,18 @@ TEST_CASE("/LogSystem/RetireOldLogRoles/FinalCoreState") { // Local recovery and backups can finish before the expected remote log set exists. logSystem->oldestBackupEpoch = epoch; logSystem->toCoreState(partialState); - ASSERT(partialState.oldTLogData.empty()); + ASSERT(logSystem->storageRecovered()); + ASSERT_EQ(partialState.oldTLogData.size(), 1); ASSERT_EQ(partialState.tLogs.size(), 1); logSystem->coreStateWritten(partialState); - ASSERT(logSystem->recoveryCompleteWrittenToCoreState.get()); + ASSERT(!logSystem->recoveryCompleteWrittenToCoreState.get()); ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); logSystem->tLogs.push_back(makeSingleLogSet({ TLogInterface(locality) }, false)); + logSystem->tLogs.back()->startVersion = 100; logSystem->remoteRecovery = Void(); ASSERT(!logSystem->remoteRecoveryComplete.isReady()); + co_await timeoutError(logSystem->onRemoteLogPrefixDurable(), 30.0); DBCoreState finalState; logSystem->toCoreState(finalState); finalState.recoveryCount = epoch; @@ -122,7 +281,190 @@ TEST_CASE("/LogSystem/RetireOldLogRoles/FinalCoreState") { logSystem->retireOldLogRoles(finalState); ASSERT(!unchanged.isReady()); ASSERT(logSystem->getLogSystemConfig() == expected); - return Void(); + co_return; +} + +TEST_CASE("/LogSystem/RetireOldLogRoles/RemotePrefixTrackerInstallation") { + constexpr double timeoutSeconds = 30.0; + LocalityData locality; + TLogInterface remote(locality); + auto logSystem = makeLaggingRetirementLogSystem({ remote }, TLogInterface(locality), TLogInterface(locality)); + Reference remoteSet = logSystem->tLogs.back(); + logSystem->tLogs.pop_back(); + Promise remoteRecruitment; + logSystem->remoteRecovery = remoteRecruitment.getFuture(); + + DBCoreState partialState; + logSystem->toCoreState(partialState); + partialState.recoveryCount = logSystem->epoch; + ASSERT_EQ(partialState.tLogs.size(), 1); + ASSERT_EQ(partialState.oldTLogData.size(), 1); + logSystem->coreStateWritten(partialState); + Future beforeInstallation = logSystem->onCoreStateChanged(); + ASSERT(!beforeInstallation.isReady()); + + // newRemoteEpoch installs the tracker before its remoteRecovery future becomes ready. + logSystem->tLogs.push_back(remoteSet); + Future prefixDurable = logSystem->onRemoteLogPrefixDurable(); + TLogQueuingMetricsRequest initialRequest = + co_await timeoutError(waitAndForward(remote.getQueuingMetrics.getFuture()), timeoutSeconds); + initialRequest.reply.send(makeRetirementMetricsReply(99)); + ASSERT(!prefixDurable.isReady()); + remoteRecruitment.send(Void()); + co_await timeoutError(beforeInstallation, timeoutSeconds); + + makePendingRetirementCoreState(logSystem); + Future afterInstallation = logSystem->onCoreStateChanged(); + ASSERT(!afterInstallation.isReady()); + TLogQueuingMetricsRequest caughtUpRequest = + co_await timeoutError(waitAndForward(remote.getQueuingMetrics.getFuture()), timeoutSeconds); + ASSERT(!afterInstallation.isReady()); + caughtUpRequest.reply.send(makeRetirementMetricsReply(100)); + co_await timeoutError(afterInstallation, timeoutSeconds); + co_await timeoutError(prefixDurable, timeoutSeconds); + ASSERT(!logSystem->remoteRecoveryComplete.isReady()); + const DBCoreState finalState = makeRetirementCoreState(logSystem); + logSystem->retireOldLogRoles(finalState); + co_return; +} + +TEST_CASE("/LogSystem/RetireOldLogRoles/RemotePrefixRemainsReadable") { + constexpr double timeoutSeconds = 30.0; + LocalityData locality; + TLogInterface remoteA(locality); + TLogInterface remoteB(locality); + TLogInterface oldRouter(locality); + TLogInterface currentRouter(locality); + auto versionA = makeReference>(100); + auto versionB = makeReference>(99); + PromiseStream reportsA; + PromiseStream reportsB; + PromiseStream oldRequests; + PromiseStream currentRequests; + Future mockActors = waitForAll(std::vector>{ + serveRetirementMetrics(remoteA, versionA, reportsA), + serveRetirementMetrics(remoteB, versionB, reportsB), + serveRetirementRouter(oldRouter, 60, 100, oldRequests), + serveRetirementRouter(currentRouter, 100, 110, currentRequests), + }); + auto logSystem = makeLaggingRetirementLogSystem({ remoteA, remoteB }, oldRouter, currentRouter); + const auto oldRoles = logSystem->getLogSystemConfig().oldTLogs; + ASSERT(logSystem->storageRecovered()); + const DBCoreState beforeTracking = makePendingRetirementCoreState(logSystem); + Future prefixDurable = logSystem->onRemoteLogPrefixDurable(); + Future coreStateChanged = logSystem->onCoreStateChanged(); + Future configChanged = logSystem->onLogSystemConfigChange(); + const Version reportedA = co_await timeoutError(waitAndForward(reportsA.getFuture()), timeoutSeconds); + const Version reportedB = co_await timeoutError(waitAndForward(reportsB.getFuture()), timeoutSeconds); + ASSERT_EQ(reportedA, 100); + ASSERT_EQ(reportedB, 99); + ASSERT(!prefixDurable.isReady()); + ASSERT(!coreStateChanged.isReady()); + ASSERT(!configChanged.isReady()); + const DBCoreState pendingState = makePendingRetirementCoreState(logSystem); + ASSERT(pendingState.oldTLogData == beforeTracking.oldTLogData); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + + // The real remote cursor must still reach the handoff through the old router. + auto oldConsumer = + makeLogSystemFromLogSystemConfig(UID(), locality, logSystem->getLogSystemConfig())->makeConsumer(); + auto oldCursor = oldConsumer->peek(UID(), 60, Optional(99), Tag(tagLocalityRemoteLog, 0), false); + co_await timeoutError(advanceRetirementCursorTo(oldCursor, 100) || mockActors, timeoutSeconds); + const Version oldBegin = co_await timeoutError(waitAndForward(oldRequests.getFuture()), timeoutSeconds); + ASSERT_EQ(oldBegin, 60); + ASSERT_EQ(oldCursor->version().version, 100); + ASSERT(!prefixDurable.isReady()); + + versionB->set(100); + co_await timeoutError(prefixDurable || mockActors, timeoutSeconds); + co_await timeoutError(coreStateChanged || mockActors, timeoutSeconds); + ASSERT(!configChanged.isReady()); + ASSERT(!logSystem->remoteRecoveryComplete.isReady()); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + DBCoreState purgeCandidate = pendingState; + logSystem->purgeOldRecoveredGenerationsCoreState(purgeCandidate); + ASSERT(purgeCandidate.oldTLogData.empty()); + const DBCoreState finalState = makeRetirementCoreState(logSystem); + logSystem->retireOldLogRoles(finalState); + ASSERT(configChanged.isReady() && !configChanged.isError()); + ASSERT(logSystem->getLogSystemConfig().oldTLogs.empty()); + + auto currentConsumer = + makeLogSystemFromLogSystemConfig(UID(), locality, logSystem->getLogSystemConfig())->makeConsumer(); + auto currentCursor = currentConsumer->peek(UID(), 100, Optional(109), Tag(tagLocalityRemoteLog, 0), false); + co_await timeoutError(advanceRetirementCursorTo(currentCursor, 110) || mockActors, timeoutSeconds); + const Version currentBegin = co_await timeoutError(waitAndForward(currentRequests.getFuture()), timeoutSeconds); + ASSERT_EQ(currentBegin, 100); + ASSERT_EQ(currentCursor->version().version, 110); + co_return; +} + +TEST_CASE("/LogSystem/RetireOldLogRoles/InterruptedRemotePrefixWait") { + constexpr double timeoutSeconds = 30.0; + LocalityData locality; + { + TLogInterface remote(locality); + auto logSystem = makeLaggingRetirementLogSystem({ remote }, TLogInterface(locality), TLogInterface(locality)); + const auto oldRoles = logSystem->getLogSystemConfig().oldTLogs; + makePendingRetirementCoreState(logSystem); + Future prefixDurable = logSystem->onRemoteLogPrefixDurable(); + Future coreStateChanged = logSystem->onCoreStateChanged(); + TLogQueuingMetricsRequest request = + co_await timeoutError(waitAndForward(remote.getQueuingMetrics.getFuture()), timeoutSeconds); + prefixDurable.cancel(); + ErrorOr result = co_await timeoutError(errorOr(prefixDurable), timeoutSeconds); + ASSERT(result.isError() && result.getError().code() == error_code_actor_cancelled); + ErrorOr changedResult = co_await timeoutError(errorOr(coreStateChanged), timeoutSeconds); + ASSERT(changedResult.isError() && changedResult.getError().code() == error_code_actor_cancelled); + request.reply.send(makeRetirementMetricsReply(100)); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + assertRetirementCoreStateError(logSystem, error_code_actor_cancelled); + } + { + TLogInterface remote(locality); + auto logSystem = makeLaggingRetirementLogSystem({ remote }, TLogInterface(locality), TLogInterface(locality)); + const auto oldRoles = logSystem->getLogSystemConfig().oldTLogs; + makePendingRetirementCoreState(logSystem); + Future prefixDurable = logSystem->onRemoteLogPrefixDurable(); + Future coreStateChanged = logSystem->onCoreStateChanged(); + TLogQueuingMetricsRequest request = + co_await timeoutError(waitAndForward(remote.getQueuingMetrics.getFuture()), timeoutSeconds); + request.reply.sendError(operation_failed()); + ErrorOr result = co_await timeoutError(errorOr(prefixDurable), timeoutSeconds); + ASSERT(result.isError() && result.getError().code() == error_code_tlog_failed); + ErrorOr changedResult = co_await timeoutError(errorOr(coreStateChanged), timeoutSeconds); + ASSERT(changedResult.isError() && changedResult.getError().code() == error_code_tlog_failed); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + assertRetirementCoreStateError(logSystem, error_code_tlog_failed); + } + + TLogInterface original(locality); + TLogInterface replacement(original.id(), original.getSharedTLogID(), locality); + auto logSystem = makeLaggingRetirementLogSystem({ original }, TLogInterface(locality), TLogInterface(locality)); + const auto oldRoles = logSystem->getLogSystemConfig().oldTLogs; + makePendingRetirementCoreState(logSystem); + Future prefixDurable = logSystem->onRemoteLogPrefixDurable(); + TLogQueuingMetricsRequest staleRequest = + co_await timeoutError(waitAndForward(original.getQueuingMetrics.getFuture()), timeoutSeconds); + logSystem->tLogs[1]->logServers[0]->setUnconditional(OptionalInterface(replacement)); + TLogQueuingMetricsRequest replacementRequest = + co_await timeoutError(waitAndForward(replacement.getQueuingMetrics.getFuture()), timeoutSeconds); + staleRequest.reply.send(makeRetirementMetricsReply(100)); + replacementRequest.reply.send(makeRetirementMetricsReply(99)); + ASSERT(!prefixDurable.isReady()); + makePendingRetirementCoreState(logSystem); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + + TLogQueuingMetricsRequest caughtUpRequest = + co_await timeoutError(waitAndForward(replacement.getQueuingMetrics.getFuture()), timeoutSeconds); + caughtUpRequest.reply.send(makeRetirementMetricsReply(100)); + co_await timeoutError(prefixDurable, timeoutSeconds); + ASSERT(!logSystem->remoteRecoveryComplete.isReady()); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + const DBCoreState finalState = makeRetirementCoreState(logSystem); + logSystem->retireOldLogRoles(finalState); + ASSERT(logSystem->getLogSystemConfig().oldTLogs.empty()); + co_return; } TEST_CASE("/LogSystem/GetPseudoPopTag/LogRouterWithoutMappedLocality") { diff --git a/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h b/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h index c4917187333..5fec2a1a17f 100644 --- a/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h +++ b/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h @@ -368,7 +368,11 @@ struct LogSystem : ReferenceCounted { // Convert LogSystem to DBCoreState and override input newState as return value void toCoreState(DBCoreState& newState) const; + // The storage/backup recovery policy can be satisfied before remote logs have copied their old prefix. + bool storageRecovered() const; bool remoteStorageRecovered() const; + // Requires every expected current log set. The returned future is shared by this recovery. + Future onRemoteLogPrefixDurable(); // Removes no-longer-needed older TLog generations from the outgoing core state. void purgeOldRecoveredGenerationsCoreState(DBCoreState&); @@ -553,6 +557,8 @@ struct LogSystem : ReferenceCounted { static std::vector getReadyNonError(std::vector> const& futures); private: + bool remoteLogPrefixRecovered() const; + Future remoteLogPrefixComplete; bool oldLogRolesRetired = false; }; From 79a8514179af80d21d0a30a4c23a00c86988e5f8 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Wed, 19 Aug 2026 22:09:33 -0700 Subject: [PATCH 005/170] Preserve old log history through remote prefix recovery --- .../clustercontroller/ClusterRecovery.cpp | 4 +- fdbserver/logsystem/LogSystem.cpp | 95 ++++- .../logsystem/LogSystemRecoveryTests.cpp | 340 ++++++++++++++++++ .../include/fdbserver/logsystem/LogSystem.h | 8 + 4 files changed, 443 insertions(+), 4 deletions(-) diff --git a/fdbserver/clustercontroller/ClusterRecovery.cpp b/fdbserver/clustercontroller/ClusterRecovery.cpp index fc30414f4b7..58a9b33e025 100644 --- a/fdbserver/clustercontroller/ClusterRecovery.cpp +++ b/fdbserver/clustercontroller/ClusterRecovery.cpp @@ -531,6 +531,8 @@ Future trackTlogRecovery(Reference self, bool allLogs = newState.tLogs.size() == configuration.expectedLogSets(!self->primaryDcId.empty() ? self->primaryDcId[0] : Optional()); + // Anti-quorum recovery must still permit removing a lost region while its old history remains durable. + bool storageRecovered = newState.oldTLogData.empty() || self->logSystem->storageRecovered(); bool finalUpdate = newState.oldTLogData.empty() && allLogs; TraceEvent("TrackTLogRecovery") .detail("FinalUpdate", finalUpdate) @@ -565,7 +567,7 @@ Future trackTlogRecovery(Reference self, self->dbgid) .detail("ActiveGenerations", 1) .trackLatest(self->clusterRecoveryGenerationsEventHolder->trackingKey); - } else if (newState.oldTLogData.empty() && self->recoveryState < RecoveryState::STORAGE_RECOVERED) { + } else if (storageRecovered && self->recoveryState < RecoveryState::STORAGE_RECOVERED) { self->recoveryState = RecoveryState::STORAGE_RECOVERED; TraceEvent(getRecoveryEventName(ClusterRecoveryEventType::CLUSTER_RECOVERY_STATE_EVENT_NAME).c_str(), self->dbgid) diff --git a/fdbserver/logsystem/LogSystem.cpp b/fdbserver/logsystem/LogSystem.cpp index 43ae3e14122..8d9cc98de68 100644 --- a/fdbserver/logsystem/LogSystem.cpp +++ b/fdbserver/logsystem/LogSystem.cpp @@ -491,6 +491,11 @@ Reference LogSystem::fromOldLogSystemConfig(UID const& dbgid, } void LogSystem::purgeOldRecoveredGenerationsCoreState(DBCoreState& newState) { + // Anti-quorum recovery may still need these generations after a controller or router failure. + // Do not let incremental purging bypass the remote-prefix durability barrier. + if (!remoteLogPrefixRecovered()) { + return; + } Version oldestGenerationRecoverAtVersion = std::min(recoveredVersion->get(), remoteRecoveredVersion->get()); TraceEvent("ToCoreStateOldestGenerationRecoverAtVersion") .detail("RecoveredVersion", recoveredVersion->get()) @@ -537,6 +542,8 @@ void LogSystem::toCoreState(DBCoreState& newState) const { if (remoteRecoveryComplete.isValid() && remoteRecoveryComplete.isError()) throw remoteRecoveryComplete.getError(); + if (remoteLogPrefixComplete.isValid() && remoteLogPrefixComplete.isError()) + throw remoteLogPrefixComplete.getError(); newState.tLogs.clear(); newState.logRouterTags = logRouterTags; @@ -554,9 +561,7 @@ void LogSystem::toCoreState(DBCoreState& newState) const { } newState.oldTLogData.clear(); - if (!recoveryComplete.isValid() || !recoveryComplete.isReady() || - (repopulateRegionAntiQuorum == 0 && (!remoteRecoveryComplete.isValid() || !remoteRecoveryComplete.isReady())) || - epoch != oldestBackupEpoch) { + if (!storageRecovered() || !remoteLogPrefixRecovered()) { for (const auto& oldData : oldLogData) { newState.oldTLogData.push_back(toOldTLogCoreData(oldData)); TraceEvent("BWToCore") @@ -572,10 +577,23 @@ void LogSystem::toCoreState(DBCoreState& newState) const { newState.logSystemType = logSystemType; } +bool LogSystem::storageRecovered() const { + return recoveryComplete.isValid() && recoveryComplete.isReady() && !recoveryComplete.isError() && + (repopulateRegionAntiQuorum != 0 || (remoteStorageRecovered() && !remoteRecoveryComplete.isError())) && + epoch == oldestBackupEpoch; +} + bool LogSystem::remoteStorageRecovered() const { return remoteRecoveryComplete.isValid() && remoteRecoveryComplete.isReady(); } +bool LogSystem::remoteLogPrefixRecovered() const { + return repopulateRegionAntiQuorum == 0 || !hasRemoteServers || oldLogData.empty() || + (remoteStorageRecovered() && !remoteRecoveryComplete.isError()) || + (remoteLogPrefixComplete.isValid() && remoteLogPrefixComplete.isReady() && + !remoteLogPrefixComplete.isError()); +} + Future LogSystem::onCoreStateChanged() const { std::vector> changes; changes.push_back(Never()); @@ -588,6 +606,9 @@ Future LogSystem::onCoreStateChanged() const { if (remoteRecoveryComplete.isValid() && !remoteRecoveryComplete.isReady()) { changes.push_back(remoteRecoveryComplete); } + if (remoteLogPrefixComplete.isValid() && !remoteLogPrefixComplete.isReady()) { + changes.push_back(remoteLogPrefixComplete); + } changes.push_back(backupWorkerChanged.onTrigger()); // changes to oldestBackupEpoch changes.push_back(recoveredVersion->onChange()); changes.push_back(remoteRecoveredVersion->onChange()); @@ -596,6 +617,7 @@ Future LogSystem::onCoreStateChanged() const { void LogSystem::coreStateWritten(DBCoreState const& newState) { if (newState.oldTLogData.empty()) { + ASSERT(remoteLogPrefixRecovered()); recoveryCompleteWrittenToCoreState.set(true); } for (auto& t : newState.tLogs) { @@ -607,6 +629,72 @@ void LogSystem::coreStateWritten(DBCoreState const& newState) { } } +namespace { + +Future waitForRemoteTLogPrefixDurable(UID dbgid, + Reference>> tlog, + Version handoffVersion) { + TraceEvent("WaitForRemoteTLogPrefixDurable", dbgid) + .detail("TLogID", tlog->get().id()) + .detail("HandoffVersion", handoffVersion); + while (true) { + Future changed = tlog->onChange(); + if (!tlog->get().present()) { + co_await changed; + continue; + } + + auto result = co_await race(tlog->get().interf().getQueuingMetrics.getReplyUnlessFailedFor( + TLogQueuingMetricsRequest(), + SERVER_KNOBS->TLOG_TIMEOUT, + SERVER_KNOBS->MASTER_FAILURE_SLOPE_DURING_RECOVERY), + changed); + if (result.index() != 0 || changed.isReady()) { + continue; + } + ErrorOr reply = std::get<0>(result); + if (reply.isError()) { + if (reply.getError().code() == error_code_actor_cancelled) { + throw reply.getError(); + } + throw tlog_failed(); + } + + // Old-router pops advance from this remote log's durable acknowledgements. The current router's initial + // pop becomes visible only after the old prefix is consumed. Since the metric is durable known-committed + // progress and router pop positions are exclusive, require the handoff rather than its preceding version. + if (reply.get().v >= handoffVersion) { + TraceEvent("RemoteTLogPrefixDurable", dbgid) + .detail("TLogID", tlog->get().id()) + .detail("HandoffVersion", handoffVersion) + .detail("DurableKnownCommittedVersion", reply.get().v); + co_return; + } + co_await (delay(SERVER_KNOBS->METRIC_UPDATE_RATE) || changed); + } +} + +} // namespace + +Future LogSystem::onRemoteLogPrefixDurable() { + ASSERT(expectedLogSets > 0 && tLogs.size() == static_cast(expectedLogSets)); + if (!remoteLogPrefixComplete.isValid()) { + const Version handoffVersion = getMaxLocalStartVersion(tLogs); + std::vector> remotePrefixes; + if (!remoteLogPrefixRecovered()) { + for (const auto& logSet : tLogs) { + if (!logSet->isLocal && logSet->startVersion < handoffVersion) { + for (const auto& tlog : logSet->logServers) { + remotePrefixes.push_back(waitForRemoteTLogPrefixDurable(dbgid, tlog, handoffVersion)); + } + } + } + } + remoteLogPrefixComplete = waitForAll(remotePrefixes); + } + return remoteLogPrefixComplete; +} + Future LogSystem::onError() const { // Never returns normally, but throws an error if the subsystem stops working while (true) { @@ -2493,6 +2581,7 @@ Future LogSystem::newRemoteEpoch(Reference oldLogSystem, remoteRecoveryComplete = waitForAll(recoveryComplete); remoteTrackTLogRecovery = LogSystem::trackTLogRecoveryActor(allRemoteTLogServers, remoteRecoveredVersion); tLogs.push_back(logSet); + onRemoteLogPrefixDurable(); TraceEvent("RemoteLogRecruitment_CompletingRecovery").log(); } diff --git a/fdbserver/logsystem/LogSystemRecoveryTests.cpp b/fdbserver/logsystem/LogSystemRecoveryTests.cpp index 70baf43d283..5ce4c6d16ec 100644 --- a/fdbserver/logsystem/LogSystemRecoveryTests.cpp +++ b/fdbserver/logsystem/LogSystemRecoveryTests.cpp @@ -20,6 +20,8 @@ #include "fdbserver/logsystem/LogSystem.h" #include "fdbserver/logsystem/LogSystemConsumer.h" +#include "fdbserver/logsystem/LogSystemFactory.h" +#include "flow/CoroUtils.h" #include "flow/UnitTest.h" namespace { @@ -34,6 +36,164 @@ Reference makeSingleLogSet(const std::vector& tlogs, bool return logSet; } +Reference makeRemotePrefixLogSet(const std::vector& tlogs, + bool isLocal, + int8_t locality, + Version startVersion) { + auto logSet = makeSingleLogSet(tlogs, isLocal); + logSet->locality = locality; + logSet->startVersion = startVersion; + logSet->tLogVersion = TLogVersion::V6; + logSet->tLogReplicationFactor = 1; + logSet->tLogPolicy = makeReference(); + for (const auto& tlog : tlogs) { + logSet->tLogLocalities.push_back(tlog.filteredLocality); + } + return logSet; +} + +Reference makeLaggingRemoteLogSystem(const std::vector& remoteLogs, + const TLogInterface& oldRouter, + const TLogInterface& currentRouter) { + constexpr LogEpoch epoch = 2; + LocalityData locality; + auto logSystem = makeReference(UID(), locality, epoch); + logSystem->logSystemType = LogSystemType::tagPartitioned; + logSystem->expectedLogSets = 2; + logSystem->oldestBackupEpoch = epoch; + logSystem->repopulateRegionAntiQuorum = 1; + logSystem->recoveryComplete = Void(); + logSystem->remoteRecovery = Void(); + logSystem->remoteRecoveryComplete = Never(); + logSystem->hasRemoteServers = true; + logSystem->logRouterTags = 1; + logSystem->tLogs.push_back(makeRemotePrefixLogSet({ TLogInterface(locality) }, true, 0, 100)); + auto remote = makeRemotePrefixLogSet(remoteLogs, false, 1, 60); + remote->logRouters.push_back( + makeReference>>(OptionalInterface(currentRouter))); + logSystem->tLogs.push_back(remote); + + OldLogData old; + old.epoch = epoch - 1; + old.epochBegin = 50; + old.epochEnd = 100; + old.recoverAt = 109; + old.logRouterTags = 1; + old.tLogs.push_back(makeRemotePrefixLogSet({ TLogInterface(locality) }, true, 0, 50)); + auto oldRemote = makeRemotePrefixLogSet({ TLogInterface(locality) }, false, 1, 60); + oldRemote->logRouters.push_back( + makeReference>>(OptionalInterface(oldRouter))); + old.tLogs.push_back(oldRemote); + logSystem->oldLogData.push_back(old); + // Make the ordinary generation-purge criteria eligible so the prefix barrier is the only retention gate. + logSystem->recoveredVersion->set(old.recoverAt + 1); + logSystem->remoteRecoveredVersion->set(old.recoverAt + 1); + return logSystem; +} + +DBCoreState makePendingRemotePrefixCoreState(const Reference& logSystem) { + DBCoreState pendingState; + logSystem->toCoreState(pendingState); + pendingState.recoveryCount = logSystem->epoch; + ASSERT(!pendingState.oldTLogData.empty()); + ASSERT_EQ(pendingState.oldTLogData.size(), logSystem->oldLogData.size()); + ASSERT_EQ(pendingState.tLogs.size(), logSystem->expectedLogSets); + const auto oldGenerations = pendingState.oldTLogData; + logSystem->purgeOldRecoveredGenerationsCoreState(pendingState); + ASSERT(pendingState.oldTLogData == oldGenerations); + logSystem->coreStateWritten(pendingState); + ASSERT(logSystem->remoteLogsWrittenToCoreState); + ASSERT(!logSystem->recoveryCompleteWrittenToCoreState.get()); + return pendingState; +} + +DBCoreState makeRecoveredRemotePrefixCoreState(const Reference& logSystem) { + DBCoreState finalState; + logSystem->toCoreState(finalState); + finalState.recoveryCount = logSystem->epoch; + ASSERT(finalState.oldTLogData.empty()); + ASSERT_EQ(finalState.tLogs.size(), logSystem->expectedLogSets); + logSystem->coreStateWritten(finalState); + ASSERT(logSystem->recoveryCompleteWrittenToCoreState.get()); + return finalState; +} + +void assertRemotePrefixCoreStateError(const Reference& logSystem, int errorCode) { + Optional error; + try { + DBCoreState state; + logSystem->toCoreState(state); + } catch (Error& e) { + error = e; + } + ASSERT(error.present()); + ASSERT_EQ(error.get().code(), errorCode); + ASSERT(!logSystem->recoveryCompleteWrittenToCoreState.get()); +} + +TLogQueuingMetricsReply makeRemotePrefixMetricsReply(Version version) { + TLogQueuingMetricsReply reply{}; + reply.localTime = now(); + reply.instanceID = 1; + reply.v = version; + return reply; +} + +Future serveRemotePrefixMetrics(TLogInterface tlog, + Reference> version, + PromiseStream reports) { + while (true) { + TLogQueuingMetricsRequest req = co_await tlog.getQueuingMetrics.getFuture(); + const Version reported = version->get(); + req.reply.send(makeRemotePrefixMetricsReply(reported)); + reports.send(reported); + } +} + +TLogPeekReply makeRemotePrefixPeekReply(Version begin, Optional requestedEnd, Version end) { + TLogPeekReply reply; + reply.begin = begin; + reply.end = std::min(end, requestedEnd.orDefault(end)); + ASSERT_LT(begin, reply.end); + reply.popped = begin; + reply.maxKnownVersion = reply.end - 1; + reply.minKnownCommittedVersion = reply.end - 1; + return reply; +} + +Future serveRemotePrefixRouter(TLogInterface router, + Version begin, + Version end, + PromiseStream requests) { + while (true) { + co_await Choose() + .When(router.peekMessages.getFuture(), + [&](const TLogPeekRequest& req) { + ASSERT_GE(req.begin, begin); + req.reply.send(makeRemotePrefixPeekReply(req.begin, req.end, end)); + requests.send(req.begin); + }) + .When(router.peekStreamMessages.getFuture(), + [&](const TLogPeekStreamRequest& req) { + ASSERT_GE(req.begin, begin); + req.reply.setByteLimit(req.limitBytes); + Future ready = req.reply.onReady(); + ASSERT(ready.isReady() && !ready.isError()); + req.reply.send(TLogPeekStreamReply(makeRemotePrefixPeekReply(req.begin, req.end, end))); + req.reply.sendError(end_of_stream()); + requests.send(req.begin); + }) + .run(); + } +} + +Future advanceRemotePrefixCursorTo(Reference cursor, Version end) { + while (cursor->version().version < end) { + co_await cursor->getMore(); + } + co_return; +} + std::tuple, bool> makeLogGroupResults( int replicationFactor, const std::vector>& perTLogUCV, @@ -58,6 +218,186 @@ std::tuple, bool> makeLogGroupResults( void forceLinkLogSystemRecoveryTests() {} +TEST_CASE("/LogSystem/RemoteLogPrefix/TrackerInstallation") { + constexpr double timeoutSeconds = 30.0; + LocalityData locality; + TLogInterface remote(locality); + auto logSystem = makeLaggingRemoteLogSystem({ remote }, TLogInterface(locality), TLogInterface(locality)); + Reference remoteSet = logSystem->tLogs.back(); + logSystem->tLogs.pop_back(); + Promise remoteRecruitment; + logSystem->remoteRecovery = remoteRecruitment.getFuture(); + + DBCoreState partialState; + logSystem->toCoreState(partialState); + partialState.recoveryCount = logSystem->epoch; + ASSERT_EQ(partialState.tLogs.size(), 1); + ASSERT_EQ(partialState.oldTLogData.size(), 1); + logSystem->coreStateWritten(partialState); + Future beforeInstallation = logSystem->onCoreStateChanged(); + ASSERT(!beforeInstallation.isReady()); + + // The prefix tracker must be installed before remote recruitment reports completion. + logSystem->tLogs.push_back(remoteSet); + Future prefixDurable = logSystem->onRemoteLogPrefixDurable(); + TLogQueuingMetricsRequest initialRequest = + co_await timeoutError(waitAndForward(remote.getQueuingMetrics.getFuture()), timeoutSeconds); + initialRequest.reply.send(makeRemotePrefixMetricsReply(99)); + ASSERT(!prefixDurable.isReady()); + remoteRecruitment.send(Void()); + co_await timeoutError(beforeInstallation, timeoutSeconds); + + makePendingRemotePrefixCoreState(logSystem); + Future afterInstallation = logSystem->onCoreStateChanged(); + ASSERT(!afterInstallation.isReady()); + TLogQueuingMetricsRequest caughtUpRequest = + co_await timeoutError(waitAndForward(remote.getQueuingMetrics.getFuture()), timeoutSeconds); + ASSERT(!afterInstallation.isReady()); + caughtUpRequest.reply.send(makeRemotePrefixMetricsReply(100)); + co_await timeoutError(afterInstallation, timeoutSeconds); + co_await timeoutError(prefixDurable, timeoutSeconds); + ASSERT(!logSystem->remoteRecoveryComplete.isReady()); + makeRecoveredRemotePrefixCoreState(logSystem); + co_return; +} + +TEST_CASE("/LogSystem/RemoteLogPrefix/RemainsReadable") { + constexpr double timeoutSeconds = 30.0; + LocalityData locality; + TLogInterface remoteA(locality); + TLogInterface remoteB(locality); + TLogInterface oldRouter(locality); + TLogInterface currentRouter(locality); + auto versionA = makeReference>(100); + auto versionB = makeReference>(99); + PromiseStream reportsA; + PromiseStream reportsB; + PromiseStream oldRequests; + PromiseStream currentRequests; + Future mockActors = waitForAll(std::vector>{ + serveRemotePrefixMetrics(remoteA, versionA, reportsA), + serveRemotePrefixMetrics(remoteB, versionB, reportsB), + serveRemotePrefixRouter(oldRouter, 60, 100, oldRequests), + serveRemotePrefixRouter(currentRouter, 100, 110, currentRequests), + }); + auto logSystem = makeLaggingRemoteLogSystem({ remoteA, remoteB }, oldRouter, currentRouter); + const auto oldRoles = logSystem->getLogSystemConfig().oldTLogs; + ASSERT(logSystem->storageRecovered()); + const DBCoreState beforeTracking = makePendingRemotePrefixCoreState(logSystem); + Future prefixDurable = logSystem->onRemoteLogPrefixDurable(); + Future coreStateChanged = logSystem->onCoreStateChanged(); + Future configChanged = logSystem->onLogSystemConfigChange(); + const Version reportedA = co_await timeoutError(waitAndForward(reportsA.getFuture()), timeoutSeconds); + const Version reportedB = co_await timeoutError(waitAndForward(reportsB.getFuture()), timeoutSeconds); + ASSERT_EQ(reportedA, 100); + ASSERT_EQ(reportedB, 99); + ASSERT(!prefixDurable.isReady()); + ASSERT(!coreStateChanged.isReady()); + ASSERT(!configChanged.isReady()); + const DBCoreState pendingState = makePendingRemotePrefixCoreState(logSystem); + ASSERT(pendingState.oldTLogData == beforeTracking.oldTLogData); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + + // The real remote cursor must still reach the handoff through the old router. + auto oldConsumer = + makeLogSystemFromLogSystemConfig(UID(), locality, logSystem->getLogSystemConfig())->makeConsumer(); + auto oldCursor = oldConsumer->peek(UID(), 60, Optional(99), Tag(tagLocalityRemoteLog, 0), false); + co_await timeoutError(advanceRemotePrefixCursorTo(oldCursor, 100) || mockActors, timeoutSeconds); + const Version oldBegin = co_await timeoutError(waitAndForward(oldRequests.getFuture()), timeoutSeconds); + ASSERT_EQ(oldBegin, 60); + ASSERT_EQ(oldCursor->version().version, 100); + ASSERT(!prefixDurable.isReady()); + + versionB->set(100); + co_await timeoutError(prefixDurable || mockActors, timeoutSeconds); + co_await timeoutError(coreStateChanged || mockActors, timeoutSeconds); + ASSERT(!configChanged.isReady()); + ASSERT(!logSystem->remoteRecoveryComplete.isReady()); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + DBCoreState purgeCandidate = pendingState; + logSystem->purgeOldRecoveredGenerationsCoreState(purgeCandidate); + ASSERT(purgeCandidate.oldTLogData.empty()); + makeRecoveredRemotePrefixCoreState(logSystem); + ASSERT(!configChanged.isReady()); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + + auto currentConsumer = + makeLogSystemFromLogSystemConfig(UID(), locality, logSystem->getLogSystemConfig())->makeConsumer(); + auto currentCursor = currentConsumer->peek(UID(), 100, Optional(109), Tag(tagLocalityRemoteLog, 0), false); + co_await timeoutError(advanceRemotePrefixCursorTo(currentCursor, 110) || mockActors, timeoutSeconds); + const Version currentBegin = co_await timeoutError(waitAndForward(currentRequests.getFuture()), timeoutSeconds); + ASSERT_EQ(currentBegin, 100); + ASSERT_EQ(currentCursor->version().version, 110); + co_return; +} + +TEST_CASE("/LogSystem/RemoteLogPrefix/InterruptedWait") { + constexpr double timeoutSeconds = 30.0; + LocalityData locality; + { + TLogInterface remote(locality); + auto logSystem = makeLaggingRemoteLogSystem({ remote }, TLogInterface(locality), TLogInterface(locality)); + const auto oldRoles = logSystem->getLogSystemConfig().oldTLogs; + makePendingRemotePrefixCoreState(logSystem); + Future prefixDurable = logSystem->onRemoteLogPrefixDurable(); + Future coreStateChanged = logSystem->onCoreStateChanged(); + TLogQueuingMetricsRequest request = + co_await timeoutError(waitAndForward(remote.getQueuingMetrics.getFuture()), timeoutSeconds); + prefixDurable.cancel(); + ErrorOr result = co_await timeoutError(errorOr(prefixDurable), timeoutSeconds); + ASSERT(result.isError() && result.getError().code() == error_code_actor_cancelled); + ErrorOr changedResult = co_await timeoutError(errorOr(coreStateChanged), timeoutSeconds); + ASSERT(changedResult.isError() && changedResult.getError().code() == error_code_actor_cancelled); + request.reply.send(makeRemotePrefixMetricsReply(100)); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + assertRemotePrefixCoreStateError(logSystem, error_code_actor_cancelled); + } + { + TLogInterface remote(locality); + auto logSystem = makeLaggingRemoteLogSystem({ remote }, TLogInterface(locality), TLogInterface(locality)); + const auto oldRoles = logSystem->getLogSystemConfig().oldTLogs; + makePendingRemotePrefixCoreState(logSystem); + Future prefixDurable = logSystem->onRemoteLogPrefixDurable(); + Future coreStateChanged = logSystem->onCoreStateChanged(); + TLogQueuingMetricsRequest request = + co_await timeoutError(waitAndForward(remote.getQueuingMetrics.getFuture()), timeoutSeconds); + request.reply.sendError(operation_failed()); + ErrorOr result = co_await timeoutError(errorOr(prefixDurable), timeoutSeconds); + ASSERT(result.isError() && result.getError().code() == error_code_tlog_failed); + ErrorOr changedResult = co_await timeoutError(errorOr(coreStateChanged), timeoutSeconds); + ASSERT(changedResult.isError() && changedResult.getError().code() == error_code_tlog_failed); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + assertRemotePrefixCoreStateError(logSystem, error_code_tlog_failed); + } + + TLogInterface original(locality); + TLogInterface replacement(original.id(), original.getSharedTLogID(), locality); + auto logSystem = makeLaggingRemoteLogSystem({ original }, TLogInterface(locality), TLogInterface(locality)); + const auto oldRoles = logSystem->getLogSystemConfig().oldTLogs; + makePendingRemotePrefixCoreState(logSystem); + Future prefixDurable = logSystem->onRemoteLogPrefixDurable(); + TLogQueuingMetricsRequest staleRequest = + co_await timeoutError(waitAndForward(original.getQueuingMetrics.getFuture()), timeoutSeconds); + logSystem->tLogs[1]->logServers[0]->setUnconditional(OptionalInterface(replacement)); + TLogQueuingMetricsRequest replacementRequest = + co_await timeoutError(waitAndForward(replacement.getQueuingMetrics.getFuture()), timeoutSeconds); + staleRequest.reply.send(makeRemotePrefixMetricsReply(100)); + replacementRequest.reply.send(makeRemotePrefixMetricsReply(99)); + ASSERT(!prefixDurable.isReady()); + makePendingRemotePrefixCoreState(logSystem); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + + TLogQueuingMetricsRequest caughtUpRequest = + co_await timeoutError(waitAndForward(replacement.getQueuingMetrics.getFuture()), timeoutSeconds); + caughtUpRequest.reply.send(makeRemotePrefixMetricsReply(100)); + co_await timeoutError(prefixDurable, timeoutSeconds); + ASSERT(!logSystem->remoteRecoveryComplete.isReady()); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + makeRecoveredRemotePrefixCoreState(logSystem); + ASSERT(logSystem->getLogSystemConfig().oldTLogs == oldRoles); + co_return; +} + TEST_CASE("/LogSystem/GetPseudoPopTag/LogRouterWithoutMappedLocality") { LocalityData locality; auto logSystem = makeReference(UID(), locality, LogEpoch(1)); diff --git a/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h b/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h index 01f443cdcf6..4f580446930 100644 --- a/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h +++ b/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h @@ -368,7 +368,11 @@ struct LogSystem : ReferenceCounted { // Convert LogSystem to DBCoreState and override input newState as return value void toCoreState(DBCoreState& newState) const; + // The storage/backup recovery policy can be satisfied before remote logs have copied their old prefix. + bool storageRecovered() const; bool remoteStorageRecovered() const; + // Requires every expected current log set. The returned future is shared by this recovery. + Future onRemoteLogPrefixDurable(); // Removes no-longer-needed older TLog generations from the outgoing core state. void purgeOldRecoveredGenerationsCoreState(DBCoreState&); @@ -547,6 +551,10 @@ struct LogSystem : ReferenceCounted { static Future lockTLog(UID myID, Reference>> tlog); template static std::vector getReadyNonError(std::vector> const& futures); + +private: + bool remoteLogPrefixRecovered() const; + Future remoteLogPrefixComplete; }; // Recovery version calculation for version vector unicast From 45a475da763bad0dca6c7efc33d455c7ed230a5c Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Thu, 20 Aug 2026 22:23:41 -0700 Subject: [PATCH 006/170] Add Python bindings for native CDC --- bindings/python/CMakeLists.txt | 14 + bindings/python/fdb/__init__.py | 7 + bindings/python/fdb/impl.py | 403 ++++++++++++++++++++ bindings/python/tests/native_cdc_tests.py | 408 +++++++++++++++++++++ design/cdc.md | 17 +- documentation/sphinx/source/api-python.rst | 195 +++++++++- 6 files changed, 1035 insertions(+), 9 deletions(-) create mode 100644 bindings/python/tests/native_cdc_tests.py diff --git a/bindings/python/CMakeLists.txt b/bindings/python/CMakeLists.txt index ab1708f6291..083bde73a62 100644 --- a/bindings/python/CMakeLists.txt +++ b/bindings/python/CMakeLists.txt @@ -126,6 +126,20 @@ if(NOT OPEN_FOR_IDE AND NOT USE_SANITIZER) DISABLE_LOG_DUMP ) + add_fdbclient_test( + NAME python_native_cdc_tests + COMMAND python ${CMAKE_CURRENT_SOURCE_DIR}/tests/native_cdc_tests.py + --cluster-file @CLUSTER_FILE@ --verbose + ) + set_property(TEST python_native_cdc_tests APPEND PROPERTY ENVIRONMENT + "FDB_KNOB_enable_native_cdc=true") + + add_fdbclient_test( + NAME python_native_cdc_legacy_api_tests + COMMAND python ${CMAKE_CURRENT_SOURCE_DIR}/tests/native_cdc_tests.py + --cluster-file @CLUSTER_FILE@ --api-version 740 --verbose + ) + # FIXME Windows support set(PYPKG_TEST_DIR "${CMAKE_BINARY_DIR}/pypkg-test-venv") set(PYPKG_TEST_PY3 "${PYPKG_TEST_DIR}/bin/python3") diff --git a/bindings/python/fdb/__init__.py b/bindings/python/fdb/__init__.py index b4dea358a43..64566ed5a96 100644 --- a/bindings/python/fdb/__init__.py +++ b/bindings/python/fdb/__init__.py @@ -109,6 +109,13 @@ def api_version(ver): "transactional", "options", "StreamingMode", + "CdcMutationType", + "CdcCursor", + "CdcStreamInfo", + "CdcMutation", + "CdcVersionedMutations", + "CdcConsumeResult", + "CdcConsumer", ) _add_symbols(fdb.impl, list) diff --git a/bindings/python/fdb/impl.py b/bindings/python/fdb/impl.py index 6b7a71f4c1a..8d009b72a21 100644 --- a/bindings/python/fdb/impl.py +++ b/bindings/python/fdb/impl.py @@ -23,9 +23,11 @@ import atexit import ctypes import ctypes.util +import enum import functools import inspect import multiprocessing +import operator import os import platform import struct @@ -33,6 +35,7 @@ import threading import traceback import weakref +from typing import NamedTuple, Tuple import fdb from fdb.tuple import int2byte @@ -42,6 +45,7 @@ _network_thread = None _network_thread_reentrant_lock = threading.RLock() +_cdc_c_api_initialized = False _open_file = open @@ -891,6 +895,81 @@ def wait(self): return list(strings[0 : count.value]) +class FutureCdcStreamInfoArray(Future): + def wait(self): + self.block_until_ready() + streams = ctypes.POINTER(CdcStreamInfoStruct)() + count = ctypes.c_int() + self.capi.fdb_future_get_cdc_stream_info_array( + self.fpointer, ctypes.byref(streams), ctypes.byref(count) + ) + return [ + CdcStreamInfo( + ctypes.string_at(stream.name.key, stream.name.key_length), + stream.stream_id, + ctypes.string_at( + stream.key_range.begin_key, stream.key_range.begin_key_length + ), + ctypes.string_at( + stream.key_range.end_key, stream.key_range.end_key_length + ), + stream.min_version, + ) + for stream in streams[: count.value] + ] + + +class FutureCdcConsumer(Future): + def __init__(self, fpointer): + super().__init__(fpointer) + self._consumer = None + self._lock = threading.Lock() + + def wait(self): + self.block_until_ready() + with self._lock: + if self._consumer is None: + consumer = CdcConsumer() + self.capi.fdb_future_get_cdc_consumer( + self.fpointer, ctypes.byref(consumer._pointer) + ) + # Each C getter call transfers a reference. Keep one Python + # owner even when callbacks or callers retrieve the result again. + self._consumer = consumer + return self._consumer + + +class FutureCdcConsumeResult(Future): + def wait(self): + self.block_until_ready() + groups = ctypes.POINTER(CdcVersionedMutationsStruct)() + count = ctypes.c_int() + last_consumed_version = ctypes.c_int64() + self.capi.fdb_future_get_cdc_versioned_mutations( + self.fpointer, + ctypes.byref(groups), + ctypes.byref(count), + ctypes.byref(last_consumed_version), + ) + return CdcConsumeResult( + tuple( + CdcVersionedMutations( + group.version, + tuple( + CdcMutation( + mutation.type, + ctypes.string_at(mutation.param1, mutation.param1_length), + ctypes.string_at(mutation.param2, mutation.param2_length), + ) + for mutation in group.mutations[: group.mutation_count] + ), + ) + for group in groups[: count.value] + ), + last_consumed_version.value, + ) + + class replaceable_property(object): def __get__(self, obj, cls=None): return self.method(obj) @@ -1342,10 +1421,201 @@ def create_transaction(self): def get_client_status(self): return Key(self.capi.fdb_database_get_client_status(self.dpointer)) + def register_cdc_stream(self, name, begin_key, end_key): + """Register a named CDC range and return a future containing its stream ID.""" + _require_cdc_api_version() + name = keyToBytes(name) + begin_key = keyToBytes(begin_key) + end_key = keyToBytes(end_key) + return FutureUInt64( + self.capi.fdb_database_register_cdc_stream( + self.dpointer, + name, + len(name), + begin_key, + len(begin_key), + end_key, + len(end_key), + ) + ) + + def remove_cdc_stream(self, name): + """Remove a stream and relinquish its unread history; return a void future.""" + _require_cdc_api_version() + name = keyToBytes(name) + return FutureVoid( + self.capi.fdb_database_remove_cdc_stream(self.dpointer, name, len(name)) + ) + + def list_cdc_streams(self): + """Return a future containing a list of CdcStreamInfo records.""" + _require_cdc_api_version() + return FutureCdcStreamInfoArray( + self.capi.fdb_database_list_cdc_streams(self.dpointer) + ) + + def create_cdc_consumer(self, name): + """Return a future containing a consumer for an existing stream name.""" + _require_cdc_api_version() + name = keyToBytes(name) + return FutureCdcConsumer( + self.capi.fdb_database_create_cdc_consumer(self.dpointer, name, len(name)) + ) + + def resume_cdc_consumer(self, cursor): + """Return a consumer future from a durably checkpointed CdcCursor. + + The native client validates the stream and position on consume or + acknowledge, not when constructing the handle. + """ + _require_cdc_api_version() + stream_id = operator.index(cursor.stream_id) + version = operator.index(cursor.last_consumed_version) + if not 0 <= stream_id < 2**64: + raise ValueError("stream_id must fit in an unsigned 64-bit integer") + if not -(2**63) <= version < 2**63: + raise ValueError( + "last_consumed_version must fit in a signed 64-bit integer" + ) + return FutureCdcConsumer( + self.capi.fdb_database_resume_cdc_consumer( + self.dpointer, stream_id, version + ) + ) + fill_operations() +class CdcMutationType(enum.IntEnum): + """Known raw CDC mutation codes; a reply may also contain unknown codes.""" + + SET_VALUE = 0 + CLEAR_RANGE = 1 + ADD = 2 + AND = 6 + OR = 7 + XOR = 8 + APPEND_IF_FITS = 9 + MAX = 12 + MIN = 13 + SET_VERSIONSTAMPED_KEY = 14 + SET_VERSIONSTAMPED_VALUE = 15 + BYTE_MIN = 16 + BYTE_MAX = 17 + MIN_V2 = 18 + AND_V2 = 19 + COMPARE_AND_CLEAR = 20 + + +class CdcCursor(NamedTuple): + """A stream's stable ID and delivered position, suitable for checkpointing.""" + + stream_id: int + last_consumed_version: int + + +class CdcStreamInfo(NamedTuple): + """A registered stream and its durable minimum required version.""" + + name: bytes + stream_id: int + begin_key: bytes + end_key: bytes + min_version: int + + +class CdcMutation(NamedTuple): + """One raw mutation with Python-owned parameter bytes.""" + + type: int + param1: bytes + param2: bytes + + +class CdcVersionedMutations(NamedTuple): + """A complete group of mutations sharing one commit version.""" + + version: int + mutations: Tuple[CdcMutation, ...] + + +class CdcConsumeResult(NamedTuple): + """Copied mutation groups and the delivered watermark, even for empty replies.""" + + mutations: Tuple[CdcVersionedMutations, ...] + last_consumed_version: int + + +class CdcConsumer(_FDBBase): + """An owned native CDC consumer handle. + + Only one consume or acknowledge operation may be outstanding per handle. + Acknowledgements affect the whole stream, which must have only one active + logical consumer. Closing a handle does not acknowledge or remove its stream. + """ + + def __init__(self): + self._lock = threading.Lock() + self._pointer = ctypes.c_void_p() + + def __del__(self): + if getattr(self, "_pointer", None): + self.close() + + def _check_open(self): + if not self._pointer: + raise ValueError("CDC consumer is closed") + + def close(self): + """Release this handle without acknowledging or removing the stream.""" + with self._lock: + if self._pointer: + pointer = self._pointer + self._pointer = None + self.capi.fdb_cdc_consumer_destroy(pointer) + + def __enter__(self): + with self._lock: + self._check_open() + return self + + def __exit__(self, exc_type, exc_value, traceback): + self.close() + + def consume(self): + """Long-poll for complete commit groups; return a CdcConsumeResult future. + + Consumption advances the delivered cursor, not durable retention. Do not + consume again until the previous reply has been durably processed. + """ + with self._lock: + self._check_open() + return FutureCdcConsumeResult( + self.capi.fdb_cdc_consumer_consume(self._pointer) + ) + + def acknowledge(self): + """Durably acknowledge the current delivered position; return a void future. + + Call only after durably processing every mutation through that position. + """ + with self._lock: + self._check_open() + return FutureVoid(self.capi.fdb_cdc_consumer_acknowledge(self._pointer)) + + def get_position(self): + """Return the current CdcCursor; this does not acknowledge the position.""" + with self._lock: + self._check_open() + stream_id = ctypes.c_uint64() + version = ctypes.c_int64() + self.capi.fdb_cdc_consumer_get_position( + self._pointer, ctypes.byref(stream_id), ctypes.byref(version) + ) + return CdcCursor(stream_id.value, version.value) + + class Cluster(_FDBBase): def __init__(self, cluster_file): self.cluster_file = cluster_file @@ -1438,6 +1708,46 @@ class KeyStruct(ctypes.Structure): _pack_ = 4 +class KeyRangeStruct(ctypes.Structure): + _pack_ = 4 + _fields_ = [ + ("begin_key", ctypes.POINTER(ctypes.c_byte)), + ("begin_key_length", ctypes.c_int), + ("end_key", ctypes.POINTER(ctypes.c_byte)), + ("end_key_length", ctypes.c_int), + ] + + +class CdcStreamInfoStruct(ctypes.Structure): + _pack_ = 4 + _fields_ = [ + ("name", KeyStruct), + ("stream_id", ctypes.c_uint64), + ("key_range", KeyRangeStruct), + ("min_version", ctypes.c_int64), + ] + + +class CdcMutationStruct(ctypes.Structure): + _pack_ = 4 + _fields_ = [ + ("type", ctypes.c_uint8), + ("param1", ctypes.POINTER(ctypes.c_byte)), + ("param1_length", ctypes.c_int), + ("param2", ctypes.POINTER(ctypes.c_byte)), + ("param2_length", ctypes.c_int), + ] + + +class CdcVersionedMutationsStruct(ctypes.Structure): + _pack_ = 4 + _fields_ = [ + ("version", ctypes.c_int64), + ("mutations", ctypes.POINTER(CdcMutationStruct)), + ("mutation_count", ctypes.c_int), + ] + + class KeyValue(object): def __init__(self, key, value): self.key = key @@ -1558,6 +1868,16 @@ def optionalParamToBytes(v): return (v, len(v)) +def _require_cdc_api_version(): + global _cdc_c_api_initialized + if fdb.get_api_version() < 800: + raise RuntimeError("Native CDC requires API version 800 or later") + with _network_thread_reentrant_lock: + if not _cdc_c_api_initialized: + _init_cdc_c_api() + _cdc_c_api_initialized = True + + _FDBBase.capi = _capi _CBFUNC = ctypes.CFUNCTYPE(None, ctypes.c_void_p) @@ -1910,6 +2230,89 @@ def init_c_api(): _capi.fdb_transaction_reset.restype = None +def _init_cdc_c_api(): + # Older libraries may support API 800 without these experimental symbols. + # Resolve the whole surface before using it, and leave non-CDC users alone. + signatures = ( + ( + "fdb_future_get_cdc_stream_info_array", + [ + ctypes.c_void_p, + ctypes.POINTER(ctypes.POINTER(CdcStreamInfoStruct)), + ctypes.POINTER(ctypes.c_int), + ], + ctypes.c_int, + ), + ( + "fdb_future_get_cdc_consumer", + [ctypes.c_void_p, ctypes.POINTER(ctypes.c_void_p)], + ctypes.c_int, + ), + ( + "fdb_future_get_cdc_versioned_mutations", + [ + ctypes.c_void_p, + ctypes.POINTER(ctypes.POINTER(CdcVersionedMutationsStruct)), + ctypes.POINTER(ctypes.c_int), + ctypes.POINTER(ctypes.c_int64), + ], + ctypes.c_int, + ), + ( + "fdb_database_register_cdc_stream", + [ + ctypes.c_void_p, + ctypes.c_void_p, + ctypes.c_int, + ctypes.c_void_p, + ctypes.c_int, + ctypes.c_void_p, + ctypes.c_int, + ], + ctypes.c_void_p, + ), + ( + "fdb_database_remove_cdc_stream", + [ctypes.c_void_p, ctypes.c_void_p, ctypes.c_int], + ctypes.c_void_p, + ), + ("fdb_database_list_cdc_streams", [ctypes.c_void_p], ctypes.c_void_p), + ( + "fdb_database_create_cdc_consumer", + [ctypes.c_void_p, ctypes.c_void_p, ctypes.c_int], + ctypes.c_void_p, + ), + ( + "fdb_database_resume_cdc_consumer", + [ctypes.c_void_p, ctypes.c_uint64, ctypes.c_int64], + ctypes.c_void_p, + ), + ("fdb_cdc_consumer_destroy", [ctypes.c_void_p], None), + ("fdb_cdc_consumer_consume", [ctypes.c_void_p], ctypes.c_void_p), + ("fdb_cdc_consumer_acknowledge", [ctypes.c_void_p], ctypes.c_void_p), + ( + "fdb_cdc_consumer_get_position", + [ + ctypes.c_void_p, + ctypes.POINTER(ctypes.c_uint64), + ctypes.POINTER(ctypes.c_int64), + ], + ctypes.c_int, + ), + ) + try: + functions = [getattr(_capi, name) for name, _, _ in signatures] + except AttributeError as error: + raise RuntimeError( + "The loaded FoundationDB C library does not support native CDC" + ) from error + for function, (_, argtypes, restype) in zip(functions, signatures): + function.argtypes = argtypes + function.restype = restype + if restype is ctypes.c_int: + function.errcheck = check_error_code + + if hasattr(ctypes.pythonapi, "Py_IncRef"): def _pin_callback(cb): diff --git a/bindings/python/tests/native_cdc_tests.py b/bindings/python/tests/native_cdc_tests.py new file mode 100644 index 00000000000..ea2d3481f68 --- /dev/null +++ b/bindings/python/tests/native_cdc_tests.py @@ -0,0 +1,408 @@ +#!/usr/bin/env python3 +# +# native_cdc_tests.py +# +# This source file is part of the FoundationDB open source project +# +# Copyright 2026 Apple Inc. and the FoundationDB project authors +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +import argparse +import ctypes +import gc +import threading +import time +import unittest +import uuid +from unittest import mock + +import fdb + + +def wait(future, timeout=30): + ready = threading.Event() + future.on_ready(lambda _: ready.set()) + if not ready.wait(timeout): + future.cancel() + raise AssertionError( + "CDC operation did not complete within {} seconds".format(timeout) + ) + return future.wait() + + +class CdcDecodingTests(unittest.TestCase): + def test_packed_c_layout(self): + impl = fdb.impl + pointer_size = ctypes.sizeof(ctypes.c_void_p) + layouts = ( + (impl.KeyRangeStruct, 2 * pointer_size + 8), + (impl.CdcStreamInfoStruct, 3 * pointer_size + 28), + (impl.CdcMutationStruct, 2 * pointer_size + 12), + (impl.CdcVersionedMutationsStruct, pointer_size + 12), + ) + for structure, size in layouts: + with self.subTest(structure=structure.__name__): + self.assertEqual(structure._pack_, 4) + self.assertEqual(ctypes.sizeof(structure), size) + self.assertEqual(impl.CdcMutationStruct.param1.offset, 4) + self.assertEqual(impl.CdcStreamInfoStruct.stream_id.offset, pointer_size + 4) + + def test_unknown_mutation_type_and_copied_bytes(self): + impl = fdb.impl + key = ctypes.create_string_buffer(b"key\x00\xff") + value = ctypes.create_string_buffer(b"\x00value\xff") + mutations = (impl.CdcMutationStruct * 2)( + impl.CdcMutationStruct( + 255, + ctypes.cast(key, ctypes.POINTER(ctypes.c_byte)), + len(key) - 1, + ctypes.cast(value, ctypes.POINTER(ctypes.c_byte)), + len(value) - 1, + ), + impl.CdcMutationStruct(fdb.CdcMutationType.SET_VALUE, None, 0, None, 0), + ) + groups = (impl.CdcVersionedMutationsStruct * 2)( + impl.CdcVersionedMutationsStruct(100, mutations, 2), + impl.CdcVersionedMutationsStruct(101, None, 0), + ) + + def get_result(pointer, out_groups, out_count, out_version): + ctypes.cast( + out_groups, + ctypes.POINTER(ctypes.POINTER(impl.CdcVersionedMutationsStruct)), + )[0] = groups + ctypes.cast(out_count, ctypes.POINTER(ctypes.c_int))[0] = len(groups) + ctypes.cast(out_version, ctypes.POINTER(ctypes.c_int64))[0] = 150 + + future = impl.FutureCdcConsumeResult(1) + future.capi = mock.Mock() + future.capi.fdb_future_is_ready.return_value = 1 + future.capi.fdb_future_get_cdc_versioned_mutations.side_effect = get_result + result = future.wait() + ctypes.memset(key, 0, len(key)) + ctypes.memset(value, 0, len(value)) + mutations[0].type = 0 + groups[0].version = 0 + del future + gc.collect() + + self.assertEqual( + result, + fdb.CdcConsumeResult( + ( + fdb.CdcVersionedMutations( + 100, + ( + fdb.CdcMutation(255, b"key\x00\xff", b"\x00value\xff"), + fdb.CdcMutation(fdb.CdcMutationType.SET_VALUE, b"", b""), + ), + ), + fdb.CdcVersionedMutations(101, ()), + ), + 150, + ), + ) + self.assertIs(type(result.mutations[0].mutations[0].type), int) + with self.assertRaises(AttributeError): + result.last_consumed_version = 0 + + def test_empty_reply_preserves_progress(self): + def get_result(pointer, out_groups, out_count, out_version): + ctypes.cast(out_count, ctypes.POINTER(ctypes.c_int))[0] = 0 + ctypes.cast(out_version, ctypes.POINTER(ctypes.c_int64))[0] = 2**63 - 1 + + future = fdb.impl.FutureCdcConsumeResult(1) + future.capi = mock.Mock() + future.capi.fdb_future_is_ready.return_value = 1 + future.capi.fdb_future_get_cdc_versioned_mutations.side_effect = get_result + self.assertEqual(future.wait(), fdb.CdcConsumeResult((), 2**63 - 1)) + + +class NativeCdcTests(unittest.TestCase): + def setUp(self): + self.prefix = b"python-cdc/" + uuid.uuid4().bytes + b"\x00/" + self.name = self.prefix + b"stream\x00\xff" + self.begin = self.prefix + b"range/" + self.end = self.prefix + b"range0" + self.addCleanup(self.db.clear_range, self.prefix, self.prefix + b"\xff") + + def register(self): + stream_id = wait(self.db.register_cdc_stream(self.name, self.begin, self.end)) + self.addCleanup(lambda: wait(self.db.remove_cdc_stream(self.name))) + return stream_id + + def stream_info(self): + future = self.db.list_cdc_streams() + streams = wait(future) + self.assertEqual(future.wait(), streams) + future._release_memory() + del future + gc.collect() + return next(stream for stream in streams if stream.name == self.name) + + def commit(self, write): + tr = self.db.create_transaction() + tr.options.set_timeout(30000) + while True: + try: + write(tr) + wait(tr.commit()) + return tr.get_committed_version() + except fdb.FDBError as error: + wait(tr.on_error(error)) + + def consume_through(self, consumer, version): + groups = {} + deadline = time.monotonic() + 60 + while time.monotonic() < deadline: + future = consumer.consume() + result = wait(future) + self.assertEqual(future.wait(), result) + future._release_memory() + del future + gc.collect() + self.assertEqual( + consumer.get_position().last_consumed_version, + result.last_consumed_version, + ) + self.assertIsInstance(result.mutations, tuple) + for group in result.mutations: + self.assertIsInstance(group.mutations, tuple) + if group.version in groups: + self.assertEqual(groups[group.version], group.mutations) + groups[group.version] = group.mutations + if result.last_consumed_version >= version: + return groups + self.fail("CDC did not deliver the committed version within 60 seconds") + + def assert_closed(self, consumer): + consumer.close() + consumer.close() + for operation in ( + consumer.consume, + consumer.acknowledge, + consumer.get_position, + ): + with self.subTest(operation=operation.__name__): + with self.assertRaises(ValueError): + operation() + + def test_stream_lifecycle_and_result_ownership(self): + stream_id = self.register() + self.assertGreater(stream_id, 0) + self.assertEqual( + wait(self.db.register_cdc_stream(self.name, self.begin, self.end)), + stream_id, + ) + info = self.stream_info() + self.assertEqual( + (info.name, info.stream_id, info.begin_key, info.end_key), + (self.name, stream_id, self.begin, self.end), + ) + self.assertGreaterEqual(info.min_version, 0) + with self.assertRaises(AttributeError): + info.name = b"different" + + create_future = self.db.create_cdc_consumer(self.name) + consumer = wait(create_future) + self.addCleanup(consumer.close) + self.assertIs(create_future.wait(), consumer) + self.assertIs(create_future.result(), consumer) + self.assertIsNone(create_future.exception()) + create_future._release_memory() + del create_future + gc.collect() + self.assertEqual(consumer.get_position(), fdb.CdcCursor(stream_id, -1)) + + first_key = self.begin + b"first\x00\xff" + second_key = self.begin + b"second" + empty_key = self.begin + b"empty" + first_value = b"\x00first\xffvalue\x00" + second_value = b"second-value" + + def write_sets(tr): + tr[first_key] = first_value + tr[second_key] = second_value + tr[self.prefix + b"outside"] = b"not in the stream" + + set_version = self.commit(write_sets) + empty_version = self.commit(lambda tr: tr.set(empty_key, b"")) + groups = self.consume_through(consumer, empty_version) + self.assertCountEqual( + groups[set_version], + ( + fdb.CdcMutation(fdb.CdcMutationType.SET_VALUE, first_key, first_value), + fdb.CdcMutation( + fdb.CdcMutationType.SET_VALUE, second_key, second_value + ), + ), + ) + self.assertEqual( + groups[empty_version], + (fdb.CdcMutation(fdb.CdcMutationType.SET_VALUE, empty_key, b""),), + ) + cursor = consumer.get_position() + self.assert_closed(consumer) + self.assertEqual(self.stream_info().min_version, info.min_version) + + resume_future = self.db.resume_cdc_consumer(cursor) + resumed = wait(resume_future) + self.addCleanup(resumed.close) + self.assertIs(resume_future.wait(), resumed) + del resume_future + gc.collect() + with resumed: + self.assertEqual(resumed.get_position(), cursor) + # A resumed handle has no local delivery proof. Its acknowledgement + # must be at or behind a fresh database read version. + deadline = time.monotonic() + 30 + while True: + remaining = deadline - time.monotonic() + self.assertGreater(remaining, 0, "Read version did not reach cursor") + tr = self.db.create_transaction() + read_version = wait( + tr.get_read_version(), + timeout=max(0, deadline - time.monotonic()), + ) + if read_version >= cursor.last_consumed_version: + break + time.sleep(min(0.01, max(0, deadline - time.monotonic()))) + # Reconcile the durable checkpoint, then reissue the acknowledgement. + for _ in range(2): + self.assertIsNone(wait(resumed.acknowledge())) + self.assertEqual( + self.stream_info().min_version, + cursor.last_consumed_version + 1, + ) + clear_end = first_key + b"\x00" + counter_key = self.begin + b"counter" + operand = b"\x01\x00\x00\x00" + + def write_raw_mutations(tr): + tr.clear_range(first_key, clear_end) + tr.add(counter_key, operand) + + raw_version = self.commit(write_raw_mutations) + groups = self.consume_through(resumed, raw_version) + self.assertCountEqual( + groups[raw_version], + ( + fdb.CdcMutation( + fdb.CdcMutationType.CLEAR_RANGE, first_key, clear_end + ), + fdb.CdcMutation(fdb.CdcMutationType.ADD, counter_key, operand), + ), + ) + self.assertIsNone(wait(resumed.acknowledge())) + self.assert_closed(resumed) + + self.assertIsNone(wait(self.db.remove_cdc_stream(self.name))) + self.assertIsNone(wait(self.db.remove_cdc_stream(self.name))) + self.assertNotIn( + self.name, [stream.name for stream in wait(self.db.list_cdc_streams())] + ) + + def test_native_errors_propagate(self): + with self.assertRaises(fdb.FDBError): + wait(self.db.create_cdc_consumer(self.name)) + self.register() + with self.assertRaises(fdb.FDBError): + wait(self.db.register_cdc_stream(self.name, self.begin, self.end + b"\x00")) + self.assertEqual(self.stream_info().end_key, self.end) + + def test_missing_cdc_symbols_preserves_normal_database_use(self): + impl = fdb.impl + capi = impl._capi + + class WithoutCdcSymbols: + def __getattr__(self, name): + if "_cdc_" in name: + raise AttributeError(name) + return getattr(capi, name) + + with mock.patch.object(impl, "_capi", WithoutCdcSymbols()): + with mock.patch.object(impl, "_cdc_c_api_initialized", False): + impl.init_c_api() + with self.assertRaisesRegex( + RuntimeError, "does not support native CDC" + ): + self.db.list_cdc_streams() + self.assertFalse(impl._cdc_c_api_initialized) + key = self.prefix + b"compatibility" + self.db[key] = b"ordinary value" + self.assertEqual(self.db[key], b"ordinary value") + + def test_cursor_integer_boundaries(self): + for cursor in ( + fdb.CdcCursor(0, -(2**63)), + fdb.CdcCursor(2**64 - 1, 2**63 - 1), + ): + with self.subTest(cursor=cursor): + with wait(self.db.resume_cdc_consumer(cursor)) as consumer: + self.assertEqual(consumer.get_position(), cursor) + for cursor in ( + fdb.CdcCursor(-1, -1), + fdb.CdcCursor(2**64, -1), + fdb.CdcCursor(1, -(2**63) - 1), + fdb.CdcCursor(1, 2**63), + ): + with self.subTest(cursor=cursor): + with self.assertRaises(ValueError): + self.db.resume_cdc_consumer(cursor) + for cursor in (fdb.CdcCursor(1.5, -1), fdb.CdcCursor(1, "0")): + with self.subTest(cursor=cursor): + with self.assertRaises(TypeError): + self.db.resume_cdc_consumer(cursor) + + +class LegacyApiTests(unittest.TestCase): + def test_normal_database_use_and_cdc_version_gate(self): + key = b"python-cdc-legacy/" + uuid.uuid4().bytes + self.addCleanup(self.db.clear, key) + self.db[key] = b"normal database operations still work" + self.assertEqual(self.db[key], b"normal database operations still work") + operations = ( + lambda: self.db.register_cdc_stream(b"legacy", b"a", b"z"), + lambda: self.db.remove_cdc_stream(b"legacy"), + lambda: self.db.list_cdc_streams(), + lambda: self.db.create_cdc_consumer(b"legacy"), + lambda: self.db.resume_cdc_consumer(fdb.CdcCursor(1, -1)), + ) + for operation in operations: + with self.assertRaisesRegex(RuntimeError, "requires API version 800"): + operation() + + +if __name__ == "__main__": + parser = argparse.ArgumentParser(description="Native CDC Python binding tests") + parser.add_argument("--cluster-file", "-C", required=True) + parser.add_argument("--api-version", type=int, default=fdb.LATEST_API_VERSION) + parser.add_argument("--verbose", "-V", action="store_true") + args = parser.parse_args() + fdb.api_version(args.api_version) + db = fdb.open(args.cluster_file) + db.options.set_transaction_timeout(30000) + NativeCdcTests.db = db + LegacyApiTests.db = db + classes = ( + (CdcDecodingTests, NativeCdcTests) + if args.api_version >= 800 + else (LegacyApiTests,) + ) + suite = unittest.TestSuite( + unittest.defaultTestLoader.loadTestsFromTestCase(cls) for cls in classes + ) + result = unittest.TextTestRunner(verbosity=2 if args.verbose else 1).run(suite) + raise SystemExit(0 if result.wasSuccessful() else 1) diff --git a/design/cdc.md b/design/cdc.md index 178110c5009..f7ea5c327be 100644 --- a/design/cdc.md +++ b/design/cdc.md @@ -11,11 +11,11 @@ or transaction-system recovery. ## Background -This design describes the native C++ interface, its C binding, and its server -implementation. The feature is disabled by default behind `ENABLE_NATIVE_CDC`; +This design describes the native C++ interface, its C and Python bindings, and +its server implementation. The feature is disabled by default behind `ENABLE_NATIVE_CDC`; the native CDC workloads explicitly enable it, and simulation may randomly -enable it. The client interface is exposed through the native C++ API and C -binding; it does not expose an external protocol compatibility guarantee. +enable it. The client interface is exposed through the native C++ API and C and +Python bindings; it does not expose an external protocol compatibility guarantee. The implementation uses the following terms: @@ -85,7 +85,7 @@ The current implementation does not attempt to provide: range requires removing and registering a stream. * Throughput-aware assignment of streams across CDC proxies. * Throughput-aware movement of streams between CDC tags. -* Language-specific bindings beyond the C API. +* Language-specific bindings beyond the C and Python APIs. ## Client interface @@ -95,7 +95,8 @@ value types and the thread-safe surface shared with language bindings are in roles are in the private `fdbclient/NativeCdcInternal.h`; cursor and wire request types are in `fdbclient/CDCProxyInterface.h`. The public C binding is declared in `bindings/c/foundationdb/fdb_c.h` and documented in -`documentation/sphinx/source/api-c.rst`. +`documentation/sphinx/source/api-c.rst`. The Python binding is documented in +`documentation/sphinx/source/api-python.rst`. `CDCStreamId` is a `uint64_t` typedef. CDC tag IDs are 16-bit, so one configured tag pool can contain at most 65,536 distinct tags. @@ -773,8 +774,8 @@ policy simple. response to load. A future implementation can use versioned tag history to make such changes without losing the ability to read earlier tagged data. * The CDC client surface does not yet provide language-specific bindings beyond - the C API, administrative tooling, or a higher-level consumer checkpoint - abstraction. + the C and Python APIs, administrative tooling, or a higher-level consumer + checkpoint abstraction. These improvements must preserve the acknowledgement and retired-pop invariants above. In particular, moving a stream between tags cannot forget an diff --git a/documentation/sphinx/source/api-python.rst b/documentation/sphinx/source/api-python.rst index 45802eb3825..ad6422c86fa 100644 --- a/documentation/sphinx/source/api-python.rst +++ b/documentation/sphinx/source/api-python.rst @@ -455,7 +455,200 @@ Database options .. method:: Database.options.set_snapshot_ryw_disable() |option-db-snapshot-ryw-disable-blurb| - + +.. _api-python-cdc: + +Native change data capture (CDC) +================================ + +CDC provides durable, named streams of committed mutations for a non-empty, +half-open user-key range. Select API version 800 or later with +:func:`api_version` before using this interface. New stream registration also +requires the cluster's ``ENABLE_NATIVE_CDC`` admission knob. Existing streams +can still be listed or removed while admission is disabled; consumer creation, +resume, consumption, and acknowledgement also remain available. Repeating an +existing same-name, same-range registration remains idempotent. + +Unlike the synchronous database key-value methods, the CDC database methods +and the consumer's ``consume()`` and ``acknowledge()`` methods return +:ref:`futures `. Use their usual ``wait()``, ``on_ready()``, +or ``result()`` interfaces to observe completion and errors. These operations +are not methods on :class:`Transaction`, and cannot be made atomic with +application writes by using :func:`transactional`. + +.. warning:: + + A stream has one shared acknowledgement frontier, not one per handle. + Use only one active logical consumer per stream. Each handle permits only + one outstanding ``consume()`` or ``acknowledge()`` operation at a time. + Acknowledge only after every mutation through the delivered position and + the corresponding application checkpoint are durable. Neither consuming + nor closing a handle acknowledges automatically. An abandoned stream can + retain unread log history indefinitely until acknowledged or removed. + +Stream management +----------------- + +.. method:: Database.register_cdc_stream(name, begin_key, end_key) + + Registers the byte-string ``name`` for ``[begin_key, end_key)`` in normal + user key space. The name and range must be non-empty. Repeating the same + name and range is idempotent; reusing a name with another range fails. + Returns a future whose value is the unsigned 64-bit stream ID as a Python + ``int``. Register long-lived streams rather than a stream per request. + +.. method:: Database.remove_cdc_stream(name) + + Removes the byte-string stream name and relinquishes its unread history. + Removing a missing name succeeds. Returns a future whose value is + ``None``. Removal is terminal for existing consumers; registering the + same name again does not redirect their cursors to the new stream. + +.. method:: Database.list_cdc_streams() + + Returns a future whose value is a list of :class:`CdcStreamInfo` records. + +.. method:: Database.create_cdc_consumer(name) + + Returns a future whose value is a :class:`CdcConsumer` for the existing + byte-string stream name at its initial position. + +.. method:: Database.resume_cdc_consumer(cursor) + + Returns a future whose value is a :class:`CdcConsumer` constructed from + a checkpointed :class:`CdcCursor`. The cursor contains no process-local + state. Stream existence and cursor validity are checked when consuming + or acknowledging, not by this method. Resume only from a durably processed + checkpoint. Before the first consume, reissue :meth:`CdcConsumer.acknowledge` + and wait for it to complete to reconcile a possibly interrupted + acknowledgement. Unacknowledged mutations may be redelivered after CDC + proxy replacement, so processing must tolerate replay. + +Result types +------------ + +The following records are immutable tuples with named fields. Returned names, +keys, and mutation parameters are Python-owned ``bytes``, copied from the +native result. They remain valid after the future or consumer is released. + +.. class:: CdcCursor(stream_id, last_consumed_version) + + A stable unsigned 64-bit stream ID and the signed 64-bit version through + which mutations have been delivered. Both fields are Python ``int`` + values. A delivered cursor is not proof of processing or acknowledgement. + +.. class:: CdcStreamInfo(name, stream_id, begin_key, end_key, min_version) + + A stream's byte-string name, integer ID, half-open registered key range, + and durable minimum required version. ``min_version`` is a retention + frontier, not a snapshot version for the registered key range. + +.. class:: CdcMutation(type, param1, param2) + + One raw mutation. ``type`` is an integer, including for unrecognized + mutation types; ``param1`` and ``param2`` are byte strings. For + ``SET_VALUE`` they are the key and value; for ``CLEAR_RANGE`` they are + the begin and end keys, clipped to the registered range; for atomic + mutations they are the key and operand. CDC returns raw mutation + operations, not a materialized post-mutation value for every key. + +.. class:: CdcMutationType + + An ``enum.IntEnum`` of known raw mutation type values, matching the C API + constants without the ``FDB_CDC_MUTATION_TYPE_`` prefix: ``SET_VALUE``, + ``CLEAR_RANGE``, ``ADD``, ``AND``, ``OR``, ``XOR``, ``APPEND_IF_FITS``, + ``MAX``, ``MIN``, ``SET_VERSIONSTAMPED_KEY``, ``SET_VERSIONSTAMPED_VALUE``, + ``BYTE_MIN``, ``BYTE_MAX``, ``MIN_V2``, ``AND_V2``, and ``COMPARE_AND_CLEAR``. + These constants are not exhaustive. Compare ``mutation.type`` to known + constants, but handle unknown integers without assuming that constructing + ``CdcMutationType(mutation.type)`` will succeed. + +.. class:: CdcVersionedMutations(version, mutations) + + One complete commit-version group, with an integer ``version`` and a + tuple of :class:`CdcMutation` records. Preserve this grouping when + processing a reply. + +.. class:: CdcConsumeResult(mutations, last_consumed_version) + + ``mutations`` is a tuple of :class:`CdcVersionedMutations` groups; + ``last_consumed_version`` is the integer delivered cursor after the reply. + The cursor may advance across commit-version gaps with no returned + mutations. Even an empty reply can advance the cursor; do not infer it + from the last mutation group or skip checkpointing and acknowledgement + solely because ``mutations`` is empty. + +Consumer lifecycle +------------------ + +.. class:: CdcConsumer + + An owned native consumer handle, obtained from a database create or resume + future. It remains valid independently of that future. Use it as a + context manager or call :meth:`CdcConsumer.close` when finished. + +.. method:: CdcConsumer.consume() + + Long-polls for the next delivered position and complete commit-version + groups. Returns a future whose value is :class:`CdcConsumeResult`. + Successful consumption advances the in-memory position, but does not + release durable retention. Finish processing a reply before consuming + again if that reply may need to be retried. Following proxy replacement, + the client may rewind to its last successful durable acknowledgement and + redeliver unacknowledged mutations. + +.. method:: CdcConsumer.acknowledge() + + Durably acknowledges the handle's current delivered position, allowing + history through that position to be released. Returns a future whose + value is ``None``. Wait for it before starting another consume or + acknowledgement. This is not atomic with writes to a downstream system + or an application checkpoint. + +.. method:: CdcConsumer.get_position() + + Returns the current :class:`CdcCursor` synchronously. + +.. method:: CdcConsumer.close() + + Releases the local handle. Repeated calls are harmless. It does not + acknowledge the delivered position or remove the stream. Exiting the + consumer's context manager calls this method, including on exceptions. + +For example, the following loop supports both an initial start and a restart. +The application-supplied ``load_checkpoint`` returns a durable +:class:`CdcCursor`, or ``None`` on the first start. ``apply_and_checkpoint`` +must durably apply complete version groups and record the cursor consistently, +and must tolerate replay without repeating non-idempotent side effects. It +must also handle an empty reply. If it raises, the loop does not acknowledge +the reply:: + + import fdb + + fdb.api_version(800) + db = fdb.open() + db.register_cdc_stream(b"orders", b"order/", b"order0").wait() + + saved_cursor = load_checkpoint() + if saved_cursor is None: + consumer_future = db.create_cdc_consumer(b"orders") + else: + consumer_future = db.resume_cdc_consumer(saved_cursor) + + with consumer_future.wait() as consumer: + if saved_cursor is not None: + consumer.acknowledge().wait() + while True: + reply = consumer.consume().wait() + apply_and_checkpoint(reply.mutations, consumer.get_position()) + consumer.acknowledge().wait() + +The acknowledgement immediately after resume closes the crash gap between +persisting the checkpoint and completing its acknowledgement; reissuing an +already durable acknowledgement is safe. This requires a cursor whose +mutations are known to have been durably processed. Never invent a cursor or +advance a checkpoint or acknowledgement beyond that position. + Transactional decoration ======================== From ee1fa01e67b5c509ea786233935f365204b14586 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Fri, 21 Aug 2026 00:05:27 -0700 Subject: [PATCH 007/170] Fix Python CDC checkpoint recovery example --- documentation/sphinx/source/api-python.rst | 28 ++++++++++++++++++---- 1 file changed, 23 insertions(+), 5 deletions(-) diff --git a/documentation/sphinx/source/api-python.rst b/documentation/sphinx/source/api-python.rst index ad6422c86fa..cf1a51b9a79 100644 --- a/documentation/sphinx/source/api-python.rst +++ b/documentation/sphinx/source/api-python.rst @@ -519,10 +519,15 @@ Stream management a checkpointed :class:`CdcCursor`. The cursor contains no process-local state. Stream existence and cursor validity are checked when consuming or acknowledging, not by this method. Resume only from a durably processed - checkpoint. Before the first consume, reissue :meth:`CdcConsumer.acknowledge` - and wait for it to complete to reconcile a possibly interrupted - acknowledgement. Unacknowledged mutations may be redelivered after CDC - proxy replacement, so processing must tolerate replay. + checkpoint. Before the first consume, wait until a fresh database read + version reaches ``cursor.last_consumed_version``, then reissue + :meth:`CdcConsumer.acknowledge` and wait for it to complete to reconcile a + possibly interrupted acknowledgement. A resumed handle lacks the original + handle's delivery proof, so an unacknowledged cursor ahead of that read + version is rejected with ``client_invalid_operation`` even if it was + previously delivered. Bound the read-version wait rather than retrying all + invalid-operation errors. Unacknowledged mutations may be redelivered after + CDC proxy replacement, so processing must tolerate replay. Result types ------------ @@ -623,6 +628,8 @@ and must tolerate replay without repeating non-idempotent side effects. It must also handle an empty reply. If it raises, the loop does not acknowledge the reply:: + import time + import fdb fdb.api_version(800) @@ -637,13 +644,24 @@ the reply:: with consumer_future.wait() as consumer: if saved_cursor is not None: + deadline = time.monotonic() + 30 + while True: + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError("CDC checkpoint is ahead of the read version") + tr = db.create_transaction() + tr.options.set_read_lock_aware() + tr.options.set_timeout(max(1, int(remaining * 1000))) + if tr.get_read_version().wait() >= saved_cursor.last_consumed_version: + break + time.sleep(min(0.01, max(0, deadline - time.monotonic()))) consumer.acknowledge().wait() while True: reply = consumer.consume().wait() apply_and_checkpoint(reply.mutations, consumer.get_position()) consumer.acknowledge().wait() -The acknowledgement immediately after resume closes the crash gap between +The acknowledgement after the read-version wait closes the crash gap between persisting the checkpoint and completing its acknowledgement; reissuing an already durable acknowledgement is safe. This requires a cursor whose mutations are known to have been durably processed. Never invent a cursor or From 786ff0f02dcd3dc214cd1484fce81e9038276ce2 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 30 Aug 2026 12:52:21 -0700 Subject: [PATCH 008/170] Support multiple disjoint ranges in native CDC streams --- bindings/c/fdb_c.cpp | 44 +++- bindings/c/foundationdb/fdb_c.h | 9 +- bindings/c/test/unit/unit_tests.cpp | 101 ++++++-- design/cdc.md | 129 +++++++---- documentation/sphinx/source/api-c.rst | 25 +- fdbclient/MultiVersionTransaction.cpp | 29 ++- fdbclient/NativeCdc.cpp | 111 +++++++-- fdbclient/NativeCdcInternal.h | 2 +- fdbclient/SystemData.cpp | 16 +- fdbclient/ThreadSafeTransaction.cpp | 8 +- .../include/fdbclient/CDCProxyInterface.h | 6 +- fdbclient/include/fdbclient/IClientApi.h | 3 +- .../fdbclient/MultiVersionTransaction.h | 13 +- fdbclient/include/fdbclient/NativeCdc.h | 6 +- fdbclient/include/fdbclient/NativeCdcClient.h | 4 +- fdbclient/include/fdbclient/SystemData.h | 6 +- .../include/fdbclient/ThreadSafeTransaction.h | 2 +- fdbserver/cdcproxy/CDCProxy.cpp | 139 ++++++++--- fdbserver/logsystem/ApplyMetadataMutation.cpp | 2 +- fdbserver/logsystem/CDCRoutingTable.cpp | 63 +++-- .../fdbserver/logsystem/CDCRoutingTable.h | 7 +- fdbserver/workloads/NativeCdcEndToEnd.cpp | 215 ++++++++++++++++-- tests/CMakeLists.txt | 1 + tests/fast/NativeCdcMultipleRanges.toml | 21 ++ 24 files changed, 747 insertions(+), 215 deletions(-) create mode 100644 tests/fast/NativeCdcMultipleRanges.toml diff --git a/bindings/c/fdb_c.cpp b/bindings/c/fdb_c.cpp index 041efd33dd3..a0293d92809 100644 --- a/bindings/c/fdb_c.cpp +++ b/bindings/c/fdb_c.cpp @@ -95,6 +95,29 @@ FDBKeyRange copyNativeCdcKeyRange(Arena& arena, KeyRangeRef source) { return FDBKeyRange{ begin.begin(), begin.size(), end.begin(), end.size() }; } +std::vector copyNativeCdcRanges(FDBKeyRange const* ranges, int rangeCount) { + if (rangeCount <= 0 || rangeCount > NATIVE_CDC_MAX_RANGES || ranges == nullptr) { + throw client_invalid_operation(); + } + std::vector result; + result.reserve(rangeCount); + for (int i = 0; i < rangeCount; ++i) { + auto const& range = ranges[i]; + if (range.begin_key_length < 0 || range.end_key_length < 0 || + (range.begin_key_length > 0 && range.begin_key == nullptr) || + (range.end_key_length > 0 && range.end_key == nullptr)) { + throw client_invalid_operation(); + } + KeyRef begin(range.begin_key, range.begin_key_length); + KeyRef end(range.end_key, range.end_key_length); + if (begin >= end) { + throw client_invalid_operation(); + } + result.emplace_back(KeyRangeRef(begin, end)); + } + return result; +} + CNativeCdcStreamInfoArray makeCNativeCdcStreamInfoArray(std::vector const& source) { CNativeCdcStreamInfoArray result; result.streams.reserve(result.arena, source.size()); @@ -102,7 +125,12 @@ CNativeCdcStreamInfoArray makeCNativeCdcStreamInfoArray(std::vectorregisterNativeCdcStream( - KeyRef(name, name_length), - KeyRangeRef(KeyRef(begin_key, begin_key_length), KeyRef(end_key, end_key_length))) - .extractPtr());); + if (name_length <= 0 || name == nullptr) { throw client_invalid_operation(); } auto rangesCopy = + copyNativeCdcRanges(ranges, range_count); + return (FDBFuture*)(DB(db)->registerNativeCdcStream(KeyRef(name, name_length), rangesCopy).extractPtr());); } extern "C" DLLEXPORT FDBFuture* fdb_database_remove_cdc_stream(FDBDatabase* db, uint8_t const* name, int name_length) { diff --git a/bindings/c/foundationdb/fdb_c.h b/bindings/c/foundationdb/fdb_c.h index cf344d51df3..05c1d045bcb 100644 --- a/bindings/c/foundationdb/fdb_c.h +++ b/bindings/c/foundationdb/fdb_c.h @@ -213,7 +213,8 @@ typedef enum { typedef struct cdc_stream_info { FDBKey name; uint64_t stream_id; - FDBKeyRange key_range; + const FDBKeyRange* ranges; + int range_count; int64_t min_version; } FDBCdcStreamInfo; @@ -437,10 +438,8 @@ DLLEXPORT WARN_UNUSED_RESULT fdb_error_t fdb_database_create_transaction(FDBData DLLEXPORT WARN_UNUSED_RESULT FDBFuture* fdb_database_register_cdc_stream(FDBDatabase* db, uint8_t const* name, int name_length, - uint8_t const* begin_key, - int begin_key_length, - uint8_t const* end_key, - int end_key_length); + FDBKeyRange const* ranges, + int range_count); DLLEXPORT WARN_UNUSED_RESULT FDBFuture* fdb_database_remove_cdc_stream(FDBDatabase* db, uint8_t const* name, diff --git a/bindings/c/test/unit/unit_tests.cpp b/bindings/c/test/unit/unit_tests.cpp index 31e2f25c746..ac584d120d1 100644 --- a/bindings/c/test/unit/unit_tests.cpp +++ b/bindings/c/test/unit/unit_tests.cpp @@ -1972,6 +1972,36 @@ TEST_CASE("fdb_database_get_server_protocol") { fdb_future_destroy(protocolFuture); } +TEST_CASE("CDC C binding rejects invalid registration ranges") { + const uint8_t name[] = "invalid-cdc-stream"; + const uint8_t begin[] = "a"; + const uint8_t end[] = "b"; + FDBKeyRange range{ begin, 1, end, 1 }; + auto checkInvalid = [&](uint8_t const* nameInput, int nameLength, FDBKeyRange const* ranges, int rangeCount) { + FDBFuture* future = fdb_database_register_cdc_stream(db, nameInput, nameLength, ranges, rangeCount); + REQUIRE(future != nullptr); + fdb_check(fdb_future_block_until_ready(future)); + CHECK(fdb_future_get_error(future) == 2000); // client_invalid_operation + fdb_future_destroy(future); + }; + + checkInvalid(name, sizeof(name) - 1, &range, -1); + checkInvalid(name, sizeof(name) - 1, &range, 0); + checkInvalid(name, sizeof(name) - 1, &range, 1025); + checkInvalid(name, sizeof(name) - 1, nullptr, 1); + checkInvalid(nullptr, 1, &range, 1); + checkInvalid(name, -1, &range, 1); + checkInvalid(name, 0, &range, 1); + for (FDBKeyRange invalid : { FDBKeyRange{ begin, -1, end, 1 }, + FDBKeyRange{ begin, 1, end, -1 }, + FDBKeyRange{ nullptr, 1, end, 1 }, + FDBKeyRange{ begin, 1, nullptr, 1 }, + FDBKeyRange{ begin, 1, begin, 1 }, + FDBKeyRange{ end, 1, begin, 1 } }) { + checkInvalid(name, sizeof(name) - 1, &invalid, 1); + } +} + TEST_CASE("CDC C binding end-to-end") { using FuturePtr = std::unique_ptr; using ConsumerPtr = std::unique_ptr; @@ -2073,10 +2103,13 @@ TEST_CASE("CDC C binding end-to-end") { }; const std::string streamName = key("cdc-stream"); - const std::string rangeBegin = key("cdc-data/"); + const std::string rangeBegin = key("cdc-data/a/"); const std::string rangeEnd = strinc_str(rangeBegin); + const std::string secondRangeBegin = key("cdc-data/c/"); + const std::string secondRangeEnd = strinc_str(secondRangeBegin); const std::string firstKey = rangeBegin + "first"; - const std::string secondKey = rangeBegin + "second"; + const std::string secondKey = secondRangeBegin + "second"; + const std::string gapKey = key("cdc-data/b/gap"); const std::string outsideKey = key("outside-cdc-range"); const std::string firstValue = "first-value"; const std::string secondValue = "second-value"; @@ -2088,18 +2121,30 @@ TEST_CASE("CDC C binding end-to-end") { std::string nameInput = streamName; std::string beginInput = rangeBegin; std::string endInput = rangeEnd; - auto registerFuture = - ownFuture(fdb_database_register_cdc_stream(db, - reinterpret_cast(nameInput.data()), - nameInput.size(), - reinterpret_cast(beginInput.data()), - beginInput.size(), - reinterpret_cast(endInput.data()), - endInput.size())); + std::string secondBeginInput = secondRangeBegin; + std::string secondEndInput = secondRangeEnd; + std::vector rangesInput{ { reinterpret_cast(secondBeginInput.data()), + static_cast(secondBeginInput.size()), + reinterpret_cast(secondEndInput.data()), + static_cast(secondEndInput.size()) }, + { reinterpret_cast(beginInput.data()), + static_cast(beginInput.size()), + reinterpret_cast(endInput.data()), + static_cast(endInput.size()) } }; + // Repeated intervals must not duplicate delivered mutations. + rangesInput.push_back(rangesInput.back()); + auto registerFuture = ownFuture(fdb_database_register_cdc_stream(db, + reinterpret_cast(nameInput.data()), + nameInput.size(), + rangesInput.data(), + rangesInput.size())); REQUIRE(registerFuture != nullptr); std::fill(nameInput.begin(), nameInput.end(), 'x'); std::fill(beginInput.begin(), beginInput.end(), 'x'); std::fill(endInput.begin(), endInput.end(), 'x'); + std::fill(secondBeginInput.begin(), secondBeginInput.end(), 'x'); + std::fill(secondEndInput.begin(), secondEndInput.end(), 'x'); + std::fill(rangesInput.begin(), rangesInput.end(), FDBKeyRange{ nullptr, -1, nullptr, -1 }); waitForSuccess(registerFuture.get()); uint64_t streamId = 0; @@ -2119,10 +2164,16 @@ TEST_CASE("CDC C binding end-to-end") { } foundStream = true; CHECK(streams[i].stream_id == streamId); - CHECK(std::string(reinterpret_cast(streams[i].key_range.begin_key), - streams[i].key_range.begin_key_length) == rangeBegin); - CHECK(std::string(reinterpret_cast(streams[i].key_range.end_key), - streams[i].key_range.end_key_length) == rangeEnd); + REQUIRE(streams[i].range_count == 2); + REQUIRE(streams[i].ranges != nullptr); + CHECK(std::string(reinterpret_cast(streams[i].ranges[0].begin_key), + streams[i].ranges[0].begin_key_length) == rangeBegin); + CHECK(std::string(reinterpret_cast(streams[i].ranges[0].end_key), + streams[i].ranges[0].end_key_length) == rangeEnd); + CHECK(std::string(reinterpret_cast(streams[i].ranges[1].begin_key), + streams[i].ranges[1].begin_key_length) == secondRangeBegin); + CHECK(std::string(reinterpret_cast(streams[i].ranges[1].end_key), + streams[i].ranges[1].end_key_length) == secondRangeEnd); CHECK(streams[i].min_version >= 0); } REQUIRE(foundStream); @@ -2145,8 +2196,10 @@ TEST_CASE("CDC C binding end-to-end") { CHECK(positionStreamId == streamId); CHECK(positionVersion == -1); - const int64_t setVersion = - commitSetValues({ { firstKey, firstValue }, { secondKey, secondValue }, { outsideKey, "outside-value" } }); + const int64_t setVersion = commitSetValues({ { firstKey, firstValue }, + { secondKey, secondValue }, + { gapKey, "gap-value" }, + { outsideKey, "outside-value" } }); auto setReply = consumeThroughVersion(consumer.get(), setVersion); CHECK(setReply.version == setVersion); CHECK(setReply.lastConsumedVersion >= setVersion); @@ -2181,14 +2234,20 @@ TEST_CASE("CDC C binding end-to-end") { CHECK(positionStreamId == streamId); CHECK(positionVersion == setReply.lastConsumedVersion); - const std::string clearEnd = strinc_str(firstKey); + const std::string clearEnd = strinc_str(secondKey); const int64_t clearVersion = commitClearRange(firstKey, clearEnd); auto clearReply = consumeThroughVersion(resumedConsumer.get(), clearVersion); CHECK(clearReply.version == clearVersion); - REQUIRE(clearReply.mutations.size() == 1); - CHECK(clearReply.mutations[0].type == FDB_CDC_MUTATION_TYPE_CLEAR_RANGE); - CHECK(clearReply.mutations[0].param1 == firstKey); - CHECK(clearReply.mutations[0].param2 == clearEnd); + REQUIRE(clearReply.mutations.size() == 2); + std::map expectedClears{ { firstKey, rangeEnd }, { secondRangeBegin, clearEnd } }; + for (auto const& mutation : clearReply.mutations) { + CHECK(mutation.type == FDB_CDC_MUTATION_TYPE_CLEAR_RANGE); + auto expected = expectedClears.find(mutation.param1); + REQUIRE(expected != expectedClears.end()); + CHECK(mutation.param2 == expected->second); + expectedClears.erase(expected); + } + CHECK(expectedClears.empty()); auto resumedAcknowledgeFuture = ownFuture(fdb_cdc_consumer_acknowledge(resumedConsumer.get())); REQUIRE(resumedAcknowledgeFuture != nullptr); diff --git a/design/cdc.md b/design/cdc.md index 254f889db8f..0379260cba5 100644 --- a/design/cdc.md +++ b/design/cdc.md @@ -3,11 +3,11 @@ ## Objective Native Change Data Capture (CDC) provides a FoundationDB-native mechanism for -reading committed mutations for a registered key range. A client registers a -named stream, creates a consumer for that name, consumes batches of mutations, -and acknowledges processed versions. The implementation persists enough state -to retain unread TLog data and to resume stream service after CDC proxy failure -or transaction-system recovery. +reading committed mutations for a registered set of key ranges. A client +registers a named stream, creates a consumer for that name, consumes batches of +mutations, and acknowledges processed versions. The implementation persists +enough state to retain unread TLog data and to resume stream service after CDC +proxy failure or transaction-system recovery. ## Background @@ -19,13 +19,13 @@ binding; it does not expose an external protocol compatibility guarantee. The implementation uses the following terms: -* A **stream** is a durable named registration for a fixed user key range. +* A **stream** is a durable named registration for a fixed set of user key ranges. * A **cursor** identifies one stream and the version through which a consumer has read. * A **CDC tag** is a TLog tag with locality `tagLocalityCDC`. Commit proxies append these tags to mutations covered by registered streams. * A **CDC proxy** reads tagged TLog mutation streams, filters mutations to a - registered range, serves consumers, and coordinates acknowledgement-driven + registered range set, serves consumers, and coordinates acknowledgement-driven log popping. CDC is not implemented as a storage server change feed. It captures mutations @@ -36,12 +36,11 @@ and release its own log history without changing user data storage. Native CDC is intended to provide: -* Durable, named registrations for single key ranges in normal user key - space. The initial API intentionally registers exactly one half-open - `[begin, end)` range per stream; callers that need multiple disjoint ranges - register multiple streams. +* Durable, named registrations for non-empty sets of half-open `[begin, end)` + ranges in normal user key space. A stream captures the union of its ranges + and excludes the gaps between them. * A consumer API in which a client only needs a stream name after - registration, rather than repeating its registered range on every read. + registration, rather than repeating its registered ranges on every read. * Ordered mutation batches identified by FoundationDB commit versions. * Durable acknowledgements that determine how much CDC-tagged TLog history may be popped. @@ -81,8 +80,8 @@ The current implementation does not attempt to provide: arbitrary application state. Such an API would be useful for queue-like transactional asynchronous processing pipelines, but the initial interface leaves that composition to the consumer. -* Dynamic stream range changes. A name is registered for one range; changing a - range requires removing and registering a stream. +* Dynamic stream range changes. A name is registered for an immutable range + set; changing membership requires removing and registering a stream. * Throughput-aware assignment of streams across CDC proxies. * Throughput-aware movement of streams between CDC tags. * Language-specific bindings beyond the C API. @@ -100,7 +99,7 @@ declared in `bindings/c/foundationdb/fdb_c.h` and documented in configured tag pool can contain at most 65,536 distinct tags. ```cpp -Future registerNativeCdcStreamClient(Database cx, Key name, KeyRange keys); +Future registerNativeCdcStreamClient(Database cx, Key name, std::vector ranges); Future removeNativeCdcStreamClient(Database cx, Key name); Future> listNativeCdcStreamsClient(Database cx); @@ -108,9 +107,11 @@ Future> createNativeCdcConsumer(Database cx, Key na Reference resumeNativeCdcConsumer(Database cx, CDCCursor position); ``` -`registerNativeCdcStreamClient()` accepts exactly one `KeyRange`. The range is -interpreted with FoundationDB's usual half-open `[begin, end)` semantics. -Multi-range registration is not part of the initial API. +`registerNativeCdcStreamClient()` accepts a non-empty vector of `KeyRange` +values, each interpreted with FoundationDB's usual half-open `[begin, end)` +semantics. Registration sorts the ranges and merges overlaps, duplicates, and +adjacent intervals into a canonical union. All ranges share one stream identity, +CDC tag assignment, proxy owner, cursor, and durable acknowledgement watermark. Registration and removal are low-rate control-plane operations intended for stable stream lifecycles, not per-request stream churn. The initial implementation does not define a supported registrations-per-second target: @@ -123,7 +124,7 @@ A stream registration contains: struct NativeCdcStreamInfo { Key name; CDCStreamId streamId; - KeyRange keys; + std::vector ranges; Version minVersion; }; ``` @@ -175,7 +176,8 @@ struct CDCConsumeReply { A typical consumer loop is: ```cpp -co_await registerNativeCdcStreamClient(db, "orders"_sr, KeyRangeRef("order/"_sr, "order0"_sr)); +co_await registerNativeCdcStreamClient( + db, "orders"_sr, { KeyRangeRef("order/"_sr, "order0"_sr), KeyRangeRef("payment/"_sr, "payment0"_sr) }); Reference consumer = co_await createNativeCdcConsumer(db, "orders"_sr); while (true) { @@ -220,17 +222,20 @@ version. ### Registration and removal semantics -`registerNativeCdcStreamClient()` accepts a non-empty stream name and a -non-empty range entirely within normal user keys. Registration of an existing -name with the same range is idempotent. Registering an existing name with a -different range is rejected. +`registerNativeCdcStreamClient()` accepts a non-empty stream name and between +1 and 1,024 non-empty ranges entirely within normal user keys. The encoded +canonical range set must fit within FoundationDB's value-size limit. +Registration of an existing name with the same canonical union is idempotent, +regardless of input order, duplication, overlap, or adjacent subdivisions. +Registering an existing name with a different union is rejected. Listing a +stream returns its canonical ranges in key order. Registration establishes an initial minimum version using the registration transaction's commit version. Mutations committed after the registration has become visible are routed to the stream's CDC tag. The initial minimum version also supplies the first retention watermark for its TLog history. When CDC admission is disabled, this gate applies only to creation of a new -name. Repeating an existing same-name/same-range registration remains +name. Repeating an existing same-name/same-range-set registration remains idempotent, including repair of a missing durable owner, so an administrator can still drain state created while the feature was enabled. @@ -244,9 +249,18 @@ will never be assigned again. ### Consumption and expiration -Consumption is ordered by commit version. Mutations from a clear range are -intersected with the stream's registered range before being returned; a -single-key mutation is returned only if its key is within that range. +Consumption is ordered by commit version. A single-key mutation is returned +only if its key is in one of the registered ranges. A clear range is intersected +with every selected interval it overlaps and emits one clear for each non-empty +intersection, in key order. These fragments remain in the original mutation +position relative to other mutations from the same commit version. For example, +a stream selecting `[a,c)` and `[x,z)` returns `[b,c)` and `[x,y)` for a clear +of `[b,y)`, and never clears the unselected gap `[c,x)`. + +A single cursor and acknowledgement cover all selected ranges. The consumer +must process every range through the acknowledged version; a slow range holds +back retention for the whole stream. Consumers that need independent progress +or lifecycle management should use separate streams. For an active stream, unacknowledged CDC mutations are retained by its durable minimum version: TLogs must not pop tagged data that the stream may still @@ -312,7 +326,7 @@ consumers. CDC proxies do not participate in committing user transactions. They consume the extra tagged log streams, buffer readable results, filter shared tagged -data back to each stream's registered range, and pop data after durable +data back to each stream's registered range set, and pop data after durable acknowledgement permits it. The cluster controller recruits CDC proxies, publishes their interfaces, and @@ -335,7 +349,7 @@ in transaction state: | --- | --- | --- | | `\xff/cdc/name/` | `CDCStreamId` | Resolves a user-visible name to its durable stream identity. | | `\xff/cdc/maxStreamId` | `CDCStreamId` | Allocates monotonic stream identifiers. | -| `\xff/cdc/keys/` | `KeyRange` | Stores the immutable registered range for an active stream. | +| `\xff/cdc/keys/` | `std::vector` | Stores the canonical immutable registered range set for an active stream. | | `\xff/cdc/tagHistory///` | empty | Records the CDC tag assignment history used for routing and historical reads. | | `\xff/cdc/proxies//` | empty | Stores the CDC proxy assigned to an active stream. | | `\xff/cdc/proxyAssignmentChange` | version/change signal | Wakes ownership monitoring when durable assignments change. | @@ -376,15 +390,16 @@ actual final pop to perform. Registration runs as a durable metadata transaction: -1. It validates the stream name and registered normal key range. +1. It validates the stream name, range count, normal key ranges, and encoded + metadata size, and canonicalizes the range union. 2. It checks whether the name is already registered and applies the idempotent - same-name/same-range rule, even when admission is disabled. + same-name/same-range-set rule, even when admission is disabled. 3. For a new name, it validates the feature knob. 4. It allocates a new monotonically increasing `CDCStreamId`. 5. It selects a CDC tag using current active stream counts. The allocator uses the least populated tag among `NATIVE_CDC_TAG_COUNT` tags (256 by default), choosing the lowest tag ID on a tie. -6. It records the stream name, range, initial tag history entry, and +6. It records the stream name, canonical ranges, initial tag history entry, and versionstamped initial minimum version. 7. It records an available CDC proxy owner and signals assignment monitoring. @@ -421,7 +436,7 @@ assigns proxy `P1`. Registration writes: * Transaction state `\xff/cdc/name/orders -> 7`. -* Transaction state `\xff/cdc/keys/7 -> ["order/", "order0")`. +* Transaction state `\xff/cdc/keys/7 -> { ["order/", "order0") }`. * Transaction state `\xff/cdc/tagHistory/7/995/tagLocalityCDC:3 -> empty`. * Transaction state `\xff/cdc/proxies/7/P1 -> empty` and the assignment-change signal. @@ -429,7 +444,7 @@ Registration writes: If the consumer later acknowledges mutations through version `1200`, `\xff\x02/cdc/minVersion/7` advances to `1201`. If the stream is then removed -at version `1500`, removal deletes the active name, range, proxy, tag-history, +at version `1500`, removal deletes the active name, ranges, proxy, tag-history, and `minVersion` rows, and writes retired final-pop work for `tagLocalityCDC:3`: a transaction-state `\xff/cdc/retiredTagPop/` marker and a storage-backed `\xff\x02/cdc/retiredTagPopVersion/` watermark for @@ -454,7 +469,7 @@ proxy processing. The cost of a broad clear range is proportional to the number of CDC stream ranges and tags it intersects, not to the number of keys in the cleared range. The logged CDC payload remains a clear-range mutation on each relevant tag, and -the CDC proxy later clips that clear to the consumer's registered range. A +the CDC proxy later clips that clear to the consumer's registered ranges. A clear that spans many CDC ranges can therefore add many CDC tag destinations and later produce many per-stream clipped clears. Commit-proxy and CDC-proxy metrics should make this visible by reporting CDC routing matches, CDC tag @@ -463,8 +478,9 @@ fanout, filtered bytes, and consumer lag. A shared CDC tag is a multiplexed log stream. A mutation routed because of stream A may be read by the proxy serving stream B if both share the tag. Consequently, the CDC proxy filters every read mutation against B's registered -range before returning it to B's consumer. Filtering also clips clear ranges -to the stream range. +range set before returning it to B's consumer. Filtering splits clear ranges +at unselected gaps. A clear intersecting multiple ranges of one stream receives +that stream's CDC tag only once. Shared-tag false positives are expected, especially when `NATIVE_CDC_TAG_COUNT` is small or active streams are unevenly distributed. The @@ -496,14 +512,14 @@ mapping for the normally configured IDs. A CDC proxy owns a set of active stream IDs. For each owned stream it loads: -* The registered key range. +* The canonical registered key ranges. * The durable minimum required version. * Its current CDC tag and versioned tag history. The proxy reads data from TLogs through `LogSystemConsumer::peekSingle()`. When a stream has historical assignments, the proxy uses the history to select the tag appropriate for the version interval it is reading. It filters -mutations to the registered range and stores versioned mutation batches in a +mutations to the registered range set and stores versioned mutation batches in a per-stream in-memory buffer. All raw peek windows and stream buffers owned by one CDC proxy share a @@ -530,9 +546,10 @@ TLog retention are the source of resumability, while the proxy buffer is a delivery optimization. One tagged TLog message can match many overlapping streams. The proxy estimates -that expansion per stream and commit version, materializes only a subset that -fits the current bounded pass, and reopens the tag cursor for the remaining -streams. It never requests raw-plus-retained permits beyond +that expansion, including clear fragments across disjoint ranges, per stream +and commit version, materializes only a subset that fits the current bounded +pass, and reopens the tag cursor for the remaining streams. It never requests +raw-plus-retained permits beyond `CDC_PROXY_BUFFER_BYTES`. If the filtered mutations for one stream at one commit version exceed the capacity remaining after the raw peek reservation, that consume fails with `server_overloaded`; operators must configure the @@ -556,7 +573,9 @@ default) and contains only complete commit-version groups. If more buffered data is available, the reply stops before the next version and advances `lastConsumedVersion` only through the delivered prefix, including any empty version gap before that next mutation. If one complete filtered version cannot -fit in the reply budget, the consume fails with `server_overloaded`. +fit in the reply budget, the consume fails with `server_overloaded`. The bound +applies to the combined mutations across every selected range; a version is +never split by range to fit a reply. The owning proxy accepts a cursor only when the position has already been delivered by that owner or is covered by the stream's durable acknowledgement watermark. A fabricated or otherwise unproven cursor is rejected instead of @@ -610,7 +629,7 @@ metadata scan. ### Removing a stream -Removing a stream eliminates its active name, range, tag history, minimum +Removing a stream eliminates its active name, ranges, tag history, minimum version, and ownership rows. Removal must not unconditionally pop each tag in the removed history: a different live stream may share a tag and still need older data. @@ -702,6 +721,14 @@ cluster role. ## Rollout and migration considerations +Native CDC is unreleased. The multi-range metadata and client interfaces replace +the earlier single-range representation without a compatibility decoder or API +overload. Test deployments using the earlier representation must remove their +streams and finish retired cleanup before upgrading. Upgrade every CDC-capable +server binary and client before registering streams in the new format; mixed +old and new CDC implementations are unsupported, and existing cursors do not +bridge that change. + ### Feature gating `ENABLE_NATIVE_CDC` defaults to false. In simulation it may be randomly enabled @@ -747,7 +774,7 @@ The implementation is structured around the following properties: reuse of a removed stream name cannot cause an existing consumer to read a new stream. * **Range correctness:** CDC proxies return only mutations within a stream's - registered range, even when its tag is shared with other streams. + registered range union, even when its tag is shared with other streams. * **Acknowledgement monotonicity:** durable minimum required versions advance only forward. * **Shared-tag retention:** tagged data is popped no farther than the minimum @@ -834,7 +861,13 @@ The basic native CDC workload covers: * Registering, listing, consuming, acknowledging, and removing streams. * Name-based consumer creation, including end-to-end clear-range clipping. -* Rejection of incompatible same-name registrations. +* Canonical multi-range registration, equivalent same-name registrations, and + rejection of incompatible same-name registrations. +* Gap exclusion for shared-tag mutations, ordered clear splitting across + disjoint intervals, and replay of a complete unacknowledged multi-range + version after proxy replacement using one cursor and acknowledgement. +* Multi-range unread history retained across transaction-system recovery and + new commits routed through the recovered range metadata. * Targeted CDC proxy termination, durable reassignment, and recovery of stream service, including independent publication when two proxies fail together. * Errors for stale consume and acknowledgement requests after removal. @@ -885,7 +918,7 @@ Production operation needs tooling beyond the initial native API: | Operator need | Current mechanism | Needed production tooling | | --- | --- | --- | -| Find a stalled consumer | `CDCProxyMetrics` reports the oldest required stream ID, acknowledgement lag, safe-pop distance, and buffer pressure. | A stream-listing view that joins stream name, range, owner, lag, retained bytes, and attributable TLog retention. | +| Find a stalled consumer | `CDCProxyMetrics` reports the oldest required stream ID, acknowledgement lag, safe-pop distance, and buffer pressure. | A stream-listing view that joins stream name, ranges, owner, lag, retained bytes, and attributable TLog retention. | | Stop retaining abandoned history | `removeNativeCdcStreamClient()` explicitly removes a named stream and relinquishes its unread history. | An authenticated force-removal command with an explicit data-loss confirmation and audit trail. | | Prevent new CDC load while draining existing work | Disabling `ENABLE_NATIVE_CDC` rejects new names while allowing existing streams to drain or be removed. | A status command that distinguishes admission state from active and retired CDC work. | | Recover a downstream system after discarding CDC history | The downstream system can rebuild from a full scan after the stream is removed and later registered again. | A runbook that coordinates stream removal, downstream rebuild, and safe re-registration. | diff --git a/documentation/sphinx/source/api-c.rst b/documentation/sphinx/source/api-c.rst index c568dea5ffe..7af87e31e04 100644 --- a/documentation/sphinx/source/api-c.rst +++ b/documentation/sphinx/source/api-c.rst @@ -586,7 +586,10 @@ select API version 800 or later before calling these functions. .. type:: FDBCdcStreamInfo A listed CDC stream, including its name, stable stream ID, registered - key range, and durable minimum required version. + key ranges, and durable minimum required version. ``ranges`` points to + ``range_count`` sorted, non-empty, disjoint ranges; overlapping and adjacent + registered ranges are merged. The array and its key bytes have the same + lifetime as the stream-info result. .. type:: FDBCdcMutation @@ -605,11 +608,21 @@ select API version 800 or later before calling these functions. remains valid after the originating future is destroyed. Destroy it exactly once with :func:`fdb_cdc_consumer_destroy()`. -.. function:: FDBFuture* fdb_database_register_cdc_stream(FDBDatabase* database, uint8_t const* name, int name_length, uint8_t const* begin_key, int begin_key_length, uint8_t const* end_key, int end_key_length) - - Registers ``name`` for the non-empty half-open range ``[begin_key, - end_key)`` in normal user key space. Repeating the same name and range is - idempotent; reusing a name with a different range fails. The future returns +.. function:: FDBFuture* fdb_database_register_cdc_stream(FDBDatabase* database, uint8_t const* name, int name_length, FDBKeyRange const* ranges, int range_count) + + Registers ``name`` for the union of ``range_count`` non-empty half-open + ranges in normal user key space. The input must contain between 1 and 1024 + ranges. Ranges may arrive in any order; overlapping, duplicate, and adjacent + ranges are merged. The encoded canonical range set must fit within the + database value-size limit. Mutations in gaps between the resulting ranges are + excluded, and clear-range mutations are clipped to each intersecting range. + + A stream's range set is immutable. Repeating the same name and equivalent + range union is idempotent; reusing a name with a different union fails. + All ranges share one consumer cursor and acknowledgement position. + The range array, key bytes, and name bytes are copied before this function + returns. Invalid counts, negative lengths, null required pointers, or invalid + ranges return a future with ``client_invalid_operation``. The future returns the ``uint64_t`` stream ID, extracted with :func:`fdb_future_get_uint64()`. .. function:: FDBFuture* fdb_database_remove_cdc_stream(FDBDatabase* database, uint8_t const* name, int name_length) diff --git a/fdbclient/MultiVersionTransaction.cpp b/fdbclient/MultiVersionTransaction.cpp index 95a8fbae282..5cfbbe958fe 100644 --- a/fdbclient/MultiVersionTransaction.cpp +++ b/fdbclient/MultiVersionTransaction.cpp @@ -399,9 +399,13 @@ NativeCdcStreamInfo copyNativeCdcStreamInfo(FdbCApi::FDBNativeCdcStreamInfo cons NativeCdcStreamInfo result; result.name = Key(StringRef(source.name.key, source.name.keyLength)); result.streamId = source.streamId; - result.keys = KeyRange( - KeyRangeRef(KeyRef(static_cast(source.keyRange.beginKey), source.keyRange.beginKeyLength), - KeyRef(static_cast(source.keyRange.endKey), source.keyRange.endKeyLength))); + result.ranges.reserve(source.rangeCount); + for (int i = 0; i < source.rangeCount; ++i) { + auto const& range = source.ranges[i]; + result.ranges.emplace_back( + KeyRangeRef(KeyRef(static_cast(range.beginKey), range.beginKeyLength), + KeyRef(static_cast(range.endKey), range.endKeyLength))); + } result.minVersion = source.minVersion; return result; } @@ -555,13 +559,21 @@ ThreadFuture DLDatabase::createSnapshot(const StringRef& uid, const String return toThreadFuture(api, f, [](FdbCApi::FDBFuture* f, FdbCApi* api) { return Void(); }); } -ThreadFuture DLDatabase::registerNativeCdcStream(const KeyRef& name, const KeyRangeRef& keys) { +ThreadFuture DLDatabase::registerNativeCdcStream(const KeyRef& name, const std::vector& ranges) { if (!api->databaseRegisterNativeCdcStream) { return unsupported_operation(); } + if (ranges.empty() || ranges.size() > NATIVE_CDC_MAX_RANGES) { + return client_invalid_operation(); + } - FdbCApi::FDBFuture* f = api->databaseRegisterNativeCdcStream( - db, name.begin(), name.size(), keys.begin.begin(), keys.begin.size(), keys.end.begin(), keys.end.size()); + std::vector cRanges; + cRanges.reserve(ranges.size()); + for (auto const& range : ranges) { + cRanges.push_back({ range.begin.begin(), range.begin.size(), range.end.begin(), range.end.size() }); + } + FdbCApi::FDBFuture* f = + api->databaseRegisterNativeCdcStream(db, name.begin(), name.size(), cRanges.data(), cRanges.size()); return toThreadFuture(api, f, [](FdbCApi::FDBFuture* f, FdbCApi* api) { uint64_t streamId; FdbCApi::fdb_error_t error = api->futureGetUInt64(f, &streamId); @@ -1656,8 +1668,9 @@ ThreadFuture MultiVersionDatabase::createSnapshot(const StringRef& uid, co return executeOperation(&IDatabase::createSnapshot, uid, snapshot_command); } -ThreadFuture MultiVersionDatabase::registerNativeCdcStream(const KeyRef& name, const KeyRangeRef& keys) { - return executeOperation(&IDatabase::registerNativeCdcStream, name, keys); +ThreadFuture MultiVersionDatabase::registerNativeCdcStream(const KeyRef& name, + const std::vector& ranges) { + return executeOperation(&IDatabase::registerNativeCdcStream, name, ranges); } ThreadFuture MultiVersionDatabase::removeNativeCdcStream(const KeyRef& name) { diff --git a/fdbclient/NativeCdc.cpp b/fdbclient/NativeCdc.cpp index b0ce1f7282c..a5cc8d24c84 100644 --- a/fdbclient/NativeCdc.cpp +++ b/fdbclient/NativeCdc.cpp @@ -95,8 +95,35 @@ class NativeCdcIdentifierAllocator { } }; -void validateNativeCdcStream(KeyRef const& name, KeyRangeRef const& keys) { - if (name.empty() || keys.empty() || !normalKeys.contains(keys)) { +void normalizeNativeCdcStreamRanges(KeyRef const& name, std::vector& ranges) { + if (name.empty() || ranges.empty() || ranges.size() > NATIVE_CDC_MAX_RANGES) { + throw client_invalid_operation(); + } + for (const auto& range : ranges) { + if (range.begin >= range.end || !normalKeys.contains(range)) { + throw client_invalid_operation(); + } + } + std::sort( + ranges.begin(), ranges.end(), [](const KeyRange& lhs, const KeyRange& rhs) { return lhs.begin < rhs.begin; }); + size_t count = 0; + for (const auto& range : ranges) { + if (count > 0 && range.begin <= ranges[count - 1].end) { + if (range.end > ranges[count - 1].end) { + ranges[count - 1] = KeyRange(KeyRangeRef(ranges[count - 1].begin, range.end)); + } + } else { + ranges[count++] = range; + } + } + ranges.resize(count); + + int64_t keyBytes = 0; + for (const auto& range : ranges) { + keyBytes += static_cast(range.begin.size()) + range.end.size(); + } + if (keyBytes > CLIENT_KNOBS->VALUE_SIZE_LIMIT || + cdcStreamKeysValue(ranges).size() > CLIENT_KNOBS->VALUE_SIZE_LIMIT) { throw client_invalid_operation(); } } @@ -386,8 +413,8 @@ Future> getNativeCdcStreamProxyForRemoval(Database c } // namespace -Future registerNativeCdcStream(Database cx, Key name, KeyRange keys, UID proxyId) { - validateNativeCdcStream(name, keys); +Future registerNativeCdcStream(Database cx, Key name, std::vector ranges, UID proxyId) { + normalizeNativeCdcStreamRanges(name, ranges); Transaction tr(cx); while (true) { @@ -401,7 +428,7 @@ Future registerNativeCdcStream(Database cx, Key name, KeyRange keys if (currentId.present()) { const CDCStreamId streamId = decodeCDCStreamNameValue(currentId.get()); Optional currentKeys = co_await tr.get(cdcStreamKeyFor(streamId)); - if (!currentKeys.present() || decodeCDCStreamKeysValue(currentKeys.get()) != keys) { + if (!currentKeys.present() || decodeCDCStreamKeysValue(currentKeys.get()) != ranges) { throw client_invalid_operation(); } if (!(co_await getNativeCdcProxyAssignment(&tr, streamId)).present()) { @@ -437,7 +464,7 @@ Future registerNativeCdcStream(Database cx, Key name, KeyRange keys tr.set(nameKey, cdcStreamNameValue(streamId)); tr.set(cdcMaxStreamIdKey, cdcMaxStreamIdValue(streamId)); - tr.set(cdcStreamKeyFor(streamId), cdcStreamKeysValue(keys)); + tr.set(cdcStreamKeyFor(streamId), cdcStreamKeysValue(ranges)); tr.set(cdcTagHistoryKeyFor(streamId, registrationVersion, tag), Value()); tr.atomicOp( cdcMinVersionKeyFor(streamId), cdcVersionstampedMinVersionValue(), MutationRef::SetVersionstampedValue); @@ -559,12 +586,12 @@ Future> listNativeCdcStreams(Database cx) { begin = keyAfter(page.back().key); } - std::unordered_map streamKeys; + std::unordered_map> streamRanges; begin = cdcStreamKeys.begin; while (begin < cdcStreamKeys.end) { RangeResult page = co_await tr.getRange(KeyRangeRef(begin, cdcStreamKeys.end), CLIENT_KNOBS->TOO_MANY); for (const auto& kv : page) { - streamKeys.emplace(decodeCDCStreamKey(kv.key), decodeCDCStreamKeysValue(kv.value)); + streamRanges.emplace(decodeCDCStreamKey(kv.key), decodeCDCStreamKeysValue(kv.value)); } if (!page.more) { break; @@ -589,11 +616,11 @@ Future> listNativeCdcStreams(Database cx) { std::vector result; result.reserve(names.size()); for (auto& [name, streamId] : names) { - auto keys = streamKeys.find(streamId); + auto ranges = streamRanges.find(streamId); auto minVersion = minVersions.find(streamId); - if (keys != streamKeys.end() && minVersion != minVersions.end()) { - result.push_back( - NativeCdcStreamInfo{ std::move(name), streamId, keys->second, minVersion->second }); + if (ranges != streamRanges.end() && minVersion != minVersions.end()) { + result.push_back(NativeCdcStreamInfo{ + std::move(name), streamId, std::move(ranges->second), minVersion->second }); } } co_return result; @@ -693,8 +720,8 @@ Future acknowledgeNativeCdcStream(Database cx, } } -Future registerNativeCdcStreamClient(Database cx, Key name, KeyRange keys) { - validateNativeCdcStream(name, keys); +Future registerNativeCdcStreamClient(Database cx, Key name, std::vector ranges) { + normalizeNativeCdcStreamRanges(name, ranges); Optional previousProxy; while (true) { Future proxyChanged = cx->clientInfo->onChange(); @@ -724,7 +751,7 @@ Future registerNativeCdcStreamClient(Database cx, Key name, KeyRang CDCProxyInterface proxy = selectedProxy.get(); try { Future> request = - proxy.registerStream.tryGetReply(CDCRegisterStreamRequest(name, keys)); + proxy.registerStream.tryGetReply(CDCRegisterStreamRequest(name, ranges)); // Assignment publications for other streams also change ClientDBInfo. Keep this request alive while its // proxy remains published; abandoning it can let a server-side retry recreate the stream after removal. while (true) { @@ -886,6 +913,60 @@ Future NativeCdcConsumer::acknowledge() { return acknowledgeImpl(Reference::addRef(this)); } +TEST_CASE("/NativeCDC/RangeNormalization") { + std::vector ranges{ + KeyRangeRef("x"_sr, "z"_sr), KeyRangeRef("b"_sr, "d"_sr), KeyRangeRef("a"_sr, "b"_sr), + KeyRangeRef("a"_sr, "c"_sr), KeyRangeRef("b"_sr, "c"_sr), KeyRangeRef("x"_sr, "z"_sr) + }; + const std::vector expected{ KeyRangeRef("a"_sr, "d"_sr), KeyRangeRef("x"_sr, "z"_sr) }; + normalizeNativeCdcStreamRanges("orders"_sr, ranges); + ASSERT(ranges == expected); + normalizeNativeCdcStreamRanges("orders"_sr, ranges); + ASSERT(ranges == expected); + + std::vector entireKeyspace{ normalKeys }; + normalizeNativeCdcStreamRanges("all"_sr, entireKeyspace); + ASSERT(entireKeyspace == std::vector{ normalKeys }); + return Void(); +} + +TEST_CASE("/NativeCDC/InvalidRanges") { + auto expectInvalid = [](KeyRef name, std::vector ranges) { + try { + normalizeNativeCdcStreamRanges(name, ranges); + } catch (Error& error) { + ASSERT_EQ(error.code(), error_code_client_invalid_operation); + return; + } + ASSERT(false); + }; + expectInvalid(KeyRef(), { normalKeys }); + expectInvalid("orders"_sr, {}); + expectInvalid("orders"_sr, { KeyRangeRef("a"_sr, "a"_sr) }); + expectInvalid("orders"_sr, { normalKeys, systemKeys }); + expectInvalid("orders"_sr, std::vector(NATIVE_CDC_MAX_RANGES + 1, normalKeys)); + + std::vector maximumCount; + std::vector oversizedMetadata; + const int endpointLength = CLIENT_KNOBS->VALUE_SIZE_LIMIT / (2 * NATIVE_CDC_MAX_RANGES); + for (int i = 0; i < NATIVE_CDC_MAX_RANGES; ++i) { + const std::string prefix = format("%04d/", i); + maximumCount.emplace_back(KeyRangeRef(prefix + "a", prefix + "z")); + ASSERT_GT(endpointLength, prefix.size()); + oversizedMetadata.emplace_back(KeyRangeRef(prefix + std::string(endpointLength - prefix.size(), 'a'), + prefix + std::string(endpointLength - prefix.size(), 'z'))); + } + normalizeNativeCdcStreamRanges("orders"_sr, maximumCount); + ASSERT_EQ(maximumCount.size(), NATIVE_CDC_MAX_RANGES); + ASSERT_LE(2 * NATIVE_CDC_MAX_RANGES * endpointLength, CLIENT_KNOBS->VALUE_SIZE_LIMIT); + ASSERT_GT(cdcStreamKeysValue(oversizedMetadata).size(), CLIENT_KNOBS->VALUE_SIZE_LIMIT); + expectInvalid("orders"_sr, oversizedMetadata); + const std::string oversizedBegin(CLIENT_KNOBS->VALUE_SIZE_LIMIT, 'a'); + const std::string oversizedEnd(CLIENT_KNOBS->VALUE_SIZE_LIMIT, 'b'); + expectInvalid("orders"_sr, { KeyRangeRef(oversizedBegin, oversizedEnd) }); + return Void(); +} + TEST_CASE("/NativeCDC/LifecycleAllocation") { ASSERT(!validNativeCdcTagCount(-1)); ASSERT(!validNativeCdcTagCount(0)); diff --git a/fdbclient/NativeCdcInternal.h b/fdbclient/NativeCdcInternal.h index ae52ee32fe1..3acd6b919f5 100644 --- a/fdbclient/NativeCdcInternal.h +++ b/fdbclient/NativeCdcInternal.h @@ -27,7 +27,7 @@ // Durable metadata operations used by CDC server roles. Registration is // feature gated; drain and cleanup operations remain available for streams // persisted before native CDC is disabled. -Future registerNativeCdcStream(Database cx, Key name, KeyRange keys, UID proxyId); +Future registerNativeCdcStream(Database cx, Key name, std::vector ranges, UID proxyId); // Persists per-tag final-pop watermarks before removing stream metadata. Future removeNativeCdcStream(Database cx, Key name, CDCStreamId streamId, UID proxyId); Future> listNativeCdcStreams(Database cx); diff --git a/fdbclient/SystemData.cpp b/fdbclient/SystemData.cpp index 0ea881e10d4..b838149bda5 100644 --- a/fdbclient/SystemData.cpp +++ b/fdbclient/SystemData.cpp @@ -839,18 +839,18 @@ CDCStreamId decodeCDCStreamKey(KeyRef const& key) { return streamId; } -Value cdcStreamKeysValue(KeyRangeRef const& keys) { +Value cdcStreamKeysValue(std::vector const& ranges) { BinaryWriter wr(IncludeVersion(ProtocolVersion::withNativeCdc())); - wr << keys; + wr << ranges; return wr.toValue(); } -KeyRange decodeCDCStreamKeysValue(ValueRef const& value) { - KeyRange keys; +std::vector decodeCDCStreamKeysValue(ValueRef const& value) { + std::vector ranges; BinaryReader reader(value, IncludeVersion()); ASSERT_WE_THINK(reader.protocolVersion().hasNativeCdc()); - reader >> keys; - return keys; + reader >> ranges; + return ranges; } static Key cdcTagHistoryPrefixFor(CDCStreamId streamId) { @@ -1944,7 +1944,7 @@ TEST_CASE("noSim/SystemData/DataMoveId") { TEST_CASE("/SystemData/NativeCDC") { const Key name = "orders"_sr; const CDCStreamId streamId = 42; - const KeyRange keys(KeyRangeRef("a"_sr, "z"_sr)); + const std::vector ranges{ KeyRangeRef("a"_sr, "c"_sr), KeyRangeRef("x"_sr, "z"_sr) }; const Version minVersion = 123456789; const Tag tag(tagLocalityCDC, 9); const UID proxyId(1, 2); @@ -1953,7 +1953,7 @@ TEST_CASE("/SystemData/NativeCDC") { ASSERT_EQ(decodeCDCStreamNameValue(cdcStreamNameValue(streamId)), streamId); ASSERT_EQ(decodeCDCMaxStreamIdValue(cdcMaxStreamIdValue(streamId)), streamId); ASSERT_EQ(decodeCDCStreamKey(cdcStreamKeyFor(streamId)), streamId); - ASSERT_EQ(decodeCDCStreamKeysValue(cdcStreamKeysValue(keys)), keys); + ASSERT(decodeCDCStreamKeysValue(cdcStreamKeysValue(ranges)) == ranges); const Key tagOwnerKey = cdcTagOwnerKeyFor(tag); ASSERT_EQ(decodeCDCTagOwnerKey(tagOwnerKey), tag); ASSERT(cdcTagOwnerKeys.contains(tagOwnerKey)); diff --git a/fdbclient/ThreadSafeTransaction.cpp b/fdbclient/ThreadSafeTransaction.cpp index 82428c4ad29..98272a25016 100644 --- a/fdbclient/ThreadSafeTransaction.cpp +++ b/fdbclient/ThreadSafeTransaction.cpp @@ -221,13 +221,13 @@ ThreadFuture ThreadSafeDatabase::createSnapshot(const StringRef& uid, cons }); } -ThreadFuture ThreadSafeDatabase::registerNativeCdcStream(const KeyRef& name, const KeyRangeRef& keys) { +ThreadFuture ThreadSafeDatabase::registerNativeCdcStream(const KeyRef& name, + const std::vector& ranges) { DatabaseContext* db = this->db; Key nameCopy(name); - KeyRange keysCopy(keys); - return onMainThread([db, nameCopy, keysCopy]() -> Future { + return onMainThread([db, nameCopy, ranges]() -> Future { db->checkDeferredError(); - return registerNativeCdcStreamClient(Database(Reference::addRef(db)), nameCopy, keysCopy); + return registerNativeCdcStreamClient(Database(Reference::addRef(db)), nameCopy, ranges); }); } diff --git a/fdbclient/include/fdbclient/CDCProxyInterface.h b/fdbclient/include/fdbclient/CDCProxyInterface.h index d5be4e2e811..6b577627567 100644 --- a/fdbclient/include/fdbclient/CDCProxyInterface.h +++ b/fdbclient/include/fdbclient/CDCProxyInterface.h @@ -72,17 +72,17 @@ struct CDCRegisterStreamReply { struct CDCRegisterStreamRequest { constexpr static FileIdentifier file_identifier = 1269096; Key name; - KeyRange keys; + std::vector ranges; ReplyPromise reply; CDCRegisterStreamRequest() = default; - CDCRegisterStreamRequest(Key name, KeyRange keys) : name(name), keys(keys) {} + CDCRegisterStreamRequest(Key name, std::vector ranges) : name(name), ranges(std::move(ranges)) {} bool verify() const { return true; } template void serialize(Ar& ar) { - serializer(ar, name, keys, reply); + serializer(ar, name, ranges, reply); } }; diff --git a/fdbclient/include/fdbclient/IClientApi.h b/fdbclient/include/fdbclient/IClientApi.h index c669966346b..ce44a8ee9ed 100644 --- a/fdbclient/include/fdbclient/IClientApi.h +++ b/fdbclient/include/fdbclient/IClientApi.h @@ -157,7 +157,8 @@ class IDatabase { // Native CDC operations. These values are intentionally independent from // NativeAPI so multi-version client wrappers can forward them without // depending on the native client implementation. - virtual ThreadFuture registerNativeCdcStream(const KeyRef& name, const KeyRangeRef& keys) = 0; + virtual ThreadFuture registerNativeCdcStream(const KeyRef& name, + const std::vector& ranges) = 0; virtual ThreadFuture removeNativeCdcStream(const KeyRef& name) = 0; virtual ThreadFuture> listNativeCdcStreams() = 0; virtual ThreadFuture> createNativeCdcConsumer(const KeyRef& name) = 0; diff --git a/fdbclient/include/fdbclient/MultiVersionTransaction.h b/fdbclient/include/fdbclient/MultiVersionTransaction.h index 084a5697af4..a6dd405a5df 100644 --- a/fdbclient/include/fdbclient/MultiVersionTransaction.h +++ b/fdbclient/include/fdbclient/MultiVersionTransaction.h @@ -95,7 +95,8 @@ struct FdbCApi : public ThreadSafeReferenceCounted { using FDBNativeCdcStreamInfo = struct native_cdc_stream_info { FDBKey name; uint64_t streamId; - FDBKeyRange keyRange; + const FDBKeyRange* ranges; + int rangeCount; int64_t minVersion; }; @@ -157,10 +158,8 @@ struct FdbCApi : public ThreadSafeReferenceCounted { FDBFuture* (*databaseRegisterNativeCdcStream)(FDBDatabase* database, uint8_t const* name, int nameLength, - uint8_t const* beginKey, - int beginKeyLength, - uint8_t const* endKey, - int endKeyLength); + FDBKeyRange const* ranges, + int rangeCount); FDBFuture* (*databaseRemoveNativeCdcStream)(FDBDatabase* database, uint8_t const* name, int nameLength); FDBFuture* (*databaseListNativeCdcStreams)(FDBDatabase* database); FDBFuture* (*databaseCreateNativeCdcConsumer)(FDBDatabase* database, uint8_t const* name, int nameLength); @@ -433,7 +432,7 @@ class DLDatabase : public IDatabase, ThreadSafeReferenceCounted { ThreadFuture rebootWorker(const StringRef& address, bool check, int duration) override; ThreadFuture forceRecoveryWithDataLoss(const StringRef& dcid) override; ThreadFuture createSnapshot(const StringRef& uid, const StringRef& snapshot_command) override; - ThreadFuture registerNativeCdcStream(const KeyRef& name, const KeyRangeRef& keys) override; + ThreadFuture registerNativeCdcStream(const KeyRef& name, const std::vector& ranges) override; ThreadFuture removeNativeCdcStream(const KeyRef& name) override; ThreadFuture> listNativeCdcStreams() override; ThreadFuture> createNativeCdcConsumer(const KeyRef& name) override; @@ -751,7 +750,7 @@ class MultiVersionDatabase final : public IDatabase, ThreadSafeReferenceCounted< ThreadFuture rebootWorker(const StringRef& address, bool check, int duration) override; ThreadFuture forceRecoveryWithDataLoss(const StringRef& dcid) override; ThreadFuture createSnapshot(const StringRef& uid, const StringRef& snapshot_command) override; - ThreadFuture registerNativeCdcStream(const KeyRef& name, const KeyRangeRef& keys) override; + ThreadFuture registerNativeCdcStream(const KeyRef& name, const std::vector& ranges) override; ThreadFuture removeNativeCdcStream(const KeyRef& name) override; ThreadFuture> listNativeCdcStreams() override; ThreadFuture> createNativeCdcConsumer(const KeyRef& name) override; diff --git a/fdbclient/include/fdbclient/NativeCdc.h b/fdbclient/include/fdbclient/NativeCdc.h index 4cd2e886124..6cb4de013d2 100644 --- a/fdbclient/include/fdbclient/NativeCdc.h +++ b/fdbclient/include/fdbclient/NativeCdc.h @@ -53,10 +53,12 @@ class NativeCdcConsumer : public ReferenceCounted { // registration and the remaining operations stay available so existing durable // streams can be drained after the feature is disabled. Requests retry when // stream ownership changes. -Future registerNativeCdcStreamClient(Database cx, Key name, KeyRange keys); +// Ranges form an immutable union. Registration normalizes overlap and adjacency +// so equivalent range sets have the same identity regardless of input order. +Future registerNativeCdcStreamClient(Database cx, Key name, std::vector ranges); Future removeNativeCdcStreamClient(Database cx, Key name); Future> listNativeCdcStreamsClient(Database cx); -// Uses the range registered for this name; consumers do not respecify it. A +// Uses the ranges registered for this name; consumers do not respecify them. A // CDCCursor remains a serializable position token and does not hold Database. Future> createNativeCdcConsumer(Database cx, Key name); Reference resumeNativeCdcConsumer(Database cx, CDCCursor position); diff --git a/fdbclient/include/fdbclient/NativeCdcClient.h b/fdbclient/include/fdbclient/NativeCdcClient.h index 7b166bfd6f1..481a08bd586 100644 --- a/fdbclient/include/fdbclient/NativeCdcClient.h +++ b/fdbclient/include/fdbclient/NativeCdcClient.h @@ -31,10 +31,12 @@ // Native CDC value types shared by thread-safe client surfaces and language // bindings. Keep this header independent from NativeAPI so multi-version // client plumbing does not depend on the native client implementation. +constexpr int NATIVE_CDC_MAX_RANGES = 1024; + struct NativeCdcStreamInfo { Key name; CDCStreamId streamId = 0; - KeyRange keys; + std::vector ranges; Version minVersion = invalidVersion; }; diff --git a/fdbclient/include/fdbclient/SystemData.h b/fdbclient/include/fdbclient/SystemData.h index 02c09cd92bd..6861d23d231 100644 --- a/fdbclient/include/fdbclient/SystemData.h +++ b/fdbclient/include/fdbclient/SystemData.h @@ -293,12 +293,12 @@ extern const KeyRef cdcMaxStreamIdKey; Value cdcMaxStreamIdValue(CDCStreamId streamId); CDCStreamId decodeCDCMaxStreamIdValue(ValueRef const& value); -// "\xff/cdc/keys/[[CDCStreamId]]" := "[[KeyRange]]" +// "\xff/cdc/keys/[[CDCStreamId]]" := "[[vector]]" extern const KeyRangeRef cdcStreamKeys; Key cdcStreamKeyFor(CDCStreamId streamId); CDCStreamId decodeCDCStreamKey(KeyRef const& key); -Value cdcStreamKeysValue(KeyRangeRef const& keys); -KeyRange decodeCDCStreamKeysValue(ValueRef const& value); +Value cdcStreamKeysValue(std::vector const& ranges); +std::vector decodeCDCStreamKeysValue(ValueRef const& value); // "\xff/cdc/tagHistory/[[CDCStreamId]][[Version]][[Tag]]" := "" struct CDCTagHistoryEntry { diff --git a/fdbclient/include/fdbclient/ThreadSafeTransaction.h b/fdbclient/include/fdbclient/ThreadSafeTransaction.h index 6a310631637..6a3d76190aa 100644 --- a/fdbclient/include/fdbclient/ThreadSafeTransaction.h +++ b/fdbclient/include/fdbclient/ThreadSafeTransaction.h @@ -58,7 +58,7 @@ class ThreadSafeDatabase : public IDatabase, public ThreadSafeReferenceCounted forceRecoveryWithDataLoss(const StringRef& dcid) override; ThreadFuture createSnapshot(const StringRef& uid, const StringRef& snapshot_command) override; - ThreadFuture registerNativeCdcStream(const KeyRef& name, const KeyRangeRef& keys) override; + ThreadFuture registerNativeCdcStream(const KeyRef& name, const std::vector& ranges) override; ThreadFuture removeNativeCdcStream(const KeyRef& name) override; ThreadFuture> listNativeCdcStreams() override; ThreadFuture> createNativeCdcConsumer(const KeyRef& name) override; diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index d3fbba680f6..67d0e898ad6 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -52,7 +52,7 @@ namespace { // Snapshot from one durable metadata read, used while initializing or validating one stream. struct CDCStreamReadState { - Optional keys; + Optional> ranges; Version minVersion = invalidVersion; Version readVersion = invalidVersion; // Each Version is the inclusive lower bound for log versions routed to its paired tag; the next entry's @@ -74,7 +74,7 @@ struct CDCTagInterval { // Proxy-owned state for one assigned stream. In-flight actors may retain it after active becomes false. struct CDCBufferedStream : ReferenceCounted { CDCStreamId streamId; - Optional keys; + Optional> ranges; bool active = true; bool initialized = false; bool initializationPausedForTesting = false; @@ -416,26 +416,32 @@ class CDCProxy { Future run(CDCProxyInterface proxy, uint64_t recoveryCount); }; -Optional clipCDCMutation(MutationRef const& mutation, KeyRangeRef const& keys) { +template +void visitClippedCDCMutations(MutationRef const& mutation, std::vector const& ranges, Visitor&& visitor) { + // Canonical stream ranges are ordered and disjoint, so their ends are strictly increasing. + auto range = std::upper_bound(ranges.begin(), ranges.end(), mutation.param1, [](KeyRef key, KeyRange const& range) { + return key < range.end; + }); if (isSingleKeyMutation((MutationRef::Type)mutation.type)) { - if (keys.contains(mutation.param1)) { - return mutation; + if (range != ranges.end() && range->contains(mutation.param1)) { + visitor(mutation); } } else if (mutation.type == MutationRef::ClearRange) { - KeyRangeRef intersection = keys & KeyRangeRef(mutation.param1, mutation.param2); - if (!intersection.empty()) { - return MutationRef(MutationRef::ClearRange, intersection.begin, intersection.end); + for (; range != ranges.end() && range->begin < mutation.param2; ++range) { + const KeyRangeRef intersection = *range & KeyRangeRef(mutation.param1, mutation.param2); + if (!intersection.empty()) { + visitor(MutationRef(MutationRef::ClearRange, intersection.begin, intersection.end)); + } } } else { ASSERT(false); } - return Optional(); } Future readCDCStreamState(Database cx, CDCStreamId streamId, UID expectedProxyId, - bool requireKeys) { + bool requireRanges) { if (streamId == 0) { throw client_invalid_operation(); } @@ -447,17 +453,17 @@ Future readCDCStreamState(Database cx, tr.setOption(FDBTransactionOptions::READ_LOCK_AWARE); tr.setOption(FDBTransactionOptions::READ_SYSTEM_KEYS); - Future> keysFuture = tr.get(cdcStreamKeyFor(streamId)); + Future> rangesFuture = tr.get(cdcStreamKeyFor(streamId)); Future> minVersionFuture = tr.get(cdcMinVersionKeyFor(streamId)); Future assignedProxiesFuture = tr.getRange(cdcProxyRangeFor(streamId), 2); KeyRange tagHistoryRange = cdcTagHistoryRangeFor(streamId); Future historyFuture = tr.getRange(tagHistoryRange, CLIENT_KNOBS->TOO_MANY); CDCStreamReadState result; - Optional keysValue = co_await keysFuture; - if (keysValue.present()) { - result.keys = decodeCDCStreamKeysValue(keysValue.get()); - } else if (requireKeys) { + Optional rangesValue = co_await rangesFuture; + if (rangesValue.present()) { + result.ranges = decodeCDCStreamKeysValue(rangesValue.get()); + } else if (requireRanges) { throw client_invalid_operation(); } @@ -842,7 +848,7 @@ void CDCProxy::visitBufferedMutations(Reference tag, } auto stream = streams.find(streamId); if (stream == streams.end() || !stream->second->active || stream->second->readDemand == 0 || - !stream->second->keys.present()) { + !stream->second->ranges.present()) { continue; } const bool coversVersion = @@ -855,10 +861,9 @@ void CDCProxy::visitBufferedMutations(Reference tag, if (!coversVersion) { continue; } - Optional clipped = clipCDCMutation(mutation, stream->second->keys.get()); - if (clipped.present()) { - visitor(stream->second, messageVersion, clipped.get()); - } + visitClippedCDCMutations(mutation, stream->second->ranges.get(), [&](MutationRef const& clipped) { + visitor(stream->second, messageVersion, clipped); + }); } } cursor->nextMessage(); @@ -1172,7 +1177,7 @@ Future CDCProxy::initializeStream(Reference stream) { CODE_PROBE(true, "CDC proxy discards stale stream initialization"); co_return; } - stream->keys = metadata.keys; + stream->ranges = metadata.ranges; stream->minVersion = metadata.minVersion; stream->bufferedThrough = metadata.minVersion - 1; for (size_t i = 0; i < metadata.tagAssignments.size(); ++i) { @@ -1629,7 +1634,7 @@ Future CDCProxy::acknowledge(CDCAckRequest request) { Future CDCProxy::registerStream(CDCRegisterStreamRequest request) { try { - const CDCStreamId streamId = co_await registerNativeCdcStream(cx, request.name, request.keys, id); + const CDCStreamId streamId = co_await registerNativeCdcStream(cx, request.name, request.ranges, id); request.reply.send(CDCRegisterStreamReply(streamId)); } catch (Error& e) { if (e.code() == error_code_actor_cancelled) { @@ -1887,23 +1892,83 @@ Future cdcProxyServer(CDCProxyInterface proxy, } TEST_CASE("/NativeCDC/ProxyMutationFiltering") { - const KeyRangeRef keys("c"_sr, "m"_sr); - - Optional inRange = clipCDCMutation(MutationRef(MutationRef::SetValue, "d"_sr, "value"_sr), keys); - ASSERT(inRange.present()); - ASSERT_EQ(inRange.get().param1, "d"_sr); + const std::vector ranges{ KeyRangeRef("c"_sr, "m"_sr), KeyRangeRef("q"_sr, "t"_sr) }; + std::vector filtered; + auto collect = [&filtered](MutationRef const& mutation) { filtered.push_back(mutation); }; + + for (const KeyRef key : { "c"_sr, "d"_sr, "q"_sr, "s"_sr }) { + filtered.clear(); + visitClippedCDCMutations(MutationRef(MutationRef::SetValue, key, "value"_sr), ranges, collect); + ASSERT_EQ(filtered.size(), 1); + ASSERT_EQ(filtered.front().param1, key); + } + for (const KeyRef key : { "a"_sr, "m"_sr, "n"_sr, "t"_sr, "z"_sr }) { + filtered.clear(); + visitClippedCDCMutations(MutationRef(MutationRef::SetValue, key, "value"_sr), ranges, collect); + ASSERT(filtered.empty()); + } + + visitClippedCDCMutations(MutationRef(MutationRef::ClearRange, "a"_sr, "r"_sr), ranges, collect); + ASSERT_EQ(filtered.size(), 2); + ASSERT_EQ(filtered[0].param1, "c"_sr); + ASSERT_EQ(filtered[0].param2, "m"_sr); + ASSERT_EQ(filtered[1].param1, "q"_sr); + ASSERT_EQ(filtered[1].param2, "r"_sr); + + filtered.clear(); + visitClippedCDCMutations(MutationRef(MutationRef::ClearRange, "l"_sr, "z"_sr), ranges, collect); + ASSERT_EQ(filtered.size(), 2); + ASSERT_EQ(filtered[0].param1, "l"_sr); + ASSERT_EQ(filtered[0].param2, "m"_sr); + ASSERT_EQ(filtered[1].param1, "q"_sr); + ASSERT_EQ(filtered[1].param2, "t"_sr); + + filtered.clear(); + visitClippedCDCMutations(MutationRef(MutationRef::ClearRange, "m"_sr, "q"_sr), ranges, collect); + visitClippedCDCMutations(MutationRef(MutationRef::ClearRange, "d"_sr, "d"_sr), ranges, collect); + visitClippedCDCMutations(MutationRef(MutationRef::ClearRange, "t"_sr, "z"_sr), ranges, collect); + ASSERT(filtered.empty()); - Optional outOfRange = clipCDCMutation(MutationRef(MutationRef::SetValue, "z"_sr, "value"_sr), keys); - ASSERT(!outOfRange.present()); - - Optional clippedClear = clipCDCMutation(MutationRef(MutationRef::ClearRange, "a"_sr, "f"_sr), keys); - ASSERT(clippedClear.present()); - ASSERT_EQ(clippedClear.get().param1, "c"_sr); - ASSERT_EQ(clippedClear.get().param2, "f"_sr); - - Optional excludedClear = clipCDCMutation(MutationRef(MutationRef::ClearRange, "n"_sr, "z"_sr), keys); - ASSERT(!excludedClear.present()); + return Void(); +} +TEST_CASE("/NativeCDC/ProxyMutationFiltering/MultiRangeBatch") { + const std::vector ranges{ KeyRangeRef("a"_sr, "c"_sr), KeyRangeRef("x"_sr, "z"_sr) }; + auto stream = makeReference(1); + CDCBufferedBatch batch; + const std::vector input{ MutationRef(MutationRef::SetValue, "b"_sr, "before"_sr), + MutationRef(MutationRef::ClearRange, "b"_sr, "y"_sr), + MutationRef(MutationRef::SetValue, "x"_sr, "after"_sr), + MutationRef(MutationRef::SetValue, "m"_sr, "gap"_sr) }; + for (const auto& mutation : input) { + visitClippedCDCMutations( + mutation, ranges, [&](MutationRef const& clipped) { addMutationToBatch(stream, &batch, 100, clipped); }); + } + ASSERT_EQ(batch.mutations.size(), 1); + const auto& versioned = batch.mutations.front(); + ASSERT_EQ(versioned.version, 100); + ASSERT_EQ(versioned.mutations.size(), 4); + const std::vector expected{ input[0], + MutationRef(MutationRef::ClearRange, "b"_sr, "c"_sr), + MutationRef(MutationRef::ClearRange, "x"_sr, "y"_sr), + input[2] }; + int64_t expectedBytes = sizeof(VersionedMutationsRef); + for (int i = 0; i < versioned.mutations.size(); ++i) { + ASSERT_EQ(versioned.mutations[i].type, expected[i].type); + ASSERT_EQ(versioned.mutations[i].param1, expected[i].param1); + ASSERT_EQ(versioned.mutations[i].param2, expected[i].param2); + expectedBytes += expected[i].expectedSize() + sizeof(MutationRef); + } + ASSERT_EQ(batch.bufferedBytes, expectedBytes); + + const int64_t versionBytes = estimatedCDCConsumeVersionBytes(versioned); + CDCConsumeReplySelection tooSmall; + ASSERT(!selectCDCConsumeReplyVersion(&tooSmall, 100, 100, versionBytes, versionBytes - 1)); + ASSERT(tooSmall.firstVersionTooLarge); + ASSERT_EQ(selectedCDCConsumeReplyThrough(tooSmall, 100), 99); + CDCConsumeReplySelection exactFit; + ASSERT(selectCDCConsumeReplyVersion(&exactFit, 100, 100, versionBytes, versionBytes)); + ASSERT_EQ(selectedCDCConsumeReplyThrough(exactFit, 100), 100); return Void(); } diff --git a/fdbserver/logsystem/ApplyMetadataMutation.cpp b/fdbserver/logsystem/ApplyMetadataMutation.cpp index ccc031bd3b2..fe2c60f5010 100644 --- a/fdbserver/logsystem/ApplyMetadataMutation.cpp +++ b/fdbserver/logsystem/ApplyMetadataMutation.cpp @@ -573,7 +573,7 @@ class ApplyMetadataMutationsImpl { return; } if (cdcStreamKeys.contains(m.param1)) { - cdcRouting->setRange(decodeCDCStreamKey(m.param1), decodeCDCStreamKeysValue(m.param2)); + cdcRouting->setRanges(decodeCDCStreamKey(m.param1), decodeCDCStreamKeysValue(m.param2)); } else if (cdcTagHistoryKeys.contains(m.param1)) { const CDCTagHistoryEntry history = decodeCDCTagHistoryKey(m.param1); cdcRouting->setTag(history.streamId, history.version, history.tag); diff --git a/fdbserver/logsystem/CDCRoutingTable.cpp b/fdbserver/logsystem/CDCRoutingTable.cpp index 240469352d6..d720adb987d 100644 --- a/fdbserver/logsystem/CDCRoutingTable.cpp +++ b/fdbserver/logsystem/CDCRoutingTable.cpp @@ -28,8 +28,8 @@ CDCRoutingTable::CDCRoutingTable() { tagsByRange.insert(allKeys, std::set()); } -void CDCRoutingTable::updateRange(CDCStreamId streamId, KeyRangeRef const& keys) { - streams[streamId].keys = KeyRange(keys); +void CDCRoutingTable::updateRanges(CDCStreamId streamId, std::vector const& ranges) { + streams[streamId].ranges = ranges; } bool CDCRoutingTable::updateTag(CDCStreamId streamId, Version version, Tag tag) { @@ -45,18 +45,20 @@ bool CDCRoutingTable::updateTag(CDCStreamId streamId, Version version, Tag tag) void CDCRoutingTable::rebuildRanges() { tagsByRange.insert(allKeys, std::set()); for (const auto& [streamId, state] : streams) { - if (!state.keys.present() || !state.tag.present()) { + if (!state.tag.present()) { continue; } - for (auto range : tagsByRange.modify(state.keys.get())) { - range->value().insert(state.tag.get().second); + for (const auto& keys : state.ranges) { + for (auto range : tagsByRange.modify(keys)) { + range->value().insert(state.tag.get().second); + } } } tagsByRange.coalesce(allKeys); } -void CDCRoutingTable::setRange(CDCStreamId streamId, KeyRangeRef const& keys) { - updateRange(streamId, keys); +void CDCRoutingTable::setRanges(CDCStreamId streamId, std::vector const& ranges) { + updateRanges(streamId, ranges); rebuildRanges(); } @@ -70,7 +72,7 @@ void CDCRoutingTable::reload(IKeyValueStore* txnStateStore) { streams.clear(); const RangeResult streamRows = txnStateStore->readRange(cdcStreamKeys).get(); for (const auto& kv : streamRows) { - updateRange(decodeCDCStreamKey(kv.key), decodeCDCStreamKeysValue(kv.value)); + updateRanges(decodeCDCStreamKey(kv.key), decodeCDCStreamKeysValue(kv.value)); } const RangeResult tagHistoryRows = txnStateStore->readRange(cdcTagHistoryKeys).get(); for (const auto& kv : tagHistoryRows) { @@ -101,9 +103,9 @@ TEST_CASE("/NativeCDC/RoutingTable") { ASSERT(table.tagsForKey("b"_sr).empty()); ASSERT(table.tagsForRange(KeyRangeRef("b"_sr, "x"_sr)).empty()); - table.setRange(1, KeyRangeRef("a"_sr, "m"_sr)); + table.setRanges(1, { KeyRangeRef("a"_sr, "m"_sr) }); table.setTag(1, 100, ordersTag); - table.setRange(2, KeyRangeRef("g"_sr, "z"_sr)); + table.setRanges(2, { KeyRangeRef("g"_sr, "z"_sr) }); table.setTag(2, 100, overlappingTag); ASSERT_EQ(table.tagsForKey("b"_sr), std::set{ ordersTag }); @@ -125,14 +127,14 @@ TEST_CASE("/NativeCDC/RoutingTable/MetadataOrdering") { const Tag staleTag(tagLocalityCDC, 3); const Tag replacementTag(tagLocalityCDC, 4); - table.setRange(1, KeyRangeRef("a"_sr, "m"_sr)); + table.setRanges(1, { KeyRangeRef("a"_sr, "m"_sr) }); ASSERT(table.tagsForKey("b"_sr).empty()); table.setTag(2, 100, tagFirstTag); ASSERT(table.tagsForKey("n"_sr).empty()); table.setTag(1, 100, rangeFirstTag); - table.setRange(2, KeyRangeRef("m"_sr, "z"_sr)); + table.setRanges(2, { KeyRangeRef("m"_sr, "z"_sr) }); ASSERT_EQ(table.tagsForKey("b"_sr), std::set{ rangeFirstTag }); ASSERT_EQ(table.tagsForKey("n"_sr), std::set{ tagFirstTag }); @@ -149,18 +151,49 @@ TEST_CASE("/NativeCDC/RoutingTable/SharedTagRangeReplacement") { CDCRoutingTable table; const Tag sharedTag(tagLocalityCDC, 1); - table.setRange(1, KeyRangeRef("a"_sr, "m"_sr)); + table.setRanges(1, { KeyRangeRef("a"_sr, "m"_sr) }); table.setTag(1, 100, sharedTag); - table.setRange(2, KeyRangeRef("g"_sr, "z"_sr)); + table.setRanges(2, { KeyRangeRef("g"_sr, "z"_sr) }); table.setTag(2, 100, sharedTag); ASSERT_EQ(table.tagsForKey("h"_sr), std::set{ sharedTag }); ASSERT_EQ(table.tagsForRange(KeyRangeRef("a"_sr, "z"_sr)), std::set{ sharedTag }); - table.setRange(1, KeyRangeRef("n"_sr, "t"_sr)); + table.setRanges(1, { KeyRangeRef("n"_sr, "t"_sr) }); ASSERT(table.tagsForKey("b"_sr).empty()); ASSERT_EQ(table.tagsForKey("h"_sr), std::set{ sharedTag }); ASSERT_EQ(table.tagsForKey("p"_sr), std::set{ sharedTag }); return Void(); } + +TEST_CASE("/NativeCDC/RoutingTable/DisjointRanges") { + CDCRoutingTable table; + const Tag sharedTag(tagLocalityCDC, 1); + const Tag gapTag(tagLocalityCDC, 2); + const Tag rotatedTag(tagLocalityCDC, 3); + + table.setRanges(1, { KeyRangeRef("a"_sr, "c"_sr), KeyRangeRef("x"_sr, "z"_sr) }); + table.setTag(1, 100, sharedTag); + ASSERT_EQ(table.tagsForKey("a"_sr), std::set{ sharedTag }); + ASSERT_EQ(table.tagsForKey("x"_sr), std::set{ sharedTag }); + ASSERT(table.tagsForKey("c"_sr).empty()); + ASSERT(table.tagsForKey("m"_sr).empty()); + ASSERT(table.tagsForKey("z"_sr).empty()); + ASSERT(table.tagsForRange(KeyRangeRef("c"_sr, "x"_sr)).empty()); + ASSERT_EQ(table.tagsForRange(KeyRangeRef("b"_sr, "y"_sr)), std::set{ sharedTag }); + + table.setRanges(2, { KeyRangeRef("b"_sr, "d"_sr) }); + table.setTag(2, 100, sharedTag); + table.setRanges(3, { KeyRangeRef("m"_sr, "q"_sr) }); + table.setTag(3, 100, gapTag); + ASSERT_EQ(table.tagsForKey("b"_sr), std::set{ sharedTag }); + ASSERT_EQ(table.tagsForRange(KeyRangeRef("b"_sr, "y"_sr)), (std::set{ sharedTag, gapTag })); + + table.setTag(1, 200, rotatedTag); + ASSERT_EQ(table.tagsForKey("a"_sr), std::set{ rotatedTag }); + ASSERT_EQ(table.tagsForKey("b"_sr), (std::set{ sharedTag, rotatedTag })); + ASSERT_EQ(table.tagsForKey("x"_sr), std::set{ rotatedTag }); + ASSERT_EQ(table.tagsForRange(KeyRangeRef("b"_sr, "y"_sr)), (std::set{ sharedTag, gapTag, rotatedTag })); + return Void(); +} diff --git a/fdbserver/logsystem/include/fdbserver/logsystem/CDCRoutingTable.h b/fdbserver/logsystem/include/fdbserver/logsystem/CDCRoutingTable.h index 6fd8be3f19e..2d9dbb90685 100644 --- a/fdbserver/logsystem/include/fdbserver/logsystem/CDCRoutingTable.h +++ b/fdbserver/logsystem/include/fdbserver/logsystem/CDCRoutingTable.h @@ -23,6 +23,7 @@ #include #include #include +#include #include "fdbclient/FDBTypes.h" #include "fdbclient/KeyRangeMap.h" @@ -32,20 +33,20 @@ class IKeyValueStore; // Active CDC write routing reconstructed from durable stream and tag-history metadata. class CDCRoutingTable : NonCopyable { struct StreamState { - Optional keys; + std::vector ranges; Optional> tag; }; std::unordered_map streams; KeyRangeMap> tagsByRange; - void updateRange(CDCStreamId streamId, KeyRangeRef const& keys); + void updateRanges(CDCStreamId streamId, std::vector const& ranges); bool updateTag(CDCStreamId streamId, Version version, Tag tag); void rebuildRanges(); public: CDCRoutingTable(); - void setRange(CDCStreamId streamId, KeyRangeRef const& keys); + void setRanges(CDCStreamId streamId, std::vector const& ranges); void setTag(CDCStreamId streamId, Version version, Tag tag); void reload(IKeyValueStore* txnStateStore); bool empty() const { return streams.empty(); } diff --git a/fdbserver/workloads/NativeCdcEndToEnd.cpp b/fdbserver/workloads/NativeCdcEndToEnd.cpp index 0d8cb777f28..31dcf6d909a 100644 --- a/fdbserver/workloads/NativeCdcEndToEnd.cpp +++ b/fdbserver/workloads/NativeCdcEndToEnd.cpp @@ -73,6 +73,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { bool injectUndeliveredProxyHalt; bool testMemoryBound; bool testReplyChunking; + bool testMultipleRanges; bool testOversizedPeek; bool testDurableAckScan; bool testDelayedRetention; @@ -224,7 +225,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { StreamState stream; stream.name = Key(StringRef(format("native-cdc-e2e/stream/%04d", nextStreamNumber++))); stream.keys = std::move(keys); - co_await timeoutError(registerNativeCdcStreamClient(cx, stream.name, stream.keys), operationTimeout); + co_await timeoutError(registerNativeCdcStreamClient(cx, stream.name, { stream.keys }), operationTimeout); stream.consumer = co_await timeoutError(createNativeCdcConsumer(cx, stream.name), operationTimeout); streams.push_back(std::move(stream)); } @@ -260,12 +261,12 @@ class NativeCdcEndToEndWorkload : public TestWorkload { const KeyRange conflictingKeys(KeyRangeRef("native-cdc-e2e/lifecycle/"_sr, "native-cdc-e2e/lifecycle1"_sr)); const CDCStreamId streamId = - co_await timeoutError(registerNativeCdcStreamClient(cx, name, keys), operationTimeout); - ASSERT_EQ(co_await timeoutError(registerNativeCdcStreamClient(cx, name, keys), operationTimeout), streamId); + co_await timeoutError(registerNativeCdcStreamClient(cx, name, { keys }), operationTimeout); + ASSERT_EQ(co_await timeoutError(registerNativeCdcStreamClient(cx, name, { keys }), operationTimeout), streamId); bool conflictingRegistrationRejected = false; try { - co_await timeoutError(registerNativeCdcStreamClient(cx, name, conflictingKeys), operationTimeout); + co_await timeoutError(registerNativeCdcStreamClient(cx, name, { conflictingKeys }), operationTimeout); } catch (Error& e) { if (e.code() != error_code_client_invalid_operation) { throw; @@ -280,7 +281,8 @@ class NativeCdcEndToEndWorkload : public TestWorkload { listed.begin(), listed.end(), [&](NativeCdcStreamInfo const& stream) { return stream.name == name; }); ASSERT_EQ(found != listed.end(), true); ASSERT_EQ(found->streamId, streamId); - ASSERT_EQ(found->keys, keys); + ASSERT_EQ(found->ranges.size(), 1); + ASSERT_EQ(found->ranges.front(), keys); bool futureConsumeRejected = false; try { @@ -344,7 +346,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { const KeyRange expectedLower(KeyRangeRef(keys.begin, lowerClear.end)); const KeyRange expectedUpper(KeyRangeRef(upperClear.begin, keys.end)); - co_await timeoutError(registerNativeCdcStreamClient(cx, name, keys), operationTimeout); + co_await timeoutError(registerNativeCdcStreamClient(cx, name, { keys }), operationTimeout); Reference consumer = co_await timeoutError(createNativeCdcConsumer(cx, name), operationTimeout); @@ -399,7 +401,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { co_await delay(0.1); const CDCStreamId streamId = - co_await timeoutError(registerNativeCdcStreamClient(cx, name, keys), operationTimeout); + co_await timeoutError(registerNativeCdcStreamClient(cx, name, { keys }), operationTimeout); Reference consumer = co_await timeoutError(createNativeCdcConsumer(cx, name), operationTimeout); ASSERT_EQ(consumer->position().streamId, streamId); @@ -409,6 +411,181 @@ class NativeCdcEndToEndWorkload : public TestWorkload { co_await timeoutError(removeNativeCdcStreamClient(cx, name), operationTimeout); } + Key multipleRangeKey(StringRef suffix) const { + return suffix.withPrefix("native-cdc-e2e/multiple-ranges/data/"_sr); + } + + KeyRange multipleRangeKeys(StringRef begin, StringRef end) const { + return KeyRange(KeyRangeRef(multipleRangeKey(begin), multipleRangeKey(end))); + } + + Future consumeExpectedVersion(Reference consumer, + Version committed, + Standalone> expected) { + const double deadline = now() + operationTimeout; + bool observed = false; + while (!observed) { + ASSERT_LT(now(), deadline); + CDCConsumeReply reply = co_await timeoutError(consumer->consume(), deadline - now()); + for (const auto& versioned : reply.mutations) { + ASSERT_LE(versioned.version, reply.lastConsumedVersion); + if (versioned.version != committed) { + continue; + } + ASSERT(!observed); + ASSERT_EQ(versioned.mutations.size(), expected.size()); + for (int i = 0; i < expected.size(); ++i) { + ASSERT_EQ(versioned.mutations[i].type, expected[i].type); + ASSERT_EQ(versioned.mutations[i].param1, expected[i].param1); + ASSERT_EQ(versioned.mutations[i].param2, expected[i].param2); + } + observed = true; + } + } + ASSERT_GE(consumer->position().lastConsumedVersion, committed); + } + + Future writeMultipleRangeClear(Database cx) { + Transaction tr(cx); + while (true) { + Error err; + try { + tr.set(multipleRangeKey("e"_sr), "before-left"_sr); + tr.set(multipleRangeKey("q"_sr), "before-middle"_sr); + tr.set(multipleRangeKey("w"_sr), "before-right"_sr); + tr.set(multipleRangeKey("h"_sr), "before-gap"_sr); + tr.clear(multipleRangeKeys("d"_sr, "x"_sr)); + tr.set(multipleRangeKey("e"_sr), "after-left"_sr); + tr.set(multipleRangeKey("q"_sr), "after-middle"_sr); + tr.set(multipleRangeKey("y"_sr), "after-right"_sr); + tr.clear(multipleRangeKeys("g"_sr, "j"_sr)); + co_await tr.commit(); + co_return tr.getCommittedVersion(); + } catch (Error& e) { + err = e; + } + co_await tr.onError(err); + } + } + + Future validateMultipleRanges(Database cx) { + ASSERT(streams.empty()); + ASSERT_EQ(cx->clientInfo->get().nativeCdcTagCount, 1); + const Key name = "native-cdc-e2e/multiple-ranges"_sr; + const Key gapName = "native-cdc-e2e/multiple-ranges-gap"_sr; + const std::vector ranges{ multipleRangeKeys("b"_sr, "f"_sr), + multipleRangeKeys("m"_sr, "r"_sr), + multipleRangeKeys("w"_sr, "z"_sr) }; + const std::vector registrationRanges{ + multipleRangeKeys("m"_sr, "r"_sr), multipleRangeKeys("c"_sr, "f"_sr), multipleRangeKeys("b"_sr, "d"_sr), + multipleRangeKeys("b"_sr, "c"_sr), multipleRangeKeys("x"_sr, "z"_sr), multipleRangeKeys("w"_sr, "x"_sr), + multipleRangeKeys("m"_sr, "r"_sr) + }; + const CDCStreamId streamId = + co_await timeoutError(registerNativeCdcStreamClient(cx, name, registrationRanges), operationTimeout); + ASSERT_EQ(co_await timeoutError(registerNativeCdcStreamClient(cx, name, ranges), operationTimeout), streamId); + bool changedGapsRejected = false; + try { + co_await timeoutError(registerNativeCdcStreamClient(cx, name, { multipleRangeKeys("b"_sr, "z"_sr) }), + operationTimeout); + } catch (Error& e) { + if (e.code() != error_code_client_invalid_operation) { + throw; + } + changedGapsRejected = true; + } + ASSERT(changedGapsRejected); + const std::vector listed = + co_await timeoutError(listNativeCdcStreamsClient(cx), operationTimeout); + const auto found = std::find_if( + listed.begin(), listed.end(), [&](NativeCdcStreamInfo const& stream) { return stream.name == name; }); + ASSERT(found != listed.end()); + ASSERT_EQ(found->streamId, streamId); + ASSERT_EQ(found->ranges, ranges); + + // Route excluded keys onto the same tag so the proxy must filter shared-tag false positives. + co_await timeoutError(registerNativeCdcStreamClient(cx, gapName, { multipleRangeKeys("a"_sr, "zz"_sr) }), + operationTimeout); + Reference consumer = + co_await timeoutError(createNativeCdcConsumer(cx, name), operationTimeout); + ASSERT_EQ(consumer->position().streamId, streamId); + std::vector> values; + for (StringRef suffix : + { "a"_sr, "b"_sr, "e"_sr, "f"_sr, "h"_sr, "m"_sr, "q"_sr, "r"_sr, "s"_sr, "w"_sr, "y"_sr, "z"_sr }) { + values.emplace_back(multipleRangeKey(suffix), suffix); + } + Standalone> expected; + for (StringRef suffix : { "b"_sr, "e"_sr, "m"_sr, "q"_sr, "w"_sr, "y"_sr }) { + expected.push_back_deep(expected.arena(), + MutationRef(MutationRef::SetValue, multipleRangeKey(suffix), suffix)); + } + const Version written = co_await writeValues(cx, values); + co_await consumeExpectedVersion(consumer, written, expected); + co_await timeoutError(consumer->acknowledge(), operationTimeout); + + expected = Standalone>(); + for (const auto& [suffix, value] : { std::pair("e"_sr, "before-left"_sr), + std::pair("q"_sr, "before-middle"_sr), + std::pair("w"_sr, "before-right"_sr) }) { + expected.push_back_deep(expected.arena(), + MutationRef(MutationRef::SetValue, multipleRangeKey(suffix), value)); + } + for (const auto& range : { multipleRangeKeys("d"_sr, "f"_sr), ranges[1], multipleRangeKeys("w"_sr, "x"_sr) }) { + expected.push_back_deep(expected.arena(), MutationRef(MutationRef::ClearRange, range.begin, range.end)); + } + for (const auto& [suffix, value] : { std::pair("e"_sr, "after-left"_sr), + std::pair("q"_sr, "after-middle"_sr), + std::pair("y"_sr, "after-right"_sr) }) { + expected.push_back_deep(expected.arena(), + MutationRef(MutationRef::SetValue, multipleRangeKey(suffix), value)); + } + const Version cleared = co_await writeMultipleRangeClear(cx); + co_await consumeExpectedVersion(consumer, cleared, expected); + + CDCProxyInterface original = co_await timeoutError(waitForAssignedProxy(cx, streamId), operationTimeout); + co_await timeoutError(haltProxyUntilReplaced(cx, original, false), operationTimeout); + CDCProxyInterface replacement = + co_await timeoutError(waitForAssignedProxy(cx, streamId, original.id()), operationTimeout); + ASSERT_NE(original.id(), replacement.id()); + // One unacknowledged version must be replayed in full, including all disjoint clear fragments. + co_await consumeExpectedVersion(consumer, cleared, expected); + co_await timeoutError(consumer->acknowledge(), operationTimeout); + const CDCCursor checkpoint = consumer->position(); + ASSERT_EQ(checkpoint.streamId, streamId); + const std::vector acknowledged = + co_await timeoutError(listNativeCdcStreamsClient(cx), operationTimeout); + const auto acknowledgedStream = std::find_if( + acknowledged.begin(), acknowledged.end(), [&](const auto& stream) { return stream.name == name; }); + ASSERT(acknowledgedStream != acknowledged.end()); + ASSERT_EQ(acknowledgedStream->ranges, ranges); + ASSERT_EQ(acknowledgedStream->minVersion, checkpoint.lastConsumedVersion + 1); + // The multi-range stream must retain unread history without another stream holding its tag back. + co_await timeoutError(removeNativeCdcStreamClient(cx, gapName), operationTimeout); + + values.clear(); + expected = Standalone>(); + for (StringRef suffix : { "b"_sr, "m"_sr, "w"_sr }) { + values.emplace_back(multipleRangeKey(suffix), "recovered-range"_sr); + expected.push_back_deep(expected.arena(), + MutationRef(MutationRef::SetValue, values.back().first, values.back().second)); + } + values.emplace_back(multipleRangeKey("h"_sr), "excluded-gap"_sr); + const Version retained = co_await writeValues(cx, values); + co_await timeoutError(forceTransactionSystemRecovery(), operationTimeout); + consumer = resumeNativeCdcConsumer(cx, checkpoint); + co_await consumeExpectedVersion(consumer, retained, expected); + co_await timeoutError(consumer->acknowledge(), operationTimeout); + + // A new commit must use the recovered range routing, including exclusion of the untracked gap. + const Version resumed = co_await writeValues(cx, values); + co_await consumeExpectedVersion(consumer, resumed, expected); + co_await timeoutError(consumer->acknowledge(), operationTimeout); + co_await timeoutError(removeNativeCdcStreamClient(cx, name), operationTimeout); + co_await timeoutError(waitForRetiredTagCleanup(cx), operationTimeout); + co_await timeoutError(waitForFullyRecovered(), operationTimeout); + CODE_PROBE(true, "Native CDC consumes and recovers one stream covering multiple disjoint ranges"); + } + Future validateAssignmentPublication(Database cx) { for (int check = 0; check < assignmentPublicationChecks; ++check) { co_await validateAssignmentPublicationOnce(cx, check); @@ -487,7 +664,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { co_await timeoutError(setAllProxyPopsPaused(cx, true), operationTimeout); streamRegistered = true; const CDCStreamId streamId = - co_await timeoutError(registerNativeCdcStreamClient(cx, name, keys), operationTimeout); + co_await timeoutError(registerNativeCdcStreamClient(cx, name, { keys }), operationTimeout); CDCProxyInterface proxy = co_await timeoutError(waitForAssignedProxy(cx, streamId), operationTimeout); const CDCCursor cursor(streamId, invalidVersion); Future> pendingConsume = proxy.consume.tryGetReply(CDCConsumeRequest(cursor)); @@ -540,7 +717,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { const Value value = "replacement-value"_sr; const CDCStreamId streamId = - co_await timeoutError(registerNativeCdcStreamClient(cx, name, keys), operationTimeout); + co_await timeoutError(registerNativeCdcStreamClient(cx, name, { keys }), operationTimeout); Reference consumer = co_await timeoutError(createNativeCdcConsumer(cx, name), operationTimeout); CDCProxyInterface original = co_await timeoutError(waitForAssignedProxy(cx, streamId), operationTimeout); @@ -615,7 +792,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { KeyRange keys, Optional sharedTagStream = Optional()) { const CDCRegisterStreamReply reply = co_await timeoutError( - proxy.registerStream.getReply(CDCRegisterStreamRequest(name, keys)), operationTimeout); + proxy.registerStream.getReply(CDCRegisterStreamRequest(name, { keys })), operationTimeout); co_await timeoutError(waitForAssignedProxy(cx, reply.streamId), operationTimeout); // Recovery can replace the owner while registration or consumption is in flight. Compare live streams // in the same published snapshot instead of retaining a proxy ID across those waits. @@ -642,7 +819,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { const Key key = "native-cdc-e2e/tag-owner/data"_sr; const KeyRange keys(KeyRangeRef(key, keyAfter(key))); const CDCStreamId firstId = - co_await timeoutError(registerNativeCdcStreamClient(cx, firstName, keys), operationTimeout); + co_await timeoutError(registerNativeCdcStreamClient(cx, firstName, { keys }), operationTimeout); CDCProxyInterface owner = co_await timeoutError(waitForAssignedProxy(cx, firstId), operationTimeout); std::vector proxies = cx->clientInfo->get().cdcProxies; ASSERT_EQ(proxies.size(), 2); @@ -748,7 +925,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { const Value value = "replacement-survives-stale-removal"_sr; const double deadline = now() + operationTimeout; CDCStreamId expectedStreamId = - co_await timeoutError(registerNativeCdcStreamClient(cx, name, keys), operationTimeout); + co_await timeoutError(registerNativeCdcStreamClient(cx, name, { keys }), operationTimeout); const auto rejectedWrongOwner = [](ErrorOr const& result) { ASSERT(!result.present()); @@ -844,7 +1021,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { co_await timeoutError(removeNativeCdcStreamClient(cx, name), operationTimeout); const CDCStreamId replacementId = - co_await timeoutError(registerNativeCdcStreamClient(cx, name, keys), operationTimeout); + co_await timeoutError(registerNativeCdcStreamClient(cx, name, { keys }), operationTimeout); ASSERT_NE(replacementId, streamId); expectedStreamId = replacementId; const ErrorOr staleRemoval = co_await timeoutError( @@ -1047,7 +1224,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { for (int i = 0; i < 4; ++i) { Key name = Key(StringRef(format("native-cdc-e2e/lease/%04d", i))); Key key = keyForIndex(keyCount / 2); - co_await timeoutError(registerNativeCdcStreamClient(cx, name, KeyRangeRef(key, keyAfter(key))), + co_await timeoutError(registerNativeCdcStreamClient(cx, name, { KeyRangeRef(key, keyAfter(key)) }), operationTimeout); CDCProxyBufferStatus status = co_await getCurrentProxyStatus(cx, streamId, proxy); ASSERT_LE(status.activeConsumeRequests, 1); @@ -1629,7 +1806,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { Optional registrationError; try { co_await timeoutError( - registerNativeCdcStreamClient(cx, "native-cdc-e2e/disabled-registration"_sr, normalKeys), + registerNativeCdcStreamClient(cx, "native-cdc-e2e/disabled-registration"_sr, { normalKeys }), operationTimeout); } catch (Error& e) { registrationError = e; @@ -1711,6 +1888,10 @@ class NativeCdcEndToEndWorkload : public TestWorkload { } Future run(Database cx) { + if (testMultipleRanges) { + co_await validateMultipleRanges(cx); + co_return; + } if (testRetiredSharedTagSnapshot) { co_await validateRetiredSharedTagSnapshot(cx); co_return; @@ -1809,6 +1990,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { injectUndeliveredProxyHalt = getOption(options, "injectUndeliveredProxyHalt"_sr, false); testMemoryBound = getOption(options, "testMemoryBound"_sr, false); testReplyChunking = getOption(options, "testReplyChunking"_sr, false); + testMultipleRanges = getOption(options, "testMultipleRanges"_sr, false); testOversizedPeek = getOption(options, "testOversizedPeek"_sr, false); testDurableAckScan = getOption(options, "testDurableAckScan"_sr, false); testDelayedRetention = getOption(options, "testDelayedRetention"_sr, false); @@ -1847,6 +2029,9 @@ class NativeCdcEndToEndWorkload : public TestWorkload { if (drainAfterRestart) { return Void(); } + if (testMultipleRanges) { + return Void(); + } if (prepareRestartDrain) { return prepareRestartDrainSetup(cx); } diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 22214551a66..a4b3787a19a 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -210,6 +210,7 @@ if(WITH_PYTHON) add_fdb_test(TEST_FILES fast/NativeCdcRetiredSharedTagSnapshot.toml) add_fdb_test(TEST_FILES fast/NativeCdcMemoryBound.toml) add_fdb_test(TEST_FILES fast/NativeCdcReplyChunking.toml) + add_fdb_test(TEST_FILES fast/NativeCdcMultipleRanges.toml) add_fdb_test(TEST_FILES fast/NativeCdcOversizedPeek.toml) add_fdb_test(TEST_FILES fast/NativeCdcDurableAckScan.toml) add_fdb_test(TEST_FILES fast/NativeCdcRetiredRecovery.toml) diff --git a/tests/fast/NativeCdcMultipleRanges.toml b/tests/fast/NativeCdcMultipleRanges.toml new file mode 100644 index 00000000000..e8407bc345b --- /dev/null +++ b/tests/fast/NativeCdcMultipleRanges.toml @@ -0,0 +1,21 @@ +[configuration] +config = 'single' +singleRegion = true +buggify = false +faultInjection = false + +[[knobs]] +enable_native_cdc = true +native_cdc_tag_count = 1 + +[[test]] +testTitle = 'NativeCdcMultipleRanges' +useDB = true +waitForQuiescenceEnd = false +connectionFailuresDisableDuration = 1000000 +runFailureWorkloads = false + + [[test.workload]] + testName = 'NativeCdcEndToEnd' + testMultipleRanges = true + operationTimeout = 180.0 From 4d13abb28f40faf106fa0d8caa48a2722fde2cb1 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 30 Aug 2026 13:42:15 -0700 Subject: [PATCH 009/170] Fix native CDC range test clang-tidy warning --- fdbclient/NativeCdc.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fdbclient/NativeCdc.cpp b/fdbclient/NativeCdc.cpp index a5cc8d24c84..0a55d943823 100644 --- a/fdbclient/NativeCdc.cpp +++ b/fdbclient/NativeCdc.cpp @@ -948,7 +948,7 @@ TEST_CASE("/NativeCDC/InvalidRanges") { std::vector maximumCount; std::vector oversizedMetadata; - const int endpointLength = CLIENT_KNOBS->VALUE_SIZE_LIMIT / (2 * NATIVE_CDC_MAX_RANGES); + const int endpointLength = CLIENT_KNOBS->VALUE_SIZE_LIMIT / (int64_t{ 2 } * NATIVE_CDC_MAX_RANGES); for (int i = 0; i < NATIVE_CDC_MAX_RANGES; ++i) { const std::string prefix = format("%04d/", i); maximumCount.emplace_back(KeyRangeRef(prefix + "a", prefix + "z")); From 2e96506b72861d617e75a0cbe81a5e4e97e86720 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 30 Aug 2026 14:47:26 -0700 Subject: [PATCH 010/170] Fix native CDC workload compilation with GCC 13 --- fdbserver/workloads/NativeCdcEndToEnd.cpp | 50 ++++++++++++++--------- 1 file changed, 31 insertions(+), 19 deletions(-) diff --git a/fdbserver/workloads/NativeCdcEndToEnd.cpp b/fdbserver/workloads/NativeCdcEndToEnd.cpp index 31dcf6d909a..2f7fbccfa4f 100644 --- a/fdbserver/workloads/NativeCdcEndToEnd.cpp +++ b/fdbserver/workloads/NativeCdcEndToEnd.cpp @@ -225,7 +225,9 @@ class NativeCdcEndToEndWorkload : public TestWorkload { StreamState stream; stream.name = Key(StringRef(format("native-cdc-e2e/stream/%04d", nextStreamNumber++))); stream.keys = std::move(keys); - co_await timeoutError(registerNativeCdcStreamClient(cx, stream.name, { stream.keys }), operationTimeout); + // GCC 13 cannot lower initializer-list vector arguments inside these awaited expressions. + const std::vector ranges{ stream.keys }; + co_await timeoutError(registerNativeCdcStreamClient(cx, stream.name, ranges), operationTimeout); stream.consumer = co_await timeoutError(createNativeCdcConsumer(cx, stream.name), operationTimeout); streams.push_back(std::move(stream)); } @@ -259,14 +261,16 @@ class NativeCdcEndToEndWorkload : public TestWorkload { const Key name = "native-cdc-e2e/lifecycle"_sr; const KeyRange keys(KeyRangeRef("native-cdc-e2e/lifecycle/"_sr, "native-cdc-e2e/lifecycle0"_sr)); const KeyRange conflictingKeys(KeyRangeRef("native-cdc-e2e/lifecycle/"_sr, "native-cdc-e2e/lifecycle1"_sr)); + const std::vector ranges{ keys }; const CDCStreamId streamId = - co_await timeoutError(registerNativeCdcStreamClient(cx, name, { keys }), operationTimeout); - ASSERT_EQ(co_await timeoutError(registerNativeCdcStreamClient(cx, name, { keys }), operationTimeout), streamId); + co_await timeoutError(registerNativeCdcStreamClient(cx, name, ranges), operationTimeout); + ASSERT_EQ(co_await timeoutError(registerNativeCdcStreamClient(cx, name, ranges), operationTimeout), streamId); bool conflictingRegistrationRejected = false; try { - co_await timeoutError(registerNativeCdcStreamClient(cx, name, { conflictingKeys }), operationTimeout); + const std::vector conflictingRanges{ conflictingKeys }; + co_await timeoutError(registerNativeCdcStreamClient(cx, name, conflictingRanges), operationTimeout); } catch (Error& e) { if (e.code() != error_code_client_invalid_operation) { throw; @@ -346,7 +350,8 @@ class NativeCdcEndToEndWorkload : public TestWorkload { const KeyRange expectedLower(KeyRangeRef(keys.begin, lowerClear.end)); const KeyRange expectedUpper(KeyRangeRef(upperClear.begin, keys.end)); - co_await timeoutError(registerNativeCdcStreamClient(cx, name, { keys }), operationTimeout); + const std::vector ranges{ keys }; + co_await timeoutError(registerNativeCdcStreamClient(cx, name, ranges), operationTimeout); Reference consumer = co_await timeoutError(createNativeCdcConsumer(cx, name), operationTimeout); @@ -400,8 +405,9 @@ class NativeCdcEndToEndWorkload : public TestWorkload { const Value value = Value(StringRef(format("assignment-value/%04d", check))); co_await delay(0.1); + const std::vector ranges{ keys }; const CDCStreamId streamId = - co_await timeoutError(registerNativeCdcStreamClient(cx, name, { keys }), operationTimeout); + co_await timeoutError(registerNativeCdcStreamClient(cx, name, ranges), operationTimeout); Reference consumer = co_await timeoutError(createNativeCdcConsumer(cx, name), operationTimeout); ASSERT_EQ(consumer->position().streamId, streamId); @@ -486,8 +492,8 @@ class NativeCdcEndToEndWorkload : public TestWorkload { ASSERT_EQ(co_await timeoutError(registerNativeCdcStreamClient(cx, name, ranges), operationTimeout), streamId); bool changedGapsRejected = false; try { - co_await timeoutError(registerNativeCdcStreamClient(cx, name, { multipleRangeKeys("b"_sr, "z"_sr) }), - operationTimeout); + const std::vector mergedRanges{ multipleRangeKeys("b"_sr, "z"_sr) }; + co_await timeoutError(registerNativeCdcStreamClient(cx, name, mergedRanges), operationTimeout); } catch (Error& e) { if (e.code() != error_code_client_invalid_operation) { throw; @@ -504,8 +510,8 @@ class NativeCdcEndToEndWorkload : public TestWorkload { ASSERT_EQ(found->ranges, ranges); // Route excluded keys onto the same tag so the proxy must filter shared-tag false positives. - co_await timeoutError(registerNativeCdcStreamClient(cx, gapName, { multipleRangeKeys("a"_sr, "zz"_sr) }), - operationTimeout); + const std::vector gapRanges{ multipleRangeKeys("a"_sr, "zz"_sr) }; + co_await timeoutError(registerNativeCdcStreamClient(cx, gapName, gapRanges), operationTimeout); Reference consumer = co_await timeoutError(createNativeCdcConsumer(cx, name), operationTimeout); ASSERT_EQ(consumer->position().streamId, streamId); @@ -645,6 +651,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { const Key name = "native-cdc-e2e/stale-initialization"_sr; const KeyRange keys( KeyRangeRef("native-cdc-e2e/stale-initialization/"_sr, "native-cdc-e2e/stale-initialization0"_sr)); + const std::vector ranges{ keys }; const double deadline = now() + operationTimeout; bool recovering = false; bool streamRegistered = false; @@ -664,7 +671,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { co_await timeoutError(setAllProxyPopsPaused(cx, true), operationTimeout); streamRegistered = true; const CDCStreamId streamId = - co_await timeoutError(registerNativeCdcStreamClient(cx, name, { keys }), operationTimeout); + co_await timeoutError(registerNativeCdcStreamClient(cx, name, ranges), operationTimeout); CDCProxyInterface proxy = co_await timeoutError(waitForAssignedProxy(cx, streamId), operationTimeout); const CDCCursor cursor(streamId, invalidVersion); Future> pendingConsume = proxy.consume.tryGetReply(CDCConsumeRequest(cursor)); @@ -716,8 +723,9 @@ class NativeCdcEndToEndWorkload : public TestWorkload { const Key key = "native-cdc-e2e/proxy-replacement/value"_sr; const Value value = "replacement-value"_sr; + const std::vector ranges{ keys }; const CDCStreamId streamId = - co_await timeoutError(registerNativeCdcStreamClient(cx, name, { keys }), operationTimeout); + co_await timeoutError(registerNativeCdcStreamClient(cx, name, ranges), operationTimeout); Reference consumer = co_await timeoutError(createNativeCdcConsumer(cx, name), operationTimeout); CDCProxyInterface original = co_await timeoutError(waitForAssignedProxy(cx, streamId), operationTimeout); @@ -791,8 +799,9 @@ class NativeCdcEndToEndWorkload : public TestWorkload { Key name, KeyRange keys, Optional sharedTagStream = Optional()) { + const std::vector ranges{ keys }; const CDCRegisterStreamReply reply = co_await timeoutError( - proxy.registerStream.getReply(CDCRegisterStreamRequest(name, { keys })), operationTimeout); + proxy.registerStream.getReply(CDCRegisterStreamRequest(name, ranges)), operationTimeout); co_await timeoutError(waitForAssignedProxy(cx, reply.streamId), operationTimeout); // Recovery can replace the owner while registration or consumption is in flight. Compare live streams // in the same published snapshot instead of retaining a proxy ID across those waits. @@ -818,8 +827,9 @@ class NativeCdcEndToEndWorkload : public TestWorkload { const Key thirdName = "native-cdc-e2e/tag-owner/third"_sr; const Key key = "native-cdc-e2e/tag-owner/data"_sr; const KeyRange keys(KeyRangeRef(key, keyAfter(key))); + const std::vector ranges{ keys }; const CDCStreamId firstId = - co_await timeoutError(registerNativeCdcStreamClient(cx, firstName, { keys }), operationTimeout); + co_await timeoutError(registerNativeCdcStreamClient(cx, firstName, ranges), operationTimeout); CDCProxyInterface owner = co_await timeoutError(waitForAssignedProxy(cx, firstId), operationTimeout); std::vector proxies = cx->clientInfo->get().cdcProxies; ASSERT_EQ(proxies.size(), 2); @@ -924,8 +934,9 @@ class NativeCdcEndToEndWorkload : public TestWorkload { const Key key = "native-cdc-e2e/stale-ownership/value"_sr; const Value value = "replacement-survives-stale-removal"_sr; const double deadline = now() + operationTimeout; + const std::vector ranges{ keys }; CDCStreamId expectedStreamId = - co_await timeoutError(registerNativeCdcStreamClient(cx, name, { keys }), operationTimeout); + co_await timeoutError(registerNativeCdcStreamClient(cx, name, ranges), operationTimeout); const auto rejectedWrongOwner = [](ErrorOr const& result) { ASSERT(!result.present()); @@ -1021,7 +1032,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { co_await timeoutError(removeNativeCdcStreamClient(cx, name), operationTimeout); const CDCStreamId replacementId = - co_await timeoutError(registerNativeCdcStreamClient(cx, name, { keys }), operationTimeout); + co_await timeoutError(registerNativeCdcStreamClient(cx, name, ranges), operationTimeout); ASSERT_NE(replacementId, streamId); expectedStreamId = replacementId; const ErrorOr staleRemoval = co_await timeoutError( @@ -1224,8 +1235,8 @@ class NativeCdcEndToEndWorkload : public TestWorkload { for (int i = 0; i < 4; ++i) { Key name = Key(StringRef(format("native-cdc-e2e/lease/%04d", i))); Key key = keyForIndex(keyCount / 2); - co_await timeoutError(registerNativeCdcStreamClient(cx, name, { KeyRangeRef(key, keyAfter(key)) }), - operationTimeout); + const std::vector ranges{ KeyRangeRef(key, keyAfter(key)) }; + co_await timeoutError(registerNativeCdcStreamClient(cx, name, ranges), operationTimeout); CDCProxyBufferStatus status = co_await getCurrentProxyStatus(cx, streamId, proxy); ASSERT_LE(status.activeConsumeRequests, 1); ASSERT_LE(status.readDemand, 1); @@ -1805,8 +1816,9 @@ class NativeCdcEndToEndWorkload : public TestWorkload { Optional registrationError; try { + const std::vector ranges{ normalKeys }; co_await timeoutError( - registerNativeCdcStreamClient(cx, "native-cdc-e2e/disabled-registration"_sr, { normalKeys }), + registerNativeCdcStreamClient(cx, "native-cdc-e2e/disabled-registration"_sr, ranges), operationTimeout); } catch (Error& e) { registrationError = e; From feed40eca221dedf3e6b454ede19e76ff1647d43 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 30 Aug 2026 15:34:17 -0700 Subject: [PATCH 011/170] Fix native CDC workload clang-format failure --- fdbserver/workloads/NativeCdcEndToEnd.cpp | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/fdbserver/workloads/NativeCdcEndToEnd.cpp b/fdbserver/workloads/NativeCdcEndToEnd.cpp index 2f7fbccfa4f..bfe4d296c7e 100644 --- a/fdbserver/workloads/NativeCdcEndToEnd.cpp +++ b/fdbserver/workloads/NativeCdcEndToEnd.cpp @@ -1817,9 +1817,8 @@ class NativeCdcEndToEndWorkload : public TestWorkload { Optional registrationError; try { const std::vector ranges{ normalKeys }; - co_await timeoutError( - registerNativeCdcStreamClient(cx, "native-cdc-e2e/disabled-registration"_sr, ranges), - operationTimeout); + co_await timeoutError(registerNativeCdcStreamClient(cx, "native-cdc-e2e/disabled-registration"_sr, ranges), + operationTimeout); } catch (Error& e) { registrationError = e; } From f998c3160aa791517267934814c9ee531df0dae3 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 30 Aug 2026 15:54:00 -0700 Subject: [PATCH 012/170] Account for compact singleton ranges in native CDC size validation --- fdbclient/NativeCdc.cpp | 21 ++++++++++++++++++++- 1 file changed, 20 insertions(+), 1 deletion(-) diff --git a/fdbclient/NativeCdc.cpp b/fdbclient/NativeCdc.cpp index 0a55d943823..2402095e19d 100644 --- a/fdbclient/NativeCdc.cpp +++ b/fdbclient/NativeCdc.cpp @@ -120,7 +120,8 @@ void normalizeNativeCdcStreamRanges(KeyRef const& name, std::vector& r int64_t keyBytes = 0; for (const auto& range : ranges) { - keyBytes += static_cast(range.begin.size()) + range.end.size(); + // Single-key ranges serialize only the end key, so count the encoded payload before allocating metadata. + keyBytes += static_cast(range.end.size()) + (range.singleKeyRange() ? 0 : range.begin.size()); } if (keyBytes > CLIENT_KNOBS->VALUE_SIZE_LIMIT || cdcStreamKeysValue(ranges).size() > CLIENT_KNOBS->VALUE_SIZE_LIMIT) { @@ -927,6 +928,24 @@ TEST_CASE("/NativeCDC/RangeNormalization") { std::vector entireKeyspace{ normalKeys }; normalizeNativeCdcStreamRanges("all"_sr, entireKeyspace); ASSERT(entireKeyspace == std::vector{ normalKeys }); + + constexpr int singletonCount = 10; + const int keyLength = CLIENT_KNOBS->VALUE_SIZE_LIMIT / (2 * singletonCount) + 1; + ASSERT_LT(keyLength, CLIENT_KNOBS->KEY_SIZE_LIMIT); + std::vector singletonRanges; + int64_t endpointBytes = 0; + for (int i = 0; i < singletonCount; ++i) { + const std::string prefix = format("%04d/", i); + ASSERT_GT(keyLength, prefix.size()); + const Key key(StringRef(prefix + std::string(keyLength - prefix.size(), 'x'))); + singletonRanges.push_back(singleKeyRange(key)); + endpointBytes += static_cast(key.size()) + key.size() + 1; + } + ASSERT_GT(endpointBytes, CLIENT_KNOBS->VALUE_SIZE_LIMIT); + ASSERT_LE(cdcStreamKeysValue(singletonRanges).size(), CLIENT_KNOBS->VALUE_SIZE_LIMIT); + const auto expectedSingletons = singletonRanges; + normalizeNativeCdcStreamRanges("singletons"_sr, singletonRanges); + ASSERT(singletonRanges == expectedSingletons); return Void(); } From c7fb905dc3994d4b2babcae5b299acaaee67a91c Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 30 Aug 2026 16:42:52 -0700 Subject: [PATCH 013/170] Fix native CDC range test clang-tidy warning --- fdbclient/NativeCdc.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fdbclient/NativeCdc.cpp b/fdbclient/NativeCdc.cpp index 2402095e19d..076cdaa0ab6 100644 --- a/fdbclient/NativeCdc.cpp +++ b/fdbclient/NativeCdc.cpp @@ -930,7 +930,7 @@ TEST_CASE("/NativeCDC/RangeNormalization") { ASSERT(entireKeyspace == std::vector{ normalKeys }); constexpr int singletonCount = 10; - const int keyLength = CLIENT_KNOBS->VALUE_SIZE_LIMIT / (2 * singletonCount) + 1; + const int keyLength = CLIENT_KNOBS->VALUE_SIZE_LIMIT / (2LL * singletonCount) + 1; ASSERT_LT(keyLength, CLIENT_KNOBS->KEY_SIZE_LIMIT); std::vector singletonRanges; int64_t endpointBytes = 0; From 9f705da93112c71b1ca7883992a06cfd4e84fa34 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Wed, 2 Sep 2026 23:00:47 -0700 Subject: [PATCH 014/170] Wake native CDC peeks when the committed frontier advances --- fdbserver/cdcproxy/CDCProxy.cpp | 30 ++- fdbserver/tlog/TLogServer.cpp | 414 ++++++++++++++++++++++++++++++-- 2 files changed, 419 insertions(+), 25 deletions(-) diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index b92acff8c3b..b8c6fd79e6d 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -195,6 +195,10 @@ bool hasCompleteLogSystemConfig(LogSystemConfig const& config) { enum class CDCBufferTagPassResult { RETRY, WAIT_FOR_COMMIT, STOP }; +double remainingCommitWait(double passStart, double waitInterval, double currentTime) { + return std::max(0.0, passStart + waitInterval - currentTime); +} + Optional calculateBufferPassLimits(int64_t bufferBytes, int64_t maximumPeekBytes, int64_t retainedReplyCount) { @@ -1146,17 +1150,20 @@ Future CDCProxy::bufferTag(Reference tag) { continue; } + const double passStart = now(); const CDCBufferTagPassResult result = co_await bufferTagPass(tag, begin.get()); if (result == CDCBufferTagPassResult::STOP) { co_return; } if (result == CDCBufferTagPassResult::WAIT_FOR_COMMIT) { - // The cursor may already hold a speculative message, so getMore() would complete immediately without - // refreshing its committed frontier. Drop that arena and reopen after one blocking-peek interval. - auto waitForCommit = co_await race(delay(SERVER_KNOBS->BLOCKING_PEEK_TIMEOUT), - logSystem->onChange(), - tag->stopped.onTrigger(), - tag->refresh.onTrigger()); + // Older TLogs can immediately return a speculative message, for which getMore() would not refresh + // the committed frontier. Retain their bounded fallback, but do not add a second blocking-peek + // interval when a TLog already waited for commit progress. The zero-delay case still yields. + auto waitForCommit = + co_await race(delay(remainingCommitWait(passStart, SERVER_KNOBS->BLOCKING_PEEK_TIMEOUT, now())), + logSystem->onChange(), + tag->stopped.onTrigger(), + tag->refresh.onTrigger()); if (waitForCommit.index() == 2) { co_return; } @@ -1958,6 +1965,17 @@ TEST_CASE("/NativeCDC/ProxyBufferCandidateSelection") { return Void(); } +TEST_CASE("/NativeCDC/CommitWaitDeadlineBudget") { + // Immediate legacy replies retain the fallback; time already spent in the pass is not charged twice. + ASSERT_EQ(remainingCommitWait(0.0, 0.4, 0.0), 0.4); + ASSERT_EQ(remainingCommitWait(0.0, 0.4, 0.2), 0.2); + ASSERT_EQ(remainingCommitWait(0.0, 0.4, 0.4), 0.0); + // Capacity waiting is part of the same pass, even when it exceeds the timeout before a peek is issued. + ASSERT_EQ(remainingCommitWait(0.0, 0.4, 2.0), 0.0); + ASSERT_EQ(remainingCommitWait(8.0, 0.5, 8.125), 0.375); + return Void(); +} + TEST_CASE("/NativeCDC/ProxyBufferPassLimits") { Optional replicated = calculateBufferPassLimits(1000, 100, 3); ASSERT(replicated.present()); diff --git a/fdbserver/tlog/TLogServer.cpp b/fdbserver/tlog/TLogServer.cpp index 4e586703b4a..570dc72b92b 100644 --- a/fdbserver/tlog/TLogServer.cpp +++ b/fdbserver/tlog/TLogServer.cpp @@ -58,6 +58,30 @@ FDB_BOOLEAN_PARAM(NothingPersistent); FDB_BOOLEAN_PARAM(PoppedRecently); FDB_BOOLEAN_PARAM(UnpoppedRecovered); +// Commit progress can advance without appending another message to this log. Keep its wakeup separate from +// LogData::version, and do not allocate a new notification on foreground commits when no reader is waiting. +class CommittedVersion : NonCopyable { +public: + Version get() const { return value; } + Future onAdvance() const { return changed.getFuture(); } + + void advance(Version version) { + if (version <= value) { + return; + } + value = version; + if (changed.getFutureReferenceCount() > 0) { + Promise notification; + notification.swap(changed); + notification.send(Void()); + } + } + +private: + Version value = 0; + Promise changed; +}; + } // namespace struct TLogQueueEntryRef { @@ -560,7 +584,7 @@ struct LogData : NonCopyable, public ReferenceCounted { Version knownCommittedVersion; // The maximum version that a proxy has told us that is committed (all TLogs have // ack'd a commit for this version). Version durableKnownCommittedVersion; - Version minKnownCommittedVersion; + CommittedVersion minKnownCommittedVersion; Version queuePoppedVersion; // The disk queue has been popped up until the location which represents this version. Version minPoppedTagVersion; Tag minPoppedTag; // The tag (locality >= 0) that makes tLog hold its data and cause tLog's disk queue increasing. @@ -707,15 +731,15 @@ struct LogData : NonCopyable, public ReferenceCounted { std::vector tags, std::string context) : initialized(false), queueCommittingVersion(0), knownCommittedVersion(0), durableKnownCommittedVersion(0), - minKnownCommittedVersion(0), queuePoppedVersion(0), minPoppedTagVersion(0), minPoppedTag(invalidTag), - minSysPopTagVersion(0), minSysPopTag(invalidTag), unpoppedRecoveredTagCount(0), - cc("TLog", interf.id().toString()), bytesInput("BytesInput", cc), tagMessageCount("tagMessageCount", cc), - bytesDurable("BytesDurable", cc), blockingPeeks("BlockingPeeks", cc), - blockingPeekTimeouts("BlockingPeekTimeouts", cc), emptyPeeks("EmptyPeeks", cc), - nonEmptyPeeks("NonEmptyPeeks", cc), persistentDataUpdateBatches("PersistentDataUpdateBatches", cc), - dirtyTagsProcessed("DirtyTagsProcessed", cc), logId(interf.id()), protocolVersion(protocolVersion), - newPersistentDataVersion(invalidVersion), tLogData(tLogData), unrecoveredBefore(1), recoveredAt(1), - recoveryTxnVersion(1), logSystem(new AsyncVar>()), + queuePoppedVersion(0), minPoppedTagVersion(0), minPoppedTag(invalidTag), minSysPopTagVersion(0), + minSysPopTag(invalidTag), unpoppedRecoveredTagCount(0), cc("TLog", interf.id().toString()), + bytesInput("BytesInput", cc), tagMessageCount("tagMessageCount", cc), bytesDurable("BytesDurable", cc), + blockingPeeks("BlockingPeeks", cc), blockingPeekTimeouts("BlockingPeekTimeouts", cc), + emptyPeeks("EmptyPeeks", cc), nonEmptyPeeks("NonEmptyPeeks", cc), + persistentDataUpdateBatches("PersistentDataUpdateBatches", cc), dirtyTagsProcessed("DirtyTagsProcessed", cc), + logId(interf.id()), protocolVersion(protocolVersion), newPersistentDataVersion(invalidVersion), + tLogData(tLogData), unrecoveredBefore(1), recoveredAt(1), recoveryTxnVersion(1), + logSystem(new AsyncVar>()), logSystemConsumer(new AsyncVar>()), remoteTag(remoteTag), isPrimary(isPrimary), logRouterTags(logRouterTags), logRouterPoppedVersion(0), logRouterPopToVersion(0), locality(tagLocalityInvalid), recruitmentID(recruitmentID), logSpillType(logSpillType), allTags(tags.begin(), tags.end()), @@ -1899,6 +1923,26 @@ Future waitForMessagesForTag(Reference self, Tag reqTag, Version } } +Future waitForCommittedVersion(Reference self, Version begin, double deadline) { + if (self->stopped() || self->minKnownCommittedVersion.get() >= begin) { + co_return; + } + ++self->blockingPeeks; + Future expired = delay(std::max(0.0, deadline - now()), TaskPriority::TLogPeekReply); + while (!self->stopped() && self->minKnownCommittedVersion.get() < begin) { + // No suspension separates the level check from registration. A canceled peek removes its callback; + // no threshold entry remains waiting for some future commit. + auto ready = + co_await race(self->minKnownCommittedVersion.onAdvance(), expired, self->stoppedPromise.getFuture()); + // Promise notification is synchronous. Never scan/materialize a reply on the tLogCommit or stop stack. + co_await delay(0, TaskPriority::TLogPeekReply); + if (ready.index() == 1) { + ++self->blockingPeekTimeouts; + co_return; + } + } +} + void peekMessagesFromMemory(Reference self, Tag tag, Version begin, @@ -2268,6 +2312,16 @@ Future tLogPeekMessages(PromiseType replyPromise, co_await delay(0, TaskPriority::TLogSpilledPeekReply); } + const bool waitForCommittedFrontier = replyByteLimit > 0 && reqTag.locality == tagLocalityCDC && + !reqReturnIfBlocked && !reqOnlySpilled && !logData->stopped() && + (!reqEnd.present() || reqEnd.get() == std::numeric_limits::max()); + if (waitForCommittedFrontier && poppedVersion(logData, reqTag) <= reqBegin) { + // A speculative message (or an empty tail) cannot advance native CDC until this frontier reaches begin. + // Waiting here lets commit progress wake the same capped peek instead of returning a stale frontier to + // the proxy. Keep finite/recovery, nonblocking, and uncapped peeks on their existing paths. + co_await waitForCommittedVersion(logData, reqBegin, now() + SERVER_KNOBS->BLOCKING_PEEK_TIMEOUT); + } + double workStart = now(); Version poppedVer{ 0 }; Version endVersion{ 0 }; @@ -2298,7 +2352,7 @@ Future tLogPeekMessages(PromiseType replyPromise, auto tagData = logData->getTagData(reqTag); bool tagRecovered = tagData && !tagData->unpoppedRecovered; - if (SERVER_KNOBS->ENABLE_VERSION_VECTOR && poppedVer <= reqBegin && + if (!waitForCommittedFrontier && SERVER_KNOBS->ENABLE_VERSION_VECTOR && poppedVer <= reqBegin && reqBegin > logData->persistentDataDurableVersion && !reqOnlySpilled && (reqTag.locality >= 0 || reqTag.locality == tagLocalityCDC) && !reqReturnIfBlocked && tagRecovered) { double startTime = now(); @@ -2323,7 +2377,7 @@ Future tLogPeekMessages(PromiseType replyPromise, if (poppedVer > reqBegin) { TLogPeekReply rep; rep.maxKnownVersion = logData->version.get(); - rep.minKnownCommittedVersion = logData->minKnownCommittedVersion; + rep.minKnownCommittedVersion = logData->minKnownCommittedVersion.get(); rep.popped = poppedVer; rep.end = poppedVer; rep.onlySpilled = false; @@ -2540,7 +2594,8 @@ Future tLogPeekMessages(PromiseType replyPromise, // - Have data return to the caller, or // - Batching empty peek is disabled, or // - Batching empty peek interval has been reached. - if (replyByteLimitReached || messages.getLength() > 0 || !SERVER_KNOBS->PEEK_BATCHING_EMPTY_MSG || + if (waitForCommittedFrontier || replyByteLimitReached || messages.getLength() > 0 || + !SERVER_KNOBS->PEEK_BATCHING_EMPTY_MSG || (now() - blockStart > SERVER_KNOBS->PEEK_BATCHING_EMPTY_MSG_INTERVAL)) { break; } @@ -2572,7 +2627,7 @@ Future tLogPeekMessages(PromiseType replyPromise, TLogPeekReply reply; reply.maxKnownVersion = logData->version.get(); - reply.minKnownCommittedVersion = logData->minKnownCommittedVersion; + reply.minKnownCommittedVersion = logData->minKnownCommittedVersion.get(); auto messagesValue = messages.toValue(); reply.arena.dependsOn(messagesValue.arena()); reply.messages = messagesValue; @@ -2847,7 +2902,7 @@ Future tLogCommit(TLogData* self, req.spanContext.spanID); } - logData->minKnownCommittedVersion = std::max(logData->minKnownCommittedVersion, req.minKnownCommittedVersion); + logData->minKnownCommittedVersion.advance(req.minKnownCommittedVersion); co_await logData->version.whenAtLeast(req.prevVersion); // Time until now has been spent waiting in the queue to do actual work. @@ -3583,8 +3638,7 @@ Future pullAsyncData(TLogData* self, if (poppedIsKnownCommitted) { logData->knownCommittedVersion = std::max(logData->knownCommittedVersion, r->popped()); - logData->minKnownCommittedVersion = - std::max(logData->minKnownCommittedVersion, r->getMinKnownCommittedVersion()); + logData->minKnownCommittedVersion.advance(r->getMinKnownCommittedVersion()); } co_await waitUntilTLogAcceptsNewData(self, logData, ver, endVersion); @@ -3624,8 +3678,7 @@ Future pullAsyncData(TLogData* self, if (poppedIsKnownCommitted) { logData->knownCommittedVersion = std::max(logData->knownCommittedVersion, r->popped()); - logData->minKnownCommittedVersion = - std::max(logData->minKnownCommittedVersion, r->getMinKnownCommittedVersion()); + logData->minKnownCommittedVersion.advance(r->getMinKnownCommittedVersion()); } co_await waitUntilTLogAcceptsNewData(self, logData, ver, endVersion); @@ -4446,6 +4499,329 @@ Future tLog(IKeyValueStore* persistentData, } // UNIT TESTS +namespace { + +class CommittedPeekTestKnobs : NonCopyable { +public: + explicit CommittedPeekTestKnobs(bool batchEmpty = true) + : versionVector(SERVER_KNOBS->ENABLE_VERSION_VECTOR), + recoveryReply(SERVER_KNOBS->ENABLE_VERSION_VECTOR_REPLY_RECOVERY), + batching(SERVER_KNOBS->PEEK_BATCHING_EMPTY_MSG), timeout(SERVER_KNOBS->BLOCKING_PEEK_TIMEOUT), + batchingInterval(SERVER_KNOBS->PEEK_BATCHING_EMPTY_MSG_INTERVAL) { + auto* knobs = const_cast(SERVER_KNOBS); + knobs->ENABLE_VERSION_VECTOR = true; + knobs->ENABLE_VERSION_VECTOR_REPLY_RECOVERY = false; + knobs->PEEK_BATCHING_EMPTY_MSG = batchEmpty; + knobs->BLOCKING_PEEK_TIMEOUT = 0.4; + knobs->PEEK_BATCHING_EMPTY_MSG_INTERVAL = 0.4; + } + + ~CommittedPeekTestKnobs() { + auto* knobs = const_cast(SERVER_KNOBS); + knobs->ENABLE_VERSION_VECTOR = versionVector; + knobs->ENABLE_VERSION_VECTOR_REPLY_RECOVERY = recoveryReply; + knobs->PEEK_BATCHING_EMPTY_MSG = batching; + knobs->BLOCKING_PEEK_TIMEOUT = timeout; + knobs->PEEK_BATCHING_EMPTY_MSG_INTERVAL = batchingInterval; + } + +private: + const bool versionVector; + const bool recoveryReply; + const bool batching; + const double timeout; + const double batchingInterval; +}; + +class CommittedPeekTestFixture : NonCopyable { +public: + explicit CommittedPeekTestFixture(bool batchEmpty = true, Version initialCommitted = 99) + : knobs(batchEmpty), shared(deterministicRandom()->randomUniqueID(), + deterministicRandom()->randomUniqueID(), + nullptr, + nullptr, + makeReference>(ServerDBInfo()), + makeReference>(false), + makeReference>(false), + "", + makeReference>(false)), + queue(shared.persistentQueue), logData(makeReference(&shared, + TLogInterface(LocalityData()), + invalidTag, + true, + 0, + 0, + deterministicRandom()->randomUniqueID(), + g_network->protocolVersion(), + TLogSpillType::VALUE, + std::vector{ cdcTag() }, + "CommittedPeekTest")) { + logData->version.set(100); + logData->queueCommittedVersion.set(100); + logData->knownCommittedVersion = initialCommitted; + logData->durableKnownCommittedVersion = initialCommitted; + logData->minKnownCommittedVersion.advance(initialCommitted); + logData->persistentDataVersion = 0; + logData->persistentDataDurableVersion = 0; + logData->locality = 0; + logData->getTagData(cdcTag()); + logData->createTagData(cdcTag(), 0, NothingPersistent::True, PoppedRecently::False, UnpoppedRecovered::False); + } + + ~CommittedPeekTestFixture() { + // This fixture only exercises in-memory peeks and duplicate commits. Termination prevents + // LogData's normal persistent-state removal from dereferencing the absent disk stores. + shared.terminated.send(Void()); + } + + static Tag cdcTag() { return Tag(tagLocalityCDC, 0); } + + static TLogPeekRequest request() { return TLogPeekRequest(100, cdcTag(), false, false, {}, {}, {}, 4096); } + + Future peek(Promise reply, const TLogPeekRequest& req = request()) { + return tLogPeekMessages(reply, + &shared, + logData, + req.begin, + req.tag, + req.returnIfBlocked, + req.onlySpilled, + req.sequence, + req.end, + req.returnEmptyIfStopped, + req.replyByteLimit); + } + + void addMessage(Tag tag = cdcTag()) { + BinaryWriter message(Unversioned()); + message << int32_t(0) << uint32_t(1) << uint16_t(1) << tag; + message << MutationRef(MutationRef::SetValue, "frontier-key"_sr, "frontier-value"_sr); + *reinterpret_cast(message.getData()) = message.getLength() - sizeof(int32_t); + auto bytes = message.toValue(); + commitMessages(&shared, logData, 100, bytes.arena(), bytes); + } + + Future commitFrontier(Version frontier, Version previous = 99) { + TLogCommitRequest req; + req.prevVersion = previous; + req.version = previous + 1; + req.knownCommittedVersion = std::max(99, frontier); + req.minKnownCommittedVersion = frontier; + req.seqPrevVersion = previous; + req.tLogCount = 1; + return tLogCommit(&shared, req, logData, PromiseStream()); + } + + void advance(Version frontier) { logData->minKnownCommittedVersion.advance(frontier); } + void stop() { logData->stop(); } + void popPastBegin() { logData->getTagData(cdcTag())->popped = 101; } + Version frontier() const { return logData->minKnownCommittedVersion.get(); } + int64_t timedOutPeeks() const { return logData->blockingPeekTimeouts.getValue(); } + void assertNoDiskCommit() const { + ASSERT(shared.diskQueueCommitBytes == 0); + ASSERT(logData->version.get() == 100); + } + +private: + CommittedPeekTestKnobs knobs; + TLogData shared; + std::unique_ptr queue; + Reference logData; +}; + +Future testCommittedPeekProgress(bool speculative) { + CommittedPeekTestFixture fixture; + if (speculative) { + fixture.addMessage(); + } + Promise reply; + Future peek = fixture.peek(reply); + ASSERT(!reply.getFuture().isReady()); + co_await delay(0.02); + ASSERT(!reply.getFuture().isReady()); + Future commit = fixture.commitFrontier(100); + ASSERT(!reply.getFuture().isReady()); + co_await commit; + TLogPeekReply result = co_await timeoutError(reply.getFuture(), 0.2); + co_await peek; + ASSERT(result.minKnownCommittedVersion == 100); + ASSERT(result.maxKnownVersion == 100); + ASSERT(result.end == 101); + ASSERT(result.messages.empty() == !speculative); + if (speculative) { + BinaryReader reader(result.messages, Unversioned()); + int32_t header; + Version version; + reader >> header >> version; + ASSERT(header == VERSION_HEADER); + ASSERT(version == 100); + } + fixture.assertNoDiskCommit(); +} + +} // namespace + +TEST_CASE("/NativeCDC/TLogCommittedFrontier/Empty") { + co_await testCommittedPeekProgress(false); +} + +TEST_CASE("/NativeCDC/TLogCommittedFrontier/Speculative") { + co_await testCommittedPeekProgress(true); +} + +TEST_CASE("/NativeCDC/TLogCommittedFrontier/AlreadyCommitted") { + CommittedPeekTestFixture fixture; + fixture.advance(100); + Promise reply; + Future peek = fixture.peek(reply); + TLogPeekReply result = co_await timeoutError(reply.getFuture(), 0.2); + co_await peek; + ASSERT(result.minKnownCommittedVersion == 100); + ASSERT(result.messages.empty()); +} + +TEST_CASE("/NativeCDC/TLogCommittedFrontier/Timeout") { + CommittedPeekTestFixture fixture; + Promise reply; + double started = now(); + Future peek = fixture.peek(reply); + TLogPeekReply result = co_await timeoutError(reply.getFuture(), 1.0); + co_await peek; + ASSERT(now() - started >= 0.35); + ASSERT(result.minKnownCommittedVersion == 99); + ASSERT(result.messages.empty()); + ASSERT(fixture.timedOutPeeks() == 1); +} + +TEST_CASE("/NativeCDC/TLogCommittedFrontier/Stop") { + CommittedPeekTestFixture fixture; + Promise reply; + Future peek = fixture.peek(reply); + co_await delay(0.02); + ASSERT(!reply.getFuture().isReady()); + fixture.stop(); + ASSERT(!reply.getFuture().isReady()); + TLogPeekReply result = co_await timeoutError(reply.getFuture(), 0.2); + co_await peek; + ASSERT(result.minKnownCommittedVersion == 99); + ASSERT(result.messages.empty()); +} + +TEST_CASE("/NativeCDC/TLogCommittedFrontier/AbsoluteDeadline") { + CommittedPeekTestFixture fixture(true, 90); + Promise reply; + double started = now(); + Future peek = fixture.peek(reply); + for (Version frontier = 91; frontier <= 93; ++frontier) { + co_await delay(0.1); + fixture.advance(frontier); + } + TLogPeekReply result = co_await timeoutError(reply.getFuture(), 0.2); + co_await peek; + ASSERT(now() - started >= 0.35 && now() - started < 0.6); + ASSERT(result.minKnownCommittedVersion == 93); + ASSERT(fixture.timedOutPeeks() == 1); +} + +TEST_CASE("/NativeCDC/TLogCommittedFrontier/Cancellation") { + CommittedPeekTestFixture fixture; + Promise abandonedReply; + Future abandoned = fixture.peek(abandonedReply); + co_await delay(0.02); + abandoned.cancel(); + ASSERT(abandoned.isReady() && abandoned.isError()); + ASSERT(abandoned.getError().code() == error_code_actor_cancelled); + Promise reply; + Future peek = fixture.peek(reply); + fixture.advance(100); + TLogPeekReply result = co_await timeoutError(reply.getFuture(), 0.2); + co_await peek; + ASSERT(result.minKnownCommittedVersion == 100); + ASSERT(!abandonedReply.getFuture().isReady()); +} + +TEST_CASE("/NativeCDC/TLogCommittedFrontier/MultipleWaiters") { + CommittedPeekTestFixture fixture; + std::vector> replies(8); + std::vector> peeks; + for (auto& reply : replies) { + peeks.push_back(fixture.peek(reply)); + } + co_await fixture.commitFrontier(98); + co_await fixture.commitFrontier(99); + co_await delay(0.02); + ASSERT(fixture.frontier() == 99); + for (auto& reply : replies) { + ASSERT(!reply.getFuture().isReady()); + } + Future commit = fixture.commitFrontier(100); + for (auto& reply : replies) { + ASSERT(!reply.getFuture().isReady()); + } + co_await commit; + for (auto& reply : replies) { + TLogPeekReply result = co_await timeoutError(reply.getFuture(), 0.2); + ASSERT(result.minKnownCommittedVersion == 100); + } + co_await waitForAll(peeks); + fixture.assertNoDiskCommit(); +} + +TEST_CASE("/NativeCDC/TLogCommittedFrontier/OutOfOrderCommit") { + CommittedPeekTestFixture fixture; + Promise reply; + Future peek = fixture.peek(reply); + Future commit = fixture.commitFrontier(100, 101); + ASSERT(!commit.isReady()); + ASSERT(!reply.getFuture().isReady()); + TLogPeekReply result = co_await timeoutError(reply.getFuture(), 0.2); + co_await peek; + ASSERT(result.minKnownCommittedVersion == 100); + ASSERT(!commit.isReady()); + commit.cancel(); + ASSERT(commit.isReady() && commit.isError()); + ASSERT(commit.getError().code() == error_code_actor_cancelled); + fixture.assertNoDiskCommit(); +} + +TEST_CASE("/NativeCDC/TLogCommittedFrontier/Bypass") { + for (int scenario = 0; scenario < 6; ++scenario) { + CommittedPeekTestFixture fixture(false); + TLogPeekRequest req = fixture.request(); + if (scenario == 0) { + req.end = 101; + } else if (scenario == 1) { + req.returnIfBlocked = true; + } else if (scenario == 2) { + req.onlySpilled = true; + } else if (scenario == 3) { + req.replyByteLimit = 0; + } else if (scenario == 4) { + req.tag = Tag(0, 0); + } else { + fixture.stop(); + } + fixture.addMessage(req.tag); + Promise reply; + Future peek = fixture.peek(reply, req); + TLogPeekReply result = co_await timeoutError(reply.getFuture(), 0.2); + co_await peek; + ASSERT(result.minKnownCommittedVersion == 99); + ASSERT(fixture.timedOutPeeks() == 0); + } +} + +TEST_CASE("/NativeCDC/TLogCommittedFrontier/Popped") { + CommittedPeekTestFixture fixture; + fixture.popPastBegin(); + Promise reply; + Future peek = fixture.peek(reply); + TLogPeekReply result = co_await timeoutError(reply.getFuture(), 0.2); + co_await peek; + ASSERT(result.popped.present() && result.popped.get() == 101); + ASSERT(result.minKnownCommittedVersion == 99); +} + TEST_CASE("/NativeCDC/TLogPeekReplyLimit") { ASSERT(tLogPeekReplyByteLimit(Tag(tagLocalityCDC, 0), 0) == 0); ASSERT(tLogPeekReplyByteLimit(Tag(tagLocalityCDC, 0), SERVER_KNOBS->MAXIMUM_PEEK_BYTES) == From 908e92f1047789597b4d7ffba95f9d2c9bd8b611 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Wed, 2 Sep 2026 23:09:42 -0700 Subject: [PATCH 015/170] Set the protocol version when constructing committed-peek test messages --- fdbserver/tlog/TLogServer.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fdbserver/tlog/TLogServer.cpp b/fdbserver/tlog/TLogServer.cpp index 570dc72b92b..01b1430649e 100644 --- a/fdbserver/tlog/TLogServer.cpp +++ b/fdbserver/tlog/TLogServer.cpp @@ -4593,7 +4593,7 @@ class CommittedPeekTestFixture : NonCopyable { } void addMessage(Tag tag = cdcTag()) { - BinaryWriter message(Unversioned()); + BinaryWriter message(AssumeVersion(g_network->protocolVersion())); message << int32_t(0) << uint32_t(1) << uint16_t(1) << tag; message << MutationRef(MutationRef::SetValue, "frontier-key"_sr, "frontier-value"_sr); *reinterpret_cast(message.getData()) = message.getLength() - sizeof(int32_t); From fde4c5483f3dd40513d30c66faaed8543f16e9c2 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Wed, 2 Sep 2026 23:16:49 -0700 Subject: [PATCH 016/170] Initialize direct commit request timestamps in simulated peek tests --- fdbserver/tlog/TLogServer.cpp | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/fdbserver/tlog/TLogServer.cpp b/fdbserver/tlog/TLogServer.cpp index 01b1430649e..c3a2020eb48 100644 --- a/fdbserver/tlog/TLogServer.cpp +++ b/fdbserver/tlog/TLogServer.cpp @@ -4603,6 +4603,8 @@ class CommittedPeekTestFixture : NonCopyable { Future commitFrontier(Version frontier, Version previous = 99) { TLogCommitRequest req; + // A real request is timestamped by transport delivery; this fixture invokes the handler directly. + req.setRequestTime(g_network->timer()); req.prevVersion = previous; req.version = previous + 1; req.knownCommittedVersion = std::max(99, frontier); @@ -4747,9 +4749,9 @@ TEST_CASE("/NativeCDC/TLogCommittedFrontier/MultipleWaiters") { for (auto& reply : replies) { peeks.push_back(fixture.peek(reply)); } + co_await delay(0.02); co_await fixture.commitFrontier(98); co_await fixture.commitFrontier(99); - co_await delay(0.02); ASSERT(fixture.frontier() == 99); for (auto& reply : replies) { ASSERT(!reply.getFuture().isReady()); @@ -4771,6 +4773,7 @@ TEST_CASE("/NativeCDC/TLogCommittedFrontier/OutOfOrderCommit") { CommittedPeekTestFixture fixture; Promise reply; Future peek = fixture.peek(reply); + co_await delay(0.02); Future commit = fixture.commitFrontier(100, 101); ASSERT(!commit.isReady()); ASSERT(!reply.getFuture().isReady()); From e9dab1120a2a8a55fdec2dc630a1ea429bb8bcf4 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Thu, 3 Sep 2026 00:24:49 -0700 Subject: [PATCH 017/170] Reserve committed-peek test waiter storage --- fdbserver/tlog/TLogServer.cpp | 1 + 1 file changed, 1 insertion(+) diff --git a/fdbserver/tlog/TLogServer.cpp b/fdbserver/tlog/TLogServer.cpp index c3a2020eb48..53bf4d8df50 100644 --- a/fdbserver/tlog/TLogServer.cpp +++ b/fdbserver/tlog/TLogServer.cpp @@ -4746,6 +4746,7 @@ TEST_CASE("/NativeCDC/TLogCommittedFrontier/MultipleWaiters") { CommittedPeekTestFixture fixture; std::vector> replies(8); std::vector> peeks; + peeks.reserve(replies.size()); for (auto& reply : replies) { peeks.push_back(fixture.peek(reply)); } From ddae261e4bef66dcc1653037ed713d6c42d04449 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Thu, 3 Sep 2026 10:21:01 -0700 Subject: [PATCH 018/170] Prioritize native CDC acknowledgement verification --- fdbserver/cdcproxy/CDCProxy.cpp | 11 ++++++----- 1 file changed, 6 insertions(+), 5 deletions(-) diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index b8c6fd79e6d..83b45950ca7 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -436,13 +436,13 @@ Optional clipCDCMutation(MutationRef const& mutation, KeyRangeRef c return Optional(); } -FDB_BOOLEAN_PARAM(PrioritizeConsume); +FDB_BOOLEAN_PARAM(PrioritizeDrain); AsyncResult readCDCStreamState(Database cx, CDCStreamId streamId, UID expectedProxyId, bool requireKeys, - PrioritizeConsume prioritizeConsume = PrioritizeConsume::False) { + PrioritizeDrain prioritizeDrain = PrioritizeDrain::False) { if (streamId == 0) { throw client_invalid_operation(); } @@ -453,7 +453,7 @@ AsyncResult readCDCStreamState(Database cx, try { tr.setOption(FDBTransactionOptions::READ_LOCK_AWARE); tr.setOption(FDBTransactionOptions::READ_SYSTEM_KEYS); - if (prioritizeConsume) { + if (prioritizeDrain) { // Draining committed CDC data must continue while ordinary transaction admission is throttled. tr.setOption(FDBTransactionOptions::PRIORITY_SYSTEM_IMMEDIATE); } @@ -1527,7 +1527,7 @@ Future CDCProxy::consume(CDCConsumeRequest request) { --stream->activeConsumes; }); const CDCStreamReadState metadata = - co_await readCDCStreamState(cx, request.cursor.streamId, id, true, PrioritizeConsume::True); + co_await readCDCStreamState(cx, request.cursor.streamId, id, true, PrioritizeDrain::True); CODE_PROBE(stream->minVersion < metadata.minVersion, "Native CDC consume reconciles a durable acknowledgement"); reconcileStreamMinVersion(stream, metadata.minVersion); if (request.cursor.lastConsumedVersion > stream->bufferedThrough) { @@ -1611,7 +1611,8 @@ Future CDCProxy::acknowledge(CDCAckRequest request) { if (request.version < 0 || request.version >= std::numeric_limits::max() - 1) { throw client_invalid_operation(); } - const CDCStreamReadState metadata = co_await readCDCStreamState(cx, request.streamId, id, false); + const CDCStreamReadState metadata = + co_await readCDCStreamState(cx, request.streamId, id, false, PrioritizeDrain::True); if (metadata.minVersion <= request.version) { throw client_invalid_operation(); } From 1d4a5793bd12ddfcc6b1bda5eb795aff6706ee66 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Thu, 3 Sep 2026 23:46:17 -0700 Subject: [PATCH 019/170] Prefetch one bounded native CDC batch after delivery --- fdbserver/cdcproxy/CDCProxy.cpp | 518 ++++++++++++++++++++++++++++++-- 1 file changed, 500 insertions(+), 18 deletions(-) diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index ea906b165fb..134102dd2a0 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -72,6 +72,54 @@ struct CDCTagInterval { : tag(tag), begin(begin), end(end), bufferedThrough(begin - 1) {} }; +struct CDCBufferedTag; + +// Speculative buffering never proves a client cursor and never creates another read-ahead credit. +class CDCStreamReadAhead { + enum class State { Idle, Armed, Claimed }; + State state = State::Idle; + Version issuedReplyThrough = invalidVersion; + Version creditThrough = invalidVersion; + CDCBufferedTag const* claimedTag = nullptr; + +public: + bool provesCursor(Version cursor, Version minVersion) const { + return cursor <= std::max(issuedReplyThrough, minVersion - 1); + } + bool issueReply(Version through, Version bufferedThrough, Version minVersion, bool hasMutations) { + const bool advanced = !provesCursor(through, minVersion); + if (state == State::Armed && creditThrough != bufferedThrough) { + cancel(); + } + issuedReplyThrough = std::max(issuedReplyThrough, through); + if (advanced && hasMutations && through == bufferedThrough && state == State::Idle) { + state = State::Armed; + creditThrough = through; + return true; + } + return false; + } + bool armedFor(Version through) const { return state == State::Armed && creditThrough == through; } + bool claim(CDCBufferedTag const* tag, Version through) { + if (!armedFor(through)) { + return false; + } + state = State::Claimed; + claimedTag = tag; + return true; + } + bool claimedBy(CDCBufferedTag const* tag) const { return state == State::Claimed && claimedTag == tag; } + void finish(CDCBufferedTag const* tag) { + if (claimedBy(tag)) { + cancel(); + } + } + void cancel() { + state = State::Idle; + claimedTag = nullptr; + } +}; + // Proxy-owned state for one assigned stream. In-flight actors may retain it after active becomes false. struct CDCBufferedStream : ReferenceCounted { CDCStreamId streamId; @@ -86,6 +134,7 @@ struct CDCBufferedStream : ReferenceCounted { int64_t bufferedBytes = 0; int readDemand = 0; int activeConsumes = 0; + CDCStreamReadAhead readAhead; std::vector tagIntervals; std::deque> mutations; AsyncTrigger changed; @@ -110,6 +159,48 @@ struct CDCBufferedTag : ReferenceCounted { explicit CDCBufferedTag(Tag tag) : tag(tag) {} }; +bool hasCDCReadInterest(Reference const& stream, Reference const& tag) { + return stream->readDemand > 0 || stream->readAhead.claimedBy(tag.getPtr()); +} + +Optional nextCDCPrefetchVersion(Reference const& stream, + Reference const& tag) { + if (!stream->active || !stream->initialized || stream->bufferLimitExceeded || + !stream->readAhead.armedFor(stream->bufferedThrough)) { + return {}; + } + const Version next = std::max(stream->minVersion, stream->bufferedThrough + 1); + for (auto const& interval : stream->tagIntervals) { + if (interval.begin <= next && next < interval.end) { + return interval.tag == tag->tag ? Optional(next) : Optional(); + } + } + return {}; +} + +// A tag has one buffering actor. Its claim retains the exact stream objects, never a replacement with the same ID. +class CDCReadAheadPass : NonCopyable { + Reference tag; + std::vector> streams; + +public: + explicit CDCReadAheadPass(Reference tag) : tag(tag) {} + ~CDCReadAheadPass() { + for (auto const& stream : streams) { + stream->readAhead.finish(tag.getPtr()); + } + } + bool empty() const { return streams.empty(); } + void claim(Reference stream) { + if (nextCDCPrefetchVersion(stream, tag).present() && + stream->readAhead.claim(tag.getPtr(), stream->bufferedThrough)) { + streams.push_back(stream); + } + } +}; + +FDB_BOOLEAN_PARAM(Prefetch); + // One stream's frontier and estimated materialization cost while selecting work for a single tag-buffering pass. struct CDCBufferCandidate { CDCStreamId streamId; @@ -365,6 +456,7 @@ Version retiredTagPopTarget(Version retiredVersion, Optional safePopVer } class CDCProxy { + friend class CDCProxyPrefetchTest; UID id; Database cx; Reference const> dbInfo; @@ -411,6 +503,7 @@ class CDCProxy { void refreshStreamTags(Reference stream); Optional nextTagReadVersionForStream(Reference tag, Reference stream); Optional nextTagReadVersion(Reference tag); + Optional nextTagPrefetchVersion(Reference tag); void advanceTagBufferedThrough(Reference tag, Version bufferedThrough, std::unordered_set const& selectedStreamIds); @@ -444,9 +537,19 @@ class CDCProxy { CDCBufferSelection const& selection, int64_t rawPeekReservation, FlowLock::Releaser& reservation, - int64_t bufferLimit); + int64_t bufferLimit, + Prefetch prefetch, + Future invalidated); Future rotateContendedPeek(); - Future bufferTagPass(Reference tag, Version begin); + Future bufferTagPass(Reference tag, Version begin, Prefetch prefetch); + Future bufferTagCursor(Reference tag, + Version begin, + Reference cursor, + Future logSystemChanged, + Prefetch prefetch); + CDCProxy() + : logSystem(makeReference>>()), + bufferLock(SERVER_KNOBS->CDC_PROXY_BUFFER_BYTES), actors(false) {} Future bufferTag(Reference tag); Future initializeStream(Reference stream); Future waitForBufferedVersion(Reference stream, Version version); @@ -605,6 +708,9 @@ void CDCProxy::refreshLogSystem() { lastLogSystemConfig = info.logSystemConfig; } if (logSystemChanged) { + for (auto const& [streamId, stream] : streams) { + stream->readAhead.cancel(); + } popLogSystemChanged.trigger(); } } @@ -786,6 +892,7 @@ void CDCProxy::deactivateStream(Reference stream) { CODE_PROBE(stream->readDemand > 0, "CDC proxy wakes pending consume when stream is unassigned"); CODE_PROBE(true, "CDC proxy drops removed or reassigned stream state"); stream->active = false; + stream->readAhead.cancel(); stream->changed.trigger(); detachStreamFromTags(stream); clearBufferedMutations(stream); @@ -802,7 +909,7 @@ void CDCProxy::refreshStreamTags(Reference stream) { Optional CDCProxy::nextTagReadVersionForStream(Reference tag, Reference stream) { - if (!stream->active || !stream->initialized || stream->bufferLimitExceeded || stream->readDemand == 0) { + if (!stream->active || !stream->initialized || stream->bufferLimitExceeded || !hasCDCReadInterest(stream, tag)) { return Optional(); } Optional begin; @@ -833,6 +940,20 @@ Optional CDCProxy::nextTagReadVersion(Reference tag) { return begin; } +Optional CDCProxy::nextTagPrefetchVersion(Reference tag) { + Optional begin; + for (const CDCStreamId streamId : tag->streamIds) { + auto stream = streams.find(streamId); + if (stream != streams.end()) { + const auto next = nextCDCPrefetchVersion(stream->second, tag); + if (next.present() && (!begin.present() || next.get() < begin.get())) { + begin = next; + } + } + } + return begin; +} + void CDCProxy::advanceTagBufferedThrough(Reference tag, Version bufferedThrough, std::unordered_set const& selectedStreamIds) { @@ -842,7 +963,7 @@ void CDCProxy::advanceTagBufferedThrough(Reference tag, continue; } auto stream = streams.find(streamId); - if (stream == streams.end() || !stream->second->active || stream->second->readDemand == 0) { + if (stream == streams.end() || !stream->second->active || !hasCDCReadInterest(stream->second, tag)) { continue; } for (auto& interval : stream->second->tagIntervals) { @@ -913,7 +1034,7 @@ void CDCProxy::visitBufferedMutations(Reference tag, continue; } auto stream = streams.find(streamId); - if (stream == streams.end() || !stream->second->active || stream->second->readDemand == 0 || + if (stream == streams.end() || !stream->second->active || !hasCDCReadInterest(stream->second, tag) || !stream->second->keys.present()) { continue; } @@ -1053,7 +1174,9 @@ Future CDCProxy::materializeBufferSelection(Reference invalidated) { const int64_t materializationReservation = reservation.remaining - rawPeekReservation; ASSERT_GE(materializationReservation, 0); if (selection.selectedBytes <= materializationReservation) { @@ -1062,6 +1185,9 @@ Future CDCProxy::materializeBufferSelection(ReferenceonChange(), tag->stopped.onTrigger(), @@ -1078,6 +1204,9 @@ Future CDCProxy::materializeBufferSelection(Referenceactive) { co_return CDCBufferTagPassResult::STOP; } + if (prefetch && invalidated.isReady()) { + co_return CDCBufferTagPassResult::RETRY; + } std::unordered_map batches = bufferMessages(tag, cursor, throughVersion, selection.selectedStreamIds); @@ -1114,13 +1243,39 @@ Future CDCProxy::materializeBufferSelection(Reference CDCProxy::bufferTagPass(Reference tag, Version begin) { - const int64_t bufferLimit = SERVER_KNOBS->CDC_PROXY_BUFFER_BYTES; +Future CDCProxy::bufferTagPass(Reference tag, + Version begin, + Prefetch prefetch) { Reference consumer = logSystem->get(); Future logSystemChanged = logSystem->onChange(); // CDC ReplayMultiCursor instances disable constructor prefetch, so constructing this cursor cannot issue a peek // before the proxy has reserved memory for every reply arena that its replicated read may retain. Reference cursor = consumer->peekSingle(id, begin, tag->tag, {}); + co_return co_await bufferTagCursor(tag, begin, std::move(cursor), logSystemChanged, prefetch); +} + +Future CDCProxy::bufferTagCursor(Reference tag, + Version begin, + Reference cursor, + Future logSystemChanged, + Prefetch prefetch) { + const int64_t bufferLimit = SERVER_KNOBS->CDC_PROXY_BUFFER_BYTES; + CDCReadAheadPass readAhead(tag); + if (prefetch) { + for (const CDCStreamId streamId : tag->streamIds) { + auto stream = streams.find(streamId); + if (stream != streams.end()) { + readAhead.claim(stream->second); + } + } + } + if (prefetch && readAhead.empty()) { + co_return CDCBufferTagPassResult::RETRY; + } + Future prefetchDeadline = prefetch ? delay(SERVER_KNOBS->BLOCKING_PEEK_TIMEOUT) : Never(); + Future invalidated = + prefetch ? logSystemChanged || tag->refresh.onTrigger() || tag->stopped.onTrigger() || prefetchDeadline + : Never(); cursor->setReplyByteLimit(SERVER_KNOBS->MAXIMUM_PEEK_BYTES); const int64_t retainedReplyCount = cursor->getMaxRetainedReplyCount(); Optional limits = @@ -1133,6 +1288,9 @@ Future CDCProxy::bufferTagPass(Reference const int64_t hardBufferedBatchLimit = limits.get().hardBufferedBytes; const int64_t preferredBufferedBatch = limits.get().preferredBufferedBytes; const int64_t passReservation = limits.get().reservationBytes; + if (prefetch && (bufferLock.waiters() != 0 || bufferLock.available() < passReservation)) { + co_return CDCBufferTagPassResult::RETRY; + } if (bufferLock.available() < passReservation) { CODE_PROBE(true, "CDC proxy applies shared buffer backpressure"); peekCapacityContended.trigger(); @@ -1150,7 +1308,7 @@ Future CDCProxy::bufferTagPass(Reference FlowLock::Releaser reservation(bufferLock, passReservation); recordBufferUsage(); // If capacity and a generation change became ready together, discard the cursor built from the old topology. - if (logSystemChanged.isReady()) { + if (logSystemChanged.isReady() || (prefetch && invalidated.isReady())) { co_return CDCBufferTagPassResult::RETRY; } if (!cursor->hasMessage()) { @@ -1161,8 +1319,9 @@ Future CDCProxy::bufferTagPass(Reference logSystemChanged, tag->stopped.onTrigger(), tag->refresh.onTrigger(), - rotateContendedPeek()); - if (result.index() == 1 || result.index() == 3 || result.index() == 4) { + rotateContendedPeek(), + prefetchDeadline); + if (result.index() == 1 || result.index() == 3 || result.index() == 4 || result.index() == 5) { co_return CDCBufferTagPassResult::RETRY; } if (result.index() == 2) { @@ -1176,6 +1335,9 @@ Future CDCProxy::bufferTagPass(Reference co_return CDCBufferTagPassResult::RETRY; } } + if (prefetch && invalidated.isReady()) { + co_return CDCBufferTagPassResult::RETRY; + } // A newly constructed replay cursor can already contain messages, especially after log-generation // changes. Initialize its reader even when getMore() was unnecessary. cursor->setProtocolVersion(g_network->protocolVersion()); @@ -1197,7 +1359,7 @@ Future CDCProxy::bufferTagPass(Reference CODE_PROBE(cursor->hasMessage() && cursor->version().version > committedThrough, "CDC proxy waits for peeked mutations to become committed"); if (throughVersion < begin) { - co_return CDCBufferTagPassResult::WAIT_FOR_COMMIT; + co_return prefetch ? CDCBufferTagPassResult::RETRY : CDCBufferTagPassResult::WAIT_FOR_COMMIT; } selection = selectBufferCandidatesForTag(tag, prefix, preferredBufferedBatch, hardBufferedBatchLimit); } @@ -1205,12 +1367,17 @@ Future CDCProxy::bufferTagPass(Reference co_return CDCBufferTagPassResult::RETRY; } co_return co_await materializeBufferSelection( - tag, cursor, throughVersion, selection, rawPeekReservation, reservation, bufferLimit); + tag, cursor, throughVersion, selection, rawPeekReservation, reservation, bufferLimit, prefetch, invalidated); } Future CDCProxy::bufferTag(Reference tag) { while (tag->active) { Optional begin = nextTagReadVersion(tag); + Prefetch prefetch = Prefetch::False; + if (!begin.present()) { + begin = nextTagPrefetchVersion(tag); + prefetch = Prefetch::True; + } if (!begin.present()) { auto waitForDemand = co_await race(tag->stopped.onTrigger(), tag->refresh.onTrigger()); if (waitForDemand.index() == 0) { @@ -1227,7 +1394,7 @@ Future CDCProxy::bufferTag(Reference tag) { continue; } - const CDCBufferTagPassResult result = co_await bufferTagPass(tag, begin.get()); + const CDCBufferTagPassResult result = co_await bufferTagPass(tag, begin.get(), prefetch); if (result == CDCBufferTagPassResult::STOP) { co_return; } @@ -1604,10 +1771,9 @@ Future CDCProxy::consume(CDCConsumeRequest request) { co_await readCDCStreamState(cx, request.cursor.streamId, id, true, PrioritizeConsume::True); CODE_PROBE(stream->minVersion < metadata.minVersion, "Native CDC consume reconciles a durable acknowledgement"); reconcileStreamMinVersion(stream, metadata.minVersion); - if (request.cursor.lastConsumedVersion > stream->bufferedThrough) { - // A cursor is trusted only when this owner has delivered through it or when it is covered by the durable - // acknowledgement watermark used to initialize bufferedThrough. This prevents a fabricated cursor from - // making the proxy retain every intervening tagged mutation while trying to reach an unproven position. + if (!stream->readAhead.provesCursor(request.cursor.lastConsumedVersion, stream->minVersion)) { + // Prefetched data is not proof of delivery. A cursor must have been issued in a reply or covered by a + // durable acknowledgement; otherwise it could skip unread data and manufacture more read-ahead work. if (request.cursor.lastConsumedVersion > metadata.readVersion) { CODE_PROBE(true, "CDC proxy rejects a consume cursor beyond its transaction read version"); } else { @@ -1671,7 +1837,14 @@ Future CDCProxy::consume(CDCConsumeRequest request) { throw server_overloaded(); } reply.lastConsumedVersion = selectedCDCConsumeReplyThrough(selection, stream->bufferedThrough); + // Record proof before send(), whose callbacks may run synchronously. Empty or capped replies do not + // extend the speculative horizon; neither does replaying an already issued cursor. + const bool armed = stream->readAhead.issueReply( + reply.lastConsumedVersion, stream->bufferedThrough, stream->minVersion, !reply.mutations.empty()); request.reply.send(reply); + if (armed) { + refreshStreamTags(stream); + } } catch (Error& e) { if (e.code() == error_code_actor_cancelled) { throw; @@ -2021,6 +2194,315 @@ Future cdcProxyServer(CDCProxyInterface proxy, } } +namespace { + +class CDCPrefetchTestCursor final : public IReplayPeekCursor, public ReferenceCounted { + Standalone payload; + Optional input; + Future ready; + LogMessageVersion position{ 100 }; + bool fetched = false; + bool done = false; + bool containsMutation; + int fetches = 0; + +public: + explicit CDCPrefetchTestCursor(Future ready, bool containsMutation = true) + : ready(ready), containsMutation(containsMutation) { + BinaryWriter writer(AssumeVersion(g_network->protocolVersion())); + writer << MutationRef(MutationRef::SetValue, "k"_sr, "value"_sr); + payload = writer.toValue(); + } + int fetchCount() const { return fetches; } + void setProtocolVersion(ProtocolVersion version) override { + input = ArenaReader(payload.arena(), payload, AssumeVersion(version)); + } + bool hasMessage() const override { return fetched && !done && containsMutation; } + VectorRef getTags() const override { return {}; } + Arena& arena() override { return payload.arena(); } + ArenaReader* reader() override { return &input.get(); } + StringRef getMessage() override { return payload; } + StringRef getMessageWithTags() override { return payload; } + void nextMessage() override { + done = true; + position = LogMessageVersion(101); + } + Future getMore(TaskPriority taskID) override { + ++fetches; + co_await ready; + fetched = true; + if (!containsMutation) { + position = LogMessageVersion(101); + } + co_return Void(); + } + bool isExhausted() const override { return fetched && !hasMessage(); } + LogMessageVersion const& version() const override { return position; } + Version popped() const override { return 0; } + Version getMinKnownCommittedVersion() const override { return 100; } + int64_t getMaxRetainedReplyCount() const override { return 1; } + void setReplyByteLimit(int limitBytes) override { ASSERT_GT(limitBytes, payload.size()); } + Optional getPrimaryPeekLocation() const override { return {}; } + Optional getCurrentPeekLocation() const override { return {}; } + Version getMaxKnownVersion() const override { return 100; } + Reference cloneNoMore() override { + auto clone = makeReference(Void(), containsMutation); + clone->position = position; + clone->fetched = fetched; + clone->done = done; + return clone; + } + void advanceTo(LogMessageVersion next) override { + if (next > position) { + nextMessage(); + } + } + void addref() override { ReferenceCounted::addref(); } + void delref() override { ReferenceCounted::delref(); } +}; + +class CDCProxyPrefetchTest { + CDCProxy proxy; + Reference tag = makeReference(Tag(tagLocalityCDC, 0)); + + Reference addStream(CDCStreamId id) { + auto stream = makeReference(id); + stream->initialized = true; + stream->minVersion = 1; + stream->bufferedThrough = 99; + stream->keys = KeyRangeRef("a"_sr, "z"_sr); + stream->tagIntervals.emplace_back(tag->tag, 1, 200); + stream->tagIntervals.back().bufferedThrough = 99; + ASSERT(stream->readAhead.issueReply(99, 99, 1, true)); + proxy.streams[id] = stream; + proxy.tags[tag->tag] = tag; + tag->streamIds.insert(id); + return stream; + } + +public: + static Future publish() { + CDCProxyPrefetchTest test; + auto first = test.addStream(1); + auto second = test.addStream(2); + auto dormant = test.addStream(3); + dormant->readAhead.cancel(); + auto cursor = makeReference(Void()); + ASSERT(!test.proxy.nextTagReadVersion(test.tag).present()); + ASSERT_EQ(test.proxy.nextTagPrefetchVersion(test.tag).get(), 100); + co_await test.proxy.bufferTagCursor(test.tag, 100, cursor, Never(), Prefetch::True); + ASSERT_EQ(cursor->fetchCount(), 1); + for (auto const& stream : { first, second }) { + ASSERT_EQ(stream->bufferedThrough, 100); + ASSERT_EQ(stream->mutations.size(), 1); + ASSERT(!stream->readAhead.claimedBy(test.tag.getPtr())); + ASSERT(!stream->readAhead.provesCursor(100, stream->minVersion)); + ASSERT(!stream->readAhead.armedFor(100)); + } + ASSERT_EQ(dormant->bufferedThrough, 99); + ASSERT(dormant->mutations.empty()); + ASSERT_EQ(dormant->bufferedBytes, 0); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), test.proxy.bufferedBytes); + ASSERT_EQ(test.proxy.bufferedBytes, first->bufferedBytes + second->bufferedBytes); + ASSERT_GT(test.proxy.bufferedBytes, 0); + ASSERT(!test.proxy.nextTagPrefetchVersion(test.tag).present()); + test.proxy.clearBufferedMutations(first); + test.proxy.clearBufferedMutations(second); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), 0); + co_return Void(); + } + + static Future capacity(bool queued = false) { + CDCProxyPrefetchTest test; + auto stream = test.addStream(1); + const int64_t limit = SERVER_KNOBS->CDC_PROXY_BUFFER_BYTES; + co_await test.proxy.bufferLock.take(TaskPriority::TLogPeekReply, limit); + FlowLock::Releaser held(test.proxy.bufferLock, limit); + Future waiting; + if (queued) { + waiting = test.proxy.bufferLock.take(TaskPriority::TLogPeekReply, limit); + ASSERT(!waiting.isReady()); + const auto passLimit = calculateBufferPassLimits(limit, SERVER_KNOBS->MAXIMUM_PEEK_BYTES, 1); + held.release(std::min(passLimit.get().reservationBytes, limit - 1)); + ASSERT_GT(test.proxy.bufferLock.waiters(), 0); + } + auto cursor = makeReference(Never()); + co_await test.proxy.bufferTagCursor(test.tag, 100, cursor, Never(), Prefetch::True); + ASSERT_EQ(cursor->fetchCount(), 0); + ASSERT(!stream->readAhead.armedFor(99)); + ASSERT(!stream->readAhead.claimedBy(test.tag.getPtr())); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), held.remaining); + if (queued) { + waiting.cancel(); + } + co_return Void(); + } + + static Future extraCapacity() { + CDCProxyPrefetchTest test; + auto stream = test.addStream(1); + const int64_t limit = SERVER_KNOBS->CDC_PROXY_BUFFER_BYTES; + const auto passLimit = calculateBufferPassLimits(limit, SERVER_KNOBS->MAXIMUM_PEEK_BYTES, 1).get(); + if (passLimit.preferredBufferedBytes == passLimit.hardBufferedBytes) { + // Small randomized budgets have no valid batch requiring an additional reservation. + ASSERT_EQ(passLimit.reservationBytes, limit); + co_return Void(); + } + co_await test.proxy.bufferLock.take(TaskPriority::TLogPeekReply, limit - passLimit.reservationBytes); + FlowLock::Releaser held(test.proxy.bufferLock, limit - passLimit.reservationBytes); + co_await test.proxy.bufferLock.take(TaskPriority::TLogPeekReply, passLimit.reservationBytes); + FlowLock::Releaser reservation(test.proxy.bufferLock, passLimit.reservationBytes); + CDCReadAheadPass pass(test.tag); + pass.claim(stream); + CDCBufferSelection selection; + selection.selectedStreamIds.insert(1); + selection.selectedBytes = passLimit.preferredBufferedBytes + 1; + auto cursor = makeReference(Void()); + co_await test.proxy.materializeBufferSelection( + test.tag, cursor, 100, selection, passLimit.rawReplyBytes, reservation, limit, Prefetch::True, Never()); + ASSERT_EQ(test.proxy.bufferLock.waiters(), 0); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), limit); + ASSERT(stream->mutations.empty()); + ASSERT_EQ(stream->bufferedThrough, 99); + co_return Void(); + } + + static Future interrupted(int action) { + CDCProxyPrefetchTest test; + auto stream = test.addStream(1); + Promise ready; + Promise generationChanged; + auto cursor = makeReference(ready.getFuture()); + auto work = test.proxy.bufferTagCursor(test.tag, 100, cursor, generationChanged.getFuture(), Prefetch::True); + co_await delay(0); + ASSERT_EQ(cursor->fetchCount(), 1); + ASSERT(stream->readAhead.claimedBy(test.tag.getPtr())); + if (action == 0) { + test.tag->refresh.trigger(); + } else if (action == 1) { + generationChanged.send(Void()); + } else if (action == 2) { + test.proxy.deactivateStream(stream); + test.proxy.streams[1] = makeReference(1); + } else if (action == 3) { + work.cancel(); + } else { + // No consumer comes back: the one speculative peek expires without granting another credit. + ASSERT_EQ(action, 4); + } + if (action != 3) { + co_await work; + } + ASSERT_EQ(stream->bufferedThrough, 99); + ASSERT(stream->mutations.empty()); + ASSERT(!stream->readAhead.claimedBy(test.tag.getPtr())); + ASSERT(!stream->readAhead.armedFor(99)); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), 0); + co_return Void(); + } + + static Future empty() { + CDCProxyPrefetchTest test; + auto stream = test.addStream(1); + auto cursor = makeReference(Void(), false); + co_await test.proxy.bufferTagCursor(test.tag, 100, cursor, Never(), Prefetch::True); + ASSERT_EQ(cursor->fetchCount(), 1); + ASSERT_EQ(stream->bufferedThrough, 100); + ASSERT(stream->mutations.empty()); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), 0); + ASSERT(!test.proxy.nextTagPrefetchVersion(test.tag).present()); + co_return Void(); + } +}; + +} // namespace + +TEST_CASE("/NativeCDC/PrefetchCreditLifecycle") { + CDCStreamReadAhead credit; + auto tag = makeReference(Tag(tagLocalityCDC, 0)); + ASSERT(!credit.armedFor(99)); + ASSERT(!credit.issueReply(99, 99, 100, true)); // Already covered by the durable floor. + ASSERT(!credit.issueReply(100, 100, 100, false)); // Empty progress is proved, but grants no lookahead. + ASSERT(credit.provesCursor(100, 100)); + ASSERT(!credit.issueReply(101, 102, 100, true)); // A capped reply has not drained the buffered tail. + ASSERT(credit.issueReply(102, 102, 100, true)); + ASSERT(credit.claim(tag.getPtr(), 102)); + ASSERT(!credit.issueReply(103, 103, 100, true)); // No credit banking while a pass is active. + credit.finish(tag.getPtr()); + ASSERT(!credit.armedFor(103)); + ASSERT(!credit.issueReply(103, 103, 100, true)); // Replayed cursor. + ASSERT(credit.issueReply(104, 104, 100, true)); + credit.cancel(); + ASSERT(!credit.claim(tag.getPtr(), 104)); + ASSERT(!credit.provesCursor(105, 100)); + ASSERT(credit.provesCursor(105, 106)); + return Void(); +} + +TEST_CASE("/NativeCDC/PrefetchCreditTailAndTags") { + auto stream = makeReference(1); + auto first = makeReference(Tag(tagLocalityCDC, 0)); + auto second = makeReference(Tag(tagLocalityCDC, 1)); + stream->initialized = true; + stream->minVersion = 1; + stream->bufferedThrough = 99; + stream->tagIntervals.emplace_back(first->tag, 1, 101); + stream->tagIntervals.emplace_back(second->tag, 101, 200); + stream->tagIntervals[0].bufferedThrough = 99; + ASSERT(stream->readAhead.issueReply(99, 99, 1, true)); + ASSERT_EQ(nextCDCPrefetchVersion(stream, first).get(), 100); + ASSERT(!nextCDCPrefetchVersion(stream, second).present()); + { + CDCReadAheadPass pass(first); + pass.claim(stream); + ASSERT(hasCDCReadInterest(stream, first)); + ASSERT(!hasCDCReadInterest(stream, second)); + ASSERT(!nextCDCPrefetchVersion(stream, second).present()); + } + ASSERT(!stream->readAhead.claimedBy(first.getPtr())); + ASSERT(stream->readAhead.issueReply(100, 100, 1, true)); + stream->bufferedThrough = 102; // Real demand filled more data before the credit could start. + stream->tagIntervals[0].bufferedThrough = 100; + stream->tagIntervals[1].bufferedThrough = 102; + ASSERT(!nextCDCPrefetchVersion(stream, second).present()); + ASSERT(!stream->readAhead.issueReply(101, 102, 1, true)); // Capped reply cannot revive the old credit. + ASSERT(!nextCDCPrefetchVersion(stream, second).present()); + ASSERT(stream->readAhead.issueReply(102, 102, 1, true)); + ASSERT_EQ(nextCDCPrefetchVersion(stream, second).get(), 103); + return Void(); +} + +TEST_CASE("/NativeCDC/PrefetchMaterializesSharedTag") { + return CDCProxyPrefetchTest::publish(); +} +TEST_CASE("/NativeCDC/PrefetchDeclinesUnavailableCapacity") { + return CDCProxyPrefetchTest::capacity(); +} +TEST_CASE("/NativeCDC/PrefetchDeclinesQueuedCapacity") { + return CDCProxyPrefetchTest::capacity(true); +} +TEST_CASE("/NativeCDC/PrefetchDeclinesExtraCapacity") { + return CDCProxyPrefetchTest::extraCapacity(); +} +TEST_CASE("/NativeCDC/PrefetchRefreshCancels") { + return CDCProxyPrefetchTest::interrupted(0); +} +TEST_CASE("/NativeCDC/PrefetchGenerationChangeCancels") { + return CDCProxyPrefetchTest::interrupted(1); +} +TEST_CASE("/NativeCDC/PrefetchReplacementCancels") { + return CDCProxyPrefetchTest::interrupted(2); +} +TEST_CASE("/NativeCDC/PrefetchActorCancellationReleases") { + return CDCProxyPrefetchTest::interrupted(3); +} +TEST_CASE("/NativeCDC/PrefetchIdleDeadline") { + return CDCProxyPrefetchTest::interrupted(4); +} +TEST_CASE("/NativeCDC/PrefetchEmptyDoesNotRetry") { + return CDCProxyPrefetchTest::empty(); +} + TEST_CASE("/NativeCDC/ProxyMutationFiltering") { const KeyRangeRef keys("c"_sr, "m"_sr); From 0ddffbc09bbfd9d1f675c079148f7339fd87ddfc Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Thu, 3 Sep 2026 23:53:54 -0700 Subject: [PATCH 020/170] Use void coroutine returns in CDC prefetch tests --- fdbserver/cdcproxy/CDCProxy.cpp | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index 134102dd2a0..8544377f956 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -2234,7 +2234,7 @@ class CDCPrefetchTestCursor final : public IReplayPeekCursor, public ReferenceCo if (!containsMutation) { position = LogMessageVersion(101); } - co_return Void(); + co_return; } bool isExhausted() const override { return fetched && !hasMessage(); } LogMessageVersion const& version() const override { return position; } @@ -2309,7 +2309,7 @@ class CDCProxyPrefetchTest { test.proxy.clearBufferedMutations(first); test.proxy.clearBufferedMutations(second); ASSERT_EQ(test.proxy.bufferLock.activePermits(), 0); - co_return Void(); + co_return; } static Future capacity(bool queued = false) { @@ -2335,7 +2335,7 @@ class CDCProxyPrefetchTest { if (queued) { waiting.cancel(); } - co_return Void(); + co_return; } static Future extraCapacity() { @@ -2346,7 +2346,7 @@ class CDCProxyPrefetchTest { if (passLimit.preferredBufferedBytes == passLimit.hardBufferedBytes) { // Small randomized budgets have no valid batch requiring an additional reservation. ASSERT_EQ(passLimit.reservationBytes, limit); - co_return Void(); + co_return; } co_await test.proxy.bufferLock.take(TaskPriority::TLogPeekReply, limit - passLimit.reservationBytes); FlowLock::Releaser held(test.proxy.bufferLock, limit - passLimit.reservationBytes); @@ -2364,7 +2364,7 @@ class CDCProxyPrefetchTest { ASSERT_EQ(test.proxy.bufferLock.activePermits(), limit); ASSERT(stream->mutations.empty()); ASSERT_EQ(stream->bufferedThrough, 99); - co_return Void(); + co_return; } static Future interrupted(int action) { @@ -2398,7 +2398,7 @@ class CDCProxyPrefetchTest { ASSERT(!stream->readAhead.claimedBy(test.tag.getPtr())); ASSERT(!stream->readAhead.armedFor(99)); ASSERT_EQ(test.proxy.bufferLock.activePermits(), 0); - co_return Void(); + co_return; } static Future empty() { @@ -2411,7 +2411,7 @@ class CDCProxyPrefetchTest { ASSERT(stream->mutations.empty()); ASSERT_EQ(test.proxy.bufferLock.activePermits(), 0); ASSERT(!test.proxy.nextTagPrefetchVersion(test.tag).present()); - co_return Void(); + co_return; } }; From e16546013e94392795faf98987eeb73010907fb6 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Fri, 4 Sep 2026 02:02:07 -0700 Subject: [PATCH 021/170] Avoid cancelling native CDC peeks for unchanged read demand --- fdbserver/cdcproxy/CDCProxy.cpp | 256 +++++++++++++++++++++++++++++++- 1 file changed, 249 insertions(+), 7 deletions(-) diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index 8544377f956..2fc54f92446 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -501,6 +501,7 @@ class CDCProxy { void detachStreamFromTags(Reference stream); void deactivateStream(Reference stream); void refreshStreamTags(Reference stream); + void changeStreamReadDemand(Reference stream, int delta); Optional nextTagReadVersionForStream(Reference tag, Reference stream); Optional nextTagReadVersion(Reference tag); Optional nextTagPrefetchVersion(Reference tag); @@ -907,6 +908,42 @@ void CDCProxy::refreshStreamTags(Reference stream) { } } +void CDCProxy::changeStreamReadDemand(Reference stream, int delta) { + ASSERT(delta == 1 || delta == -1); + ASSERT_GE(stream->readDemand + delta, 0); + struct TagDemandSnapshot { + Reference tag; + Optional before; + bool refresh = true; + }; + std::vector snapshots; + snapshots.reserve(stream->tagIntervals.size()); + for (const auto& interval : stream->tagIntervals) { + auto found = tags.find(interval.tag); + if (found != tags.end()) { + snapshots.push_back({ found->second, nextTagReadVersion(found->second) }); + } + } + stream->readDemand += delta; + for (auto& snapshot : snapshots) { + const auto after = nextTagReadVersion(snapshot.tag); + // A claimed prefetch (or another consumer) may already cover this exact earliest frontier. + // Count-only changes then need not discard its cursor. Absent or changed work still wakes it. + snapshot.refresh = !snapshot.before.present() || !after.present() || snapshot.before.get() != after.get(); + } + // Trigger callbacks can synchronously replace tags or change stream membership. Decide before triggering, + // retain no map iterators across callbacks, and never apply an old object's equality proof to a replacement. + for (const auto& snapshot : snapshots) { + auto found = tags.find(snapshot.tag->tag); + if (found != tags.end()) { + auto current = found->second; + if (current.getPtr() != snapshot.tag.getPtr() || snapshot.refresh) { + current->refresh.trigger(); + } + } + } +} + Optional CDCProxy::nextTagReadVersionForStream(Reference tag, Reference stream) { if (!stream->active || !stream->initialized || stream->bufferLimitExceeded || !hasCDCReadInterest(stream, tag)) { @@ -1718,13 +1755,8 @@ Future CDCProxy::waitForBufferedVersion(Reference strea co_return; } - ++stream->readDemand; - refreshStreamTags(stream); - ScopeExit releaseReadDemand([this, stream]() { - ASSERT_GT(stream->readDemand, 0); - --stream->readDemand; - refreshStreamTags(stream); - }); + changeStreamReadDemand(stream, 1); + ScopeExit releaseReadDemand([this, stream]() { changeStreamReadDemand(stream, -1); }); while (stream->active && !stream->bufferLimitExceeded && stream->bufferedThrough < version) { co_await stream->changed.onTrigger(); } @@ -2281,6 +2313,188 @@ class CDCProxyPrefetchTest { } public: + static Future sameFrontierDemand(bool release, bool expire = false) { + CDCProxyPrefetchTest test; + auto stream = test.addStream(1); + Future waiter; + if (release) { + waiter = test.proxy.waitForBufferedVersion(stream, 100); + ASSERT_EQ(stream->readDemand, 1); + } + Promise ready; + auto cursor = makeReference(ready.getFuture()); + auto work = test.proxy.bufferTagCursor(test.tag, 100, cursor, Never(), Prefetch::True); + co_await delay(0); + ASSERT_EQ(cursor->fetchCount(), 1); + ASSERT(stream->readAhead.claimedBy(test.tag.getPtr())); + if (release) { + waiter.cancel(); // Exercise the actual waiter's ScopeExit, not a direct count mutation. + } else { + waiter = test.proxy.waitForBufferedVersion(stream, 100); + } + co_await delay(0); + ASSERT_EQ(stream->readDemand, release ? 0 : 1); + ASSERT(!work.isReady()); + ASSERT(stream->readAhead.claimedBy(test.tag.getPtr())); + if (expire) { + // Removing same-frontier demand must not promote the pass or renew its finite deadline. + ASSERT(release); + co_await work; + ASSERT_EQ(stream->bufferedThrough, 99); + ASSERT(stream->mutations.empty()); + } else { + ready.send(Void()); + co_await work; + if (!release) { + co_await waiter; + } + ASSERT_EQ(stream->bufferedThrough, 100); + ASSERT_EQ(stream->mutations.size(), 1); + ASSERT(!stream->readAhead.provesCursor(100, stream->minVersion)); + } + ASSERT_EQ(cursor->fetchCount(), 1); + ASSERT_EQ(stream->readDemand, 0); + ASSERT(!stream->readAhead.claimedBy(test.tag.getPtr())); + ASSERT(!test.proxy.nextTagPrefetchVersion(test.tag).present()); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), test.proxy.bufferedBytes); + test.proxy.clearBufferedMutations(stream); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), 0); + co_return; + } + + static Future sharedDemand(Version next) { + CDCProxyPrefetchTest test; + auto first = test.addStream(1); + auto second = test.addStream(2); + second->readAhead.cancel(); + second->bufferedThrough = next - 1; + second->tagIntervals.back().bufferedThrough = next - 1; + Promise ready; + auto cursor = makeReference(ready.getFuture()); + auto work = test.proxy.bufferTagCursor(test.tag, 100, cursor, Never(), Prefetch::True); + co_await delay(0); + ASSERT_EQ(cursor->fetchCount(), 1); + auto waiter = test.proxy.waitForBufferedVersion(second, next); + co_await delay(0); + if (next < 100) { + co_await work; + ASSERT_EQ(test.proxy.nextTagReadVersion(test.tag).get(), next); + ASSERT_EQ(first->bufferedThrough, 99); + ASSERT_EQ(second->bufferedThrough, next - 1); + ASSERT(first->mutations.empty()); + ASSERT(second->mutations.empty()); + waiter.cancel(); + } else { + ASSERT(!work.isReady()); + ASSERT_EQ(test.proxy.nextTagReadVersion(test.tag).get(), 100); + ready.send(Void()); + co_await work; + ASSERT_EQ(first->bufferedThrough, 100); + ASSERT_EQ(first->mutations.size(), 1); + if (next == 100) { + co_await waiter; + ASSERT_EQ(second->bufferedThrough, 100); + ASSERT_EQ(second->mutations.size(), 1); + } else { + ASSERT(!waiter.isReady()); + ASSERT_EQ(second->bufferedThrough, next - 1); + ASSERT(second->mutations.empty()); + waiter.cancel(); + } + } + ASSERT_EQ(second->readDemand, 0); + ASSERT(!first->readAhead.claimedBy(test.tag.getPtr())); + ASSERT(!first->readAhead.armedFor(first->bufferedThrough)); + ASSERT(!first->readAhead.provesCursor(100, first->minVersion)); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), test.proxy.bufferedBytes); + test.proxy.clearBufferedMutations(first); + test.proxy.clearBufferedMutations(second); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), 0); + co_return; + } + + static Future lastDemandLeaves() { + CDCProxyPrefetchTest test; + auto stream = test.addStream(1); + stream->readAhead.cancel(); + auto awakened = test.tag->refresh.onTrigger(); + auto waiter = test.proxy.waitForBufferedVersion(stream, 100); + ASSERT(awakened.isReady()); // No interest -> real demand still wakes a dormant tag. + auto cursor = makeReference(Never()); + auto work = test.proxy.bufferTagCursor(test.tag, 100, cursor, Never(), Prefetch::False); + co_await delay(0); + ASSERT_EQ(cursor->fetchCount(), 1); + ASSERT(!work.isReady()); + waiter.cancel(); + co_await work; + ASSERT_EQ(stream->readDemand, 0); + ASSERT(!test.proxy.nextTagReadVersion(test.tag).present()); + ASSERT_EQ(stream->bufferedThrough, 99); + ASSERT(stream->mutations.empty()); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), 0); + co_return; + } + + static Future absentFrontierDemand() { + CDCProxyPrefetchTest test; + auto stream = test.addStream(1); + stream->readAhead.cancel(); + stream->tagIntervals.back().end = 100; + auto acquired = test.tag->refresh.onTrigger(); + auto waiter = test.proxy.waitForBufferedVersion(stream, 100); + ASSERT(acquired.isReady()); // Two absent frontiers must not count as equal, eligible work. + ASSERT(!test.proxy.nextTagReadVersion(test.tag).present()); + auto released = test.tag->refresh.onTrigger(); + waiter.cancel(); + ASSERT(released.isReady()); + ASSERT_EQ(stream->readDemand, 0); + co_return; + } + + static Future replaceTagOnRefresh(CDCProxy* proxy, + Reference first, + Reference oldTag, + Reference replacement) { + co_await first->refresh.onTrigger(); + oldTag->active = false; + oldTag->stopped.trigger(); + proxy->tags[replacement->tag] = replacement; + co_return; + } + + static Future demandRefreshReplacement() { + CDCProxyPrefetchTest test; + auto stream = test.addStream(1); + stream->readAhead.cancel(); + stream->bufferedThrough = 98; + stream->tagIntervals[0].end = 100; + stream->tagIntervals[0].bufferedThrough = 98; + auto second = makeReference(Tag(tagLocalityCDC, 1)); + stream->tagIntervals.emplace_back(second->tag, 100, 200); + stream->tagIntervals.back().bufferedThrough = 99; + auto other = test.addStream(2); + test.tag->streamIds.erase(2); + other->tagIntervals[0].tag = second->tag; + second->streamIds = { 1, 2 }; + test.proxy.tags[second->tag] = second; + CDCReadAheadPass pass(second); + pass.claim(other); + ASSERT_EQ(test.proxy.nextTagReadVersion(second).get(), 100); + auto replacement = makeReference(second->tag); + replacement->streamIds = second->streamIds; + auto notified = replacement->refresh.onTrigger(); + auto replace = replaceTagOnRefresh(&test.proxy, test.tag, second, replacement); + auto waiter = test.proxy.waitForBufferedVersion(stream, 99); + // The first notification runs the replacement coroutine synchronously. The equal second frontier + // was observed on the old object, so it cannot suppress a notification to the replacement. + ASSERT(replace.isReady()); + ASSERT(notified.isReady()); + ASSERT(test.proxy.tags.at(second->tag).getPtr() == replacement.getPtr()); + waiter.cancel(); + ASSERT_EQ(stream->readDemand, 0); + co_return; + } + static Future publish() { CDCProxyPrefetchTest test; auto first = test.addStream(1); @@ -2503,6 +2717,34 @@ TEST_CASE("/NativeCDC/PrefetchEmptyDoesNotRetry") { return CDCProxyPrefetchTest::empty(); } +TEST_CASE("/NativeCDC/DemandRefresh/PrefetchAcquirePreservesPeek") { + return CDCProxyPrefetchTest::sameFrontierDemand(false); +} +TEST_CASE("/NativeCDC/DemandRefresh/PrefetchReleasePreservesPeek") { + return CDCProxyPrefetchTest::sameFrontierDemand(true); +} +TEST_CASE("/NativeCDC/DemandRefresh/PrefetchReleasePreservesDeadline") { + return CDCProxyPrefetchTest::sameFrontierDemand(true, true); +} +TEST_CASE("/NativeCDC/DemandRefresh/SharedEarlierRestartsPeek") { + return CDCProxyPrefetchTest::sharedDemand(50); +} +TEST_CASE("/NativeCDC/DemandRefresh/SharedSamePreservesPeek") { + return CDCProxyPrefetchTest::sharedDemand(100); +} +TEST_CASE("/NativeCDC/DemandRefresh/SharedLaterPreservesPeek") { + return CDCProxyPrefetchTest::sharedDemand(101); +} +TEST_CASE("/NativeCDC/DemandRefresh/LastDemandCancelsPeek") { + return CDCProxyPrefetchTest::lastDemandLeaves(); +} +TEST_CASE("/NativeCDC/DemandRefresh/AbsentFrontiersNotify") { + return CDCProxyPrefetchTest::absentFrontierDemand(); +} +TEST_CASE("/NativeCDC/DemandRefresh/TagReplacementNotifiesCurrentObject") { + return CDCProxyPrefetchTest::demandRefreshReplacement(); +} + TEST_CASE("/NativeCDC/ProxyMutationFiltering") { const KeyRangeRef keys("c"_sr, "m"_sr); From 671d7f28b898b8ceaabb4a92c1ee87444667fa44 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Fri, 4 Sep 2026 14:28:27 -0700 Subject: [PATCH 022/170] Deduplicate native CDC demand frontier snapshots --- fdbserver/cdcproxy/CDCProxy.cpp | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index 2fc54f92446..cbd43feaa6d 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -917,10 +917,13 @@ void CDCProxy::changeStreamReadDemand(Reference stream, int d bool refresh = true; }; std::vector snapshots; - snapshots.reserve(stream->tagIntervals.size()); + std::unordered_set seenTags; + const size_t maxSnapshots = std::min(stream->tagIntervals.size(), tags.size()); + snapshots.reserve(maxSnapshots); + seenTags.reserve(maxSnapshots); for (const auto& interval : stream->tagIntervals) { auto found = tags.find(interval.tag); - if (found != tags.end()) { + if (found != tags.end() && seenTags.insert(interval.tag).second) { snapshots.push_back({ found->second, nextTagReadVersion(found->second) }); } } From b08ddd1c350c23f4d3a0cd3808c4e04abb824a4e Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Fri, 4 Sep 2026 14:49:49 -0700 Subject: [PATCH 023/170] Reduce native CDC demand refresh work --- fdbserver/cdcproxy/CDCProxy.cpp | 82 +++++++++++++++++++++++++-------- 1 file changed, 63 insertions(+), 19 deletions(-) diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index cbd43feaa6d..2a6ff3e47b3 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -502,7 +502,10 @@ class CDCProxy { void deactivateStream(Reference stream); void refreshStreamTags(Reference stream); void changeStreamReadDemand(Reference stream, int delta); - Optional nextTagReadVersionForStream(Reference tag, Reference stream); + bool tagDemandChangeNeedsRefresh(Reference tag, Reference stream, int delta); + Optional nextTagReadVersionForStream(Reference tag, + Reference stream, + bool ignoreReadInterest = false); Optional nextTagReadVersion(Reference tag); Optional nextTagPrefetchVersion(Reference tag); void advanceTagBufferedThrough(Reference tag, @@ -908,13 +911,63 @@ void CDCProxy::refreshStreamTags(Reference stream) { } } +bool CDCProxy::tagDemandChangeNeedsRefresh(Reference tag, + Reference stream, + int delta) { + const bool beforeInterest = hasCDCReadInterest(stream, tag); + const bool afterInterest = stream->readDemand + delta > 0 || stream->readAhead.claimedBy(tag.getPtr()); + Optional before; + Optional after; + for (const CDCStreamId streamId : tag->streamIds) { + auto found = streams.find(streamId); + if (found == streams.end()) { + continue; + } + const bool changedStream = found->second.getPtr() == stream.getPtr(); + const auto candidate = nextTagReadVersionForStream(tag, found->second, changedStream); + if (!candidate.present()) { + continue; + } + if ((!changedStream || beforeInterest) && (!before.present() || candidate.get() < before.get())) { + before = candidate; + } + if ((!changedStream || afterInterest) && (!after.present() || candidate.get() < after.get())) { + after = candidate; + } + } + // Absent frontiers must still wake a dormant reader; only equal, present work preserves its peek. + return !before.present() || !after.present() || before.get() != after.get(); +} + void CDCProxy::changeStreamReadDemand(Reference stream, int delta) { ASSERT(delta == 1 || delta == -1); ASSERT_GE(stream->readDemand + delta, 0); + auto refreshCurrentTag = [this](Reference const& tag, bool refresh) { + auto found = tags.find(tag->tag); + if (found != tags.end() && (found->second.getPtr() != tag.getPtr() || refresh)) { + found->second->refresh.trigger(); + } + }; + // A stream with one tag needs neither scratch container on each consume wait. + if (stream->tagIntervals.size() <= 1) { + Reference tag; + bool refresh = false; + if (!stream->tagIntervals.empty()) { + auto found = tags.find(stream->tagIntervals.front().tag); + if (found != tags.end()) { + tag = found->second; + refresh = tagDemandChangeNeedsRefresh(tag, stream, delta); + } + } + stream->readDemand += delta; + if (tag) { + refreshCurrentTag(tag, refresh); + } + return; + } struct TagDemandSnapshot { Reference tag; - Optional before; - bool refresh = true; + bool refresh; }; std::vector snapshots; std::unordered_set seenTags; @@ -924,32 +977,23 @@ void CDCProxy::changeStreamReadDemand(Reference stream, int d for (const auto& interval : stream->tagIntervals) { auto found = tags.find(interval.tag); if (found != tags.end() && seenTags.insert(interval.tag).second) { - snapshots.push_back({ found->second, nextTagReadVersion(found->second) }); + snapshots.push_back({ found->second, tagDemandChangeNeedsRefresh(found->second, stream, delta) }); } } stream->readDemand += delta; - for (auto& snapshot : snapshots) { - const auto after = nextTagReadVersion(snapshot.tag); - // A claimed prefetch (or another consumer) may already cover this exact earliest frontier. - // Count-only changes then need not discard its cursor. Absent or changed work still wakes it. - snapshot.refresh = !snapshot.before.present() || !after.present() || snapshot.before.get() != after.get(); - } + // A claimed prefetch (or another consumer) may already cover the same frontier, avoiding a restart. // Trigger callbacks can synchronously replace tags or change stream membership. Decide before triggering, // retain no map iterators across callbacks, and never apply an old object's equality proof to a replacement. for (const auto& snapshot : snapshots) { - auto found = tags.find(snapshot.tag->tag); - if (found != tags.end()) { - auto current = found->second; - if (current.getPtr() != snapshot.tag.getPtr() || snapshot.refresh) { - current->refresh.trigger(); - } - } + refreshCurrentTag(snapshot.tag, snapshot.refresh); } } Optional CDCProxy::nextTagReadVersionForStream(Reference tag, - Reference stream) { - if (!stream->active || !stream->initialized || stream->bufferLimitExceeded || !hasCDCReadInterest(stream, tag)) { + Reference stream, + bool ignoreReadInterest) { + if (!stream->active || !stream->initialized || stream->bufferLimitExceeded || + (!ignoreReadInterest && !hasCDCReadInterest(stream, tag))) { return Optional(); } Optional begin; From d1777655d518a709593c098c1b453a6a7751772f Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Fri, 4 Sep 2026 16:21:38 -0700 Subject: [PATCH 024/170] Preserve later native CDC prefetch credits on shared tags --- fdbserver/cdcproxy/CDCProxy.cpp | 69 ++++++++++++++++++++++++++------- 1 file changed, 54 insertions(+), 15 deletions(-) diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index 2a6ff3e47b3..ebd5327edb0 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -191,9 +191,10 @@ class CDCReadAheadPass : NonCopyable { } } bool empty() const { return streams.empty(); } - void claim(Reference stream) { - if (nextCDCPrefetchVersion(stream, tag).present() && - stream->readAhead.claim(tag.getPtr(), stream->bufferedThrough)) { + void claim(Reference stream, Version begin) { + const auto next = nextCDCPrefetchVersion(stream, tag); + // A later frontier keeps its credit for a pass that can actually reach it. + if (next.present() && next.get() == begin && stream->readAhead.claim(tag.getPtr(), stream->bufferedThrough)) { streams.push_back(stream); } } @@ -1349,7 +1350,7 @@ Future CDCProxy::bufferTagCursor(ReferencestreamIds) { auto stream = streams.find(streamId); if (stream != streams.end()) { - readAhead.claim(stream->second); + readAhead.claim(stream->second, begin); } } } @@ -2279,15 +2280,16 @@ class CDCPrefetchTestCursor final : public IReplayPeekCursor, public ReferenceCo Standalone payload; Optional input; Future ready; - LogMessageVersion position{ 100 }; + Version messageVersion; + LogMessageVersion position; bool fetched = false; bool done = false; bool containsMutation; int fetches = 0; public: - explicit CDCPrefetchTestCursor(Future ready, bool containsMutation = true) - : ready(ready), containsMutation(containsMutation) { + explicit CDCPrefetchTestCursor(Future ready, bool containsMutation = true, Version version = 100) + : ready(ready), messageVersion(version), position(version), containsMutation(containsMutation) { BinaryWriter writer(AssumeVersion(g_network->protocolVersion())); writer << MutationRef(MutationRef::SetValue, "k"_sr, "value"_sr); payload = writer.toValue(); @@ -2304,28 +2306,28 @@ class CDCPrefetchTestCursor final : public IReplayPeekCursor, public ReferenceCo StringRef getMessageWithTags() override { return payload; } void nextMessage() override { done = true; - position = LogMessageVersion(101); + position = LogMessageVersion(messageVersion + 1); } Future getMore(TaskPriority taskID) override { ++fetches; co_await ready; fetched = true; if (!containsMutation) { - position = LogMessageVersion(101); + position = LogMessageVersion(messageVersion + 1); } co_return; } bool isExhausted() const override { return fetched && !hasMessage(); } LogMessageVersion const& version() const override { return position; } Version popped() const override { return 0; } - Version getMinKnownCommittedVersion() const override { return 100; } + Version getMinKnownCommittedVersion() const override { return messageVersion; } int64_t getMaxRetainedReplyCount() const override { return 1; } void setReplyByteLimit(int limitBytes) override { ASSERT_GT(limitBytes, payload.size()); } Optional getPrimaryPeekLocation() const override { return {}; } Optional getCurrentPeekLocation() const override { return {}; } - Version getMaxKnownVersion() const override { return 100; } + Version getMaxKnownVersion() const override { return messageVersion; } Reference cloneNoMore() override { - auto clone = makeReference(Void(), containsMutation); + auto clone = makeReference(Void(), containsMutation, messageVersion); clone->position = position; clone->fetched = fetched; clone->done = done; @@ -2525,7 +2527,7 @@ class CDCProxyPrefetchTest { second->streamIds = { 1, 2 }; test.proxy.tags[second->tag] = second; CDCReadAheadPass pass(second); - pass.claim(other); + pass.claim(other, 100); ASSERT_EQ(test.proxy.nextTagReadVersion(second).get(), 100); auto replacement = makeReference(second->tag); replacement->streamIds = second->streamIds; @@ -2573,6 +2575,40 @@ class CDCProxyPrefetchTest { co_return; } + static Future staggeredSharedTag() { + CDCProxyPrefetchTest test; + auto first = test.addStream(1); + auto second = test.addStream(2); + second->bufferedThrough = 199; + second->tagIntervals.back().bufferedThrough = 199; + second->tagIntervals.back().end = 300; + ASSERT(second->readAhead.issueReply(199, 199, second->minVersion, true)); + + auto firstCursor = makeReference(Void()); + ASSERT_EQ(test.proxy.nextTagPrefetchVersion(test.tag).get(), 100); + co_await test.proxy.bufferTagCursor(test.tag, 100, firstCursor, Never(), Prefetch::True); + ASSERT_EQ(firstCursor->fetchCount(), 1); + ASSERT_EQ(first->bufferedThrough, 100); + ASSERT_EQ(first->mutations.size(), 1); + ASSERT_EQ(second->bufferedThrough, 199); + ASSERT(second->mutations.empty()); + ASSERT(second->readAhead.armedFor(199)); + ASSERT_EQ(test.proxy.nextTagPrefetchVersion(test.tag).get(), 200); + + auto secondCursor = makeReference(Void(), true, 200); + co_await test.proxy.bufferTagCursor(test.tag, 200, secondCursor, Never(), Prefetch::True); + ASSERT_EQ(secondCursor->fetchCount(), 1); + ASSERT_EQ(second->bufferedThrough, 200); + ASSERT_EQ(second->mutations.size(), 1); + ASSERT(!second->readAhead.provesCursor(200, second->minVersion)); + ASSERT(!test.proxy.nextTagPrefetchVersion(test.tag).present()); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), test.proxy.bufferedBytes); + test.proxy.clearBufferedMutations(first); + test.proxy.clearBufferedMutations(second); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), 0); + co_return; + } + static Future capacity(bool queued = false) { CDCProxyPrefetchTest test; auto stream = test.addStream(1); @@ -2614,7 +2650,7 @@ class CDCProxyPrefetchTest { co_await test.proxy.bufferLock.take(TaskPriority::TLogPeekReply, passLimit.reservationBytes); FlowLock::Releaser reservation(test.proxy.bufferLock, passLimit.reservationBytes); CDCReadAheadPass pass(test.tag); - pass.claim(stream); + pass.claim(stream, 100); CDCBufferSelection selection; selection.selectedStreamIds.insert(1); selection.selectedBytes = passLimit.preferredBufferedBytes + 1; @@ -2715,7 +2751,7 @@ TEST_CASE("/NativeCDC/PrefetchCreditTailAndTags") { ASSERT(!nextCDCPrefetchVersion(stream, second).present()); { CDCReadAheadPass pass(first); - pass.claim(stream); + pass.claim(stream, 100); ASSERT(hasCDCReadInterest(stream, first)); ASSERT(!hasCDCReadInterest(stream, second)); ASSERT(!nextCDCPrefetchVersion(stream, second).present()); @@ -2736,6 +2772,9 @@ TEST_CASE("/NativeCDC/PrefetchCreditTailAndTags") { TEST_CASE("/NativeCDC/PrefetchMaterializesSharedTag") { return CDCProxyPrefetchTest::publish(); } +TEST_CASE("/NativeCDC/PrefetchPreservesLaterSharedTagCredit") { + return CDCProxyPrefetchTest::staggeredSharedTag(); +} TEST_CASE("/NativeCDC/PrefetchDeclinesUnavailableCapacity") { return CDCProxyPrefetchTest::capacity(); } From 86ed42f6b0bb702ed7bbcd5125cb78eafed24137 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Fri, 4 Sep 2026 16:31:40 -0700 Subject: [PATCH 025/170] Fix loopback streams in remote TLog recovery tests --- fdbserver/logsystem/LogSystemRecoveryTests.cpp | 15 +++++++++++---- 1 file changed, 11 insertions(+), 4 deletions(-) diff --git a/fdbserver/logsystem/LogSystemRecoveryTests.cpp b/fdbserver/logsystem/LogSystemRecoveryTests.cpp index 5ce4c6d16ec..dbc0935e173 100644 --- a/fdbserver/logsystem/LogSystemRecoveryTests.cpp +++ b/fdbserver/logsystem/LogSystemRecoveryTests.cpp @@ -52,6 +52,13 @@ Reference makeRemotePrefixLogSet(const std::vector& tlogs return logSet; } +TLogInterface makeRemotePrefixRouterClient(TLogInterface router) { + // Streaming health checks and reply backpressure require transport-backed peek streams. + router.peekMessages = RequestStream(router.peekMessages.getEndpoint()); + router.peekStreamMessages = RequestStream(router.peekStreamMessages.getEndpoint()); + return router; +} + Reference makeLaggingRemoteLogSystem(const std::vector& remoteLogs, const TLogInterface& oldRouter, const TLogInterface& currentRouter) { @@ -69,8 +76,8 @@ Reference makeLaggingRemoteLogSystem(const std::vector logSystem->logRouterTags = 1; logSystem->tLogs.push_back(makeRemotePrefixLogSet({ TLogInterface(locality) }, true, 0, 100)); auto remote = makeRemotePrefixLogSet(remoteLogs, false, 1, 60); - remote->logRouters.push_back( - makeReference>>(OptionalInterface(currentRouter))); + remote->logRouters.push_back(makeReference>>( + OptionalInterface(makeRemotePrefixRouterClient(currentRouter)))); logSystem->tLogs.push_back(remote); OldLogData old; @@ -81,8 +88,8 @@ Reference makeLaggingRemoteLogSystem(const std::vector old.logRouterTags = 1; old.tLogs.push_back(makeRemotePrefixLogSet({ TLogInterface(locality) }, true, 0, 50)); auto oldRemote = makeRemotePrefixLogSet({ TLogInterface(locality) }, false, 1, 60); - oldRemote->logRouters.push_back( - makeReference>>(OptionalInterface(oldRouter))); + oldRemote->logRouters.push_back(makeReference>>( + OptionalInterface(makeRemotePrefixRouterClient(oldRouter)))); old.tLogs.push_back(oldRemote); logSystem->oldLogData.push_back(old); // Make the ordinary generation-purge criteria eligible so the prefix barrier is the only retention gate. From f815acc01c42fde8be39ab7c011a30068f7ea3e9 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 6 Sep 2026 10:37:30 -0700 Subject: [PATCH 026/170] Synchronize CDC prefetch test with cursor fetch start --- fdbserver/cdcproxy/CDCProxy.cpp | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index ebd5327edb0..c69a23e7d8f 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -2286,6 +2286,7 @@ class CDCPrefetchTestCursor final : public IReplayPeekCursor, public ReferenceCo bool done = false; bool containsMutation; int fetches = 0; + Promise fetchStarted; public: explicit CDCPrefetchTestCursor(Future ready, bool containsMutation = true, Version version = 100) @@ -2295,6 +2296,7 @@ class CDCPrefetchTestCursor final : public IReplayPeekCursor, public ReferenceCo payload = writer.toValue(); } int fetchCount() const { return fetches; } + Future onFetchStarted() { return fetchStarted.getFuture(); } void setProtocolVersion(ProtocolVersion version) override { input = ArenaReader(payload.arena(), payload, AssumeVersion(version)); } @@ -2310,6 +2312,9 @@ class CDCPrefetchTestCursor final : public IReplayPeekCursor, public ReferenceCo } Future getMore(TaskPriority taskID) override { ++fetches; + if (fetchStarted.canBeSet()) { + fetchStarted.send(Void()); + } co_await ready; fetched = true; if (!containsMutation) { @@ -2372,8 +2377,10 @@ class CDCProxyPrefetchTest { } Promise ready; auto cursor = makeReference(ready.getFuture()); + auto fetchStarted = cursor->onFetchStarted(); auto work = test.proxy.bufferTagCursor(test.tag, 100, cursor, Never(), Prefetch::True); - co_await delay(0); + auto start = co_await race(fetchStarted, work); + ASSERT_EQ(start.index(), 0); ASSERT_EQ(cursor->fetchCount(), 1); ASSERT(stream->readAhead.claimedBy(test.tag.getPtr())); if (release) { From dab8bf3e3cdf99c72db550d1fcfdac4088cf55f3 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 6 Sep 2026 11:12:07 -0700 Subject: [PATCH 027/170] Encapsulate version vector storage and serialization mutator --- fdbclient/include/fdbclient/VersionVector.h | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/fdbclient/include/fdbclient/VersionVector.h b/fdbclient/include/fdbclient/VersionVector.h index 57160149a0e..5553749b6e1 100644 --- a/fdbclient/include/fdbclient/VersionVector.h +++ b/fdbclient/include/fdbclient/VersionVector.h @@ -34,6 +34,9 @@ static const int InvalidEncodedSize = 0; struct VersionVector { constexpr static FileIdentifier file_identifier = 5253554; friend struct serializable_traits; + friend struct dynamic_size_traits; + +private: boost::container::flat_map versions; // An ordered map. (Note: // changing this to an unordered // map will break the @@ -42,6 +45,7 @@ struct VersionVector { // there may or may not be a corresponding entry for this // version in the "versions" map.) +public: VersionVector() : maxVersion(invalidVersion), cachedEncodedSize(InvalidEncodedSize) {} explicit VersionVector(Version version) : maxVersion(version), cachedEncodedSize(InvalidEncodedSize) {} @@ -54,6 +58,7 @@ struct VersionVector { } inline void invalidateCachedEncodedSize() { cachedEncodedSize = InvalidEncodedSize; } + void setMaxVersion(Version version) { maxVersion = version; } // Encoded version vector size. Introduced to help speed up serialization. // @note This encoded size is not meant to be kept in sync with the updates @@ -65,8 +70,6 @@ struct VersionVector { public: Version getMaxVersion() const { return maxVersion; } - void setMaxVersion(Version version) { maxVersion = version; } - int size() const { return versions.size(); } bool empty() const { return versions.empty(); } From 913948e1af90a04d019ad1a859e2957c9c23b6a5 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 6 Sep 2026 11:12:19 -0700 Subject: [PATCH 028/170] Keep arbitrary page object ownership fields private --- fdbserver/kvstore/IPager.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fdbserver/kvstore/IPager.h b/fdbserver/kvstore/IPager.h index 8bd6b529f89..ad3f268217d 100644 --- a/fdbserver/kvstore/IPager.h +++ b/fdbserver/kvstore/IPager.h @@ -174,12 +174,12 @@ struct ArbitraryObject { onDestruct = nullptr; } +private: // ptr can be set to any arbitrary thing. If it is not null at destruct time then // onDestruct(ptr) will be called if onDestruct is not null. void* ptr = nullptr; void (*onDestruct)(void*) = nullptr; -private: // Call onDestruct(ptr) if needed but don't reset any state void destructOnly() { if (ptr != nullptr && onDestruct != nullptr) { From c262bd283b3dc7cb20e6cdaa8b30e50af95feadb Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 6 Sep 2026 11:13:40 -0700 Subject: [PATCH 029/170] Encapsulate storage wiggle queue and handles --- fdbserver/datadistributor/StorageWiggler.h | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/fdbserver/datadistributor/StorageWiggler.h b/fdbserver/datadistributor/StorageWiggler.h index df3a905a093..5a424d21e0d 100644 --- a/fdbserver/datadistributor/StorageWiggler.h +++ b/fdbserver/datadistributor/StorageWiggler.h @@ -34,7 +34,8 @@ class DDTeamCollection; -struct StorageWiggler : ReferenceCounted { +class StorageWiggler : public ReferenceCounted { +public: static constexpr double MIN_ON_CHECK_DELAY_SEC = 5.0; using State = StorageWigglerState::Value; static constexpr State INVALID = StorageWigglerState::INVALID; @@ -46,6 +47,8 @@ struct StorageWiggler : ReferenceCounted { StorageWiggleMetrics metrics; AsyncVar stopWiggleSignal; + +private: // data structures using MetadataUIDP = std::pair; // min-heap @@ -53,6 +56,7 @@ struct StorageWiggler : ReferenceCounted { wiggle_pq; std::unordered_map pq_handles; +public: State wiggleState = INVALID; double lastStateChangeTs = 0.0; // timestamp describes when did the state change From a591f9b8e4843ad9df3bd573555e1e60b4d6b0c0 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 6 Sep 2026 11:14:03 -0700 Subject: [PATCH 030/170] Keep key range snapshot boundaries private --- fdbclient/DataDistributionConfig.cpp | 2 +- fdbclient/include/fdbclient/KeyBackedRangeMap.h | 8 +++++++- 2 files changed, 8 insertions(+), 2 deletions(-) diff --git a/fdbclient/DataDistributionConfig.cpp b/fdbclient/DataDistributionConfig.cpp index b090a483f53..fc301596685 100644 --- a/fdbclient/DataDistributionConfig.cpp +++ b/fdbclient/DataDistributionConfig.cpp @@ -50,7 +50,7 @@ json_spirit::mValue DDConfiguration::toJSON(RangeConfigMapSnapshot const& config doc["numConfiguredRanges"] = configuredRanges; doc["numDefaultRanges"] = defaultRanges; - doc["numBoundaries"] = (int)config.map.size(); + doc["numBoundaries"] = (int)config.boundaryCount(); return doc; } diff --git a/fdbclient/include/fdbclient/KeyBackedRangeMap.h b/fdbclient/include/fdbclient/KeyBackedRangeMap.h index 0d4b6649fb8..b82ad2e71eb 100644 --- a/fdbclient/include/fdbclient/KeyBackedRangeMap.h +++ b/fdbclient/include/fdbclient/KeyBackedRangeMap.h @@ -29,7 +29,8 @@ // This is ReferenceCounted as it can be large and there is no reason to copy it as // it should not be modified locally. template -struct KeyRangeMapSnapshot : public ReferenceCounted> { +class KeyRangeMapSnapshot : public ReferenceCounted> { +public: using Map = std::map; // A default constructed map snapshot can't be used to look anything up because no ranges are covered. @@ -97,6 +98,11 @@ struct KeyRangeMapSnapshot : public ReferenceCounted + friend class KeyBackedRangeMap; Map map; }; From db1dc0006f38ceda81cab5ce65338e9cbaf8112d Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 6 Sep 2026 11:14:25 -0700 Subject: [PATCH 031/170] Protect log locality index storage from external mutation --- fdbserver/logsystem/LogSystem.cpp | 2 +- fdbserver/logsystem/LogSystemPeekCursor.cpp | 6 +++--- fdbserver/logsystem/include/fdbserver/logsystem/LogSet.h | 6 ++++++ 3 files changed, 10 insertions(+), 4 deletions(-) diff --git a/fdbserver/logsystem/LogSystem.cpp b/fdbserver/logsystem/LogSystem.cpp index 2d2ebdea049..52f8b3ef537 100644 --- a/fdbserver/logsystem/LogSystem.cpp +++ b/fdbserver/logsystem/LogSystem.cpp @@ -1026,7 +1026,7 @@ Future LogSystem::confirmEpochLive_internal(Reference logSet, Opti while (true) { for (int i = 0; i < alive.size(); i++) { if (!responded[i] && alive[i].isReady() && !alive[i].isError()) { - aliveEntries.push_back(logSet->logEntryArray[i]); + aliveEntries.push_back(logSet->getLogEntry(i)); responded[i] = true; } } diff --git a/fdbserver/logsystem/LogSystemPeekCursor.cpp b/fdbserver/logsystem/LogSystemPeekCursor.cpp index 2866bc353b8..e5db866d4cc 100644 --- a/fdbserver/logsystem/LogSystemPeekCursor.cpp +++ b/fdbserver/logsystem/LogSystemPeekCursor.cpp @@ -887,7 +887,7 @@ void MergedPeekCursor::updateMessage(bool usePolicy) { locations.clear(); for (auto sortedVersion : versions) { - locations.push_back(logSet->logEntryArray[sortedVersion.second]); + locations.push_back(logSet->getLogEntry(sortedVersion.second)); if (locations.size() >= tLogReplicationFactor && logSet->satisfiesPolicy(locations)) { selectedVersion = sortedVersion.first; break; @@ -1207,7 +1207,7 @@ void SetPeekCursor::updateMessage(int logIdx, bool usePolicy) { std::sort(versions.begin(), versions.end()); locations.clear(); for (auto sortedVersion : versions) { - locations.push_back(logSets[logIdx]->logEntryArray[sortedVersion.second]); + locations.push_back(logSets[logIdx]->getLogEntry(sortedVersion.second)); if (locations.size() >= logSets[logIdx]->tLogReplicationFactor && logSets[logIdx]->satisfiesPolicy(locations)) { selectedVersion = sortedVersion.first; @@ -1305,7 +1305,7 @@ Future setPeekGetMore(SetPeekCursor* self, LogMessageVersion startVersion, for (int i = 0; i < self->serverCursors[self->bestSet].size(); i++) { if (!self->serverCursors[self->bestSet][i]->isActive() && self->serverCursors[self->bestSet][i]->version() <= self->messageVersion) { - self->locations.push_back(self->logSets[self->bestSet]->logEntryArray[i]); + self->locations.push_back(self->logSets[self->bestSet]->getLogEntry(i)); } } bestSetValid = self->locations.size() < self->logSets[self->bestSet]->tLogReplicationFactor || diff --git a/fdbserver/logsystem/include/fdbserver/logsystem/LogSet.h b/fdbserver/logsystem/include/fdbserver/logsystem/LogSet.h index d0dc5b14ada..490005b519b 100644 --- a/fdbserver/logsystem/include/fdbserver/logsystem/LogSet.h +++ b/fdbserver/logsystem/include/fdbserver/logsystem/LogSet.h @@ -57,8 +57,13 @@ class LogSet : NonCopyable, public ReferenceCounted { TLogVersion tLogVersion; Reference tLogPolicy; Reference logServerSet; + +private: + // The locality map stores pointers to these indices; only updateLocalitySet may resize them. std::vector logIndexArray; std::vector logEntryArray; + +public: bool isLocal; int8_t locality; Version startVersion; @@ -83,6 +88,7 @@ class LogSet : NonCopyable, public ReferenceCounted { void checkSatelliteTagLocations(); int bestLocationFor(Tag tag); void updateLocalitySet(std::vector const& localities); + LocalityEntry getLogEntry(int location) const { return logEntryArray[location]; } bool satisfiesPolicy(const std::vector& locations); void getPushLocations( VectorRef tags, From cf6fdd8fec255cd7346b58a1dc89813370dfc5f9 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 6 Sep 2026 11:14:35 -0700 Subject: [PATCH 032/170] Keep data distribution team indexes synchronized behind class boundary --- fdbserver/datadistributor/DDTeamCollection.h | 8 ++++---- fdbserver/datadistributor/DataDistribution.cpp | 2 +- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/fdbserver/datadistributor/DDTeamCollection.h b/fdbserver/datadistributor/DDTeamCollection.h index 2627041e4de..03faea63d43 100644 --- a/fdbserver/datadistributor/DDTeamCollection.h +++ b/fdbserver/datadistributor/DDTeamCollection.h @@ -691,13 +691,12 @@ class DDTeamCollection : public ReferenceCounted { std::map, Reference> machine_info; std::vector> machineTeams; // all machine teams - // IMPORTANT: teams and teamsByServerIDs MUST be consistent, so any time we - // mutate teams, we must also mutate teamsByServerIDs +private: + // These must be updated together when adding or removing a team. std::vector> teams; - // O(1) hash map from server ID string to team information - // Currently used by getTeamByServers std::unordered_map> teamsByServerIDs; +public: std::vector teamCollections; AsyncTrigger printDetailedTeamsInfo; Reference storageServerSet; @@ -705,6 +704,7 @@ class DDTeamCollection : public ReferenceCounted { explicit DDTeamCollection(DDTeamCollectionInitParams const& params); ~DDTeamCollection(); + size_t teamCount() const { return teams.size(); } void addLaggingStorageServer(Key zoneId); diff --git a/fdbserver/datadistributor/DataDistribution.cpp b/fdbserver/datadistributor/DataDistribution.cpp index f4eadf3ce2f..9c4c6e4ee56 100644 --- a/fdbserver/datadistributor/DataDistribution.cpp +++ b/fdbserver/datadistributor/DataDistribution.cpp @@ -3789,7 +3789,7 @@ Future ddExclusionSafetyCheck(DistributorExclusionSafetyCheckRequest req, co_return; } // If there is only 1 team, unsafe to mark failed: team building can get stuck due to lack of servers left - if (self->teamCollection->teams.size() <= 1) { + if (self->teamCollection->teamCount() <= 1) { TraceEvent("DDExclusionSafetyCheckNotEnoughTeams", self->ddId).log(); reply.safe = false; req.reply.send(reply); From 6056ea7bf93a10ba17eafe122b51925cce56021c Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 6 Sep 2026 11:17:16 -0700 Subject: [PATCH 033/170] Keep queue model actor lifecycle behind submission methods --- fdbclient/NativeAPI.cpp | 2 +- .../fdbclient/StorageServerLoadBalance.h | 14 +++++------ fdbrpc/include/fdbrpc/LoadBalance.h | 23 +++++++------------ fdbrpc/include/fdbrpc/QueueModel.h | 20 +++++++++++++--- 4 files changed, 33 insertions(+), 26 deletions(-) diff --git a/fdbclient/NativeAPI.cpp b/fdbclient/NativeAPI.cpp index 0383891cc0e..ae27b9174f5 100644 --- a/fdbclient/NativeAPI.cpp +++ b/fdbclient/NativeAPI.cpp @@ -3171,7 +3171,7 @@ Optional> maybeDuplicateTSSStr ReplyPromiseStream tssReplyStream = tssRequestStream.getReplyStream(req); PromiseStream ssDuplicateReplyStream; TSSDuplicateStreamData streamData(ssDuplicateReplyStream); - model->addActor.send(tssStreamComparison(req, streamData, tssReplyStream, tssData.get())); + model->addBackgroundActor(tssStreamComparison(req, streamData, tssReplyStream, tssData.get())); return Optional>(streamData); } } diff --git a/fdbclient/include/fdbclient/StorageServerLoadBalance.h b/fdbclient/include/fdbclient/StorageServerLoadBalance.h index 8849d5c9ee3..1352c43e500 100644 --- a/fdbclient/include/fdbclient/StorageServerLoadBalance.h +++ b/fdbclient/include/fdbclient/StorageServerLoadBalance.h @@ -368,13 +368,13 @@ struct LoadBalanceRequestHooks tssRequestStream(tssData.get().endpoint); Future> fTssResult = tssRequestStream.tryGetReply(request); - model->addActor.send(tssComparison(request, - ssResponse, - fTssResult, - tssData.get(), - stream->getEndpoint().token.first(), - alternatives, - channel)); + model->addBackgroundActor(tssComparison(request, + ssResponse, + fTssResult, + tssData.get(), + stream->getEndpoint().token.first(), + alternatives, + channel)); } } } diff --git a/fdbrpc/include/fdbrpc/LoadBalance.h b/fdbrpc/include/fdbrpc/LoadBalance.h index ec6c96c65c2..4d51aa31afc 100644 --- a/fdbrpc/include/fdbrpc/LoadBalance.h +++ b/fdbrpc/include/fdbrpc/LoadBalance.h @@ -268,22 +268,15 @@ struct RequestData : NonCopyable { ASSERT(modelHolder->model); QueueModel* model = modelHolder->model; - if (model->laggingRequestCount > FLOW_KNOBS->MAX_LAGGING_REQUESTS_OUTSTANDING || - model->laggingRequests.isReady()) { - model->laggingRequests.cancel(); - model->laggingRequestCount = 0; - model->addActor = PromiseStream>(); - model->laggingRequests = actorCollection(model->addActor.getFuture(), &model->laggingRequestCount); - } - - // We need to process the lagging request in order to update the queue model - Reference holderCapture = std::move(modelHolder); - auto triedAllOptionsCapture = triedAllOptions; - Future updateModel = map(response, [holderCapture, triedAllOptionsCapture](Reply result) { - checkAndProcessResultImpl(result, holderCapture, AtMostOnce::False, triedAllOptionsCapture); - return Void(); + model->addLaggingRequest([&] { + // We need to process the lagging request in order to update the queue model + Reference holderCapture = std::move(modelHolder); + auto triedAllOptionsCapture = triedAllOptions; + return map(response, [holderCapture, triedAllOptionsCapture](Reply result) { + checkAndProcessResultImpl(result, holderCapture, AtMostOnce::False, triedAllOptionsCapture); + return Void(); + }); }); - model->addActor.send(updateModel); } ~RequestData() { diff --git a/fdbrpc/include/fdbrpc/QueueModel.h b/fdbrpc/include/fdbrpc/QueueModel.h index 90eece2d78d..f6e95dffa7e 100644 --- a/fdbrpc/include/fdbrpc/QueueModel.h +++ b/fdbrpc/include/fdbrpc/QueueModel.h @@ -96,9 +96,6 @@ class QueueModel { double secondMultiplier; double secondBudget; - PromiseStream> addActor; - Future laggingRequests; // requests for which a different recipient already answered - int laggingRequestCount; QueueModel() : secondMultiplier(1.0), secondBudget(0), laggingRequestCount(0) { laggingRequests = actorCollection(addActor.getFuture(), &laggingRequestCount); @@ -106,7 +103,24 @@ class QueueModel { ~QueueModel() { laggingRequests.cancel(); } + void addBackgroundActor(Future actor) { addActor.send(actor); } + + // The lagging actor must be created after an exhausted collection is cancelled. + template + void addLaggingRequest(MakeActor&& makeActor) { + if (laggingRequestCount > FLOW_KNOBS->MAX_LAGGING_REQUESTS_OUTSTANDING || laggingRequests.isReady()) { + laggingRequests.cancel(); + laggingRequestCount = 0; + addActor = PromiseStream>(); + laggingRequests = actorCollection(addActor.getFuture(), &laggingRequestCount); + } + addActor.send(makeActor()); + } + private: + PromiseStream> addActor; + Future laggingRequests; // requests for which a different recipient already answered + int laggingRequestCount; std::unordered_map data; }; From 8df70aa06b0ad2eae2b263f7b637d6e17c229716 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 6 Sep 2026 11:17:55 -0700 Subject: [PATCH 034/170] Consume incompatible peer reports through transport API --- fdbrpc/FlowTransport.cpp | 13 +++++++++++-- fdbrpc/include/fdbrpc/FlowTransport.h | 8 +++++--- fdbserver/worker/worker.cpp | 10 +--------- 3 files changed, 17 insertions(+), 14 deletions(-) diff --git a/fdbrpc/FlowTransport.cpp b/fdbrpc/FlowTransport.cpp index 00a4f8ef108..6761e6f2a19 100644 --- a/fdbrpc/FlowTransport.cpp +++ b/fdbrpc/FlowTransport.cpp @@ -1945,7 +1945,7 @@ const std::unordered_map>& FlowTransport::getAll return self->peers; } -std::map>* FlowTransport::getIncompatiblePeers() { +std::vector FlowTransport::consumeReportableIncompatiblePeers() { for (auto it = self->incompatiblePeers.begin(); it != self->incompatiblePeers.end();) { if (self->multiVersionConnections.contains(it->second.first)) { it = self->incompatiblePeers.erase(it); @@ -1953,7 +1953,16 @@ std::map>* FlowTransport::getIncompa it++; } } - return &self->incompatiblePeers; + std::vector reportable; + for (auto it = self->incompatiblePeers.begin(); it != self->incompatiblePeers.end();) { + if (now() - it->second.second > FLOW_KNOBS->INCOMPATIBLE_PEER_DELAY_BEFORE_LOGGING) { + reportable.push_back(it->first); + it = self->incompatiblePeers.erase(it); + } else { + it++; + } + } + return reportable; } Future FlowTransport::onIncompatibleChanged() { diff --git a/fdbrpc/include/fdbrpc/FlowTransport.h b/fdbrpc/include/fdbrpc/FlowTransport.h index ad68b112b7d..f67e706f267 100644 --- a/fdbrpc/include/fdbrpc/FlowTransport.h +++ b/fdbrpc/include/fdbrpc/FlowTransport.h @@ -26,6 +26,7 @@ #include #include #include +#include #include "fdbrpc/DDSketch.h" #include "flow/genericactors.h" @@ -237,10 +238,11 @@ class FlowTransport : NonCopyable { // Returns all peers that the FlowTransport is monitoring. const std::unordered_map>& getAllPeers() const; - // Returns the set of all peers that have attempted to connect, but have incompatible protocol versions - std::map>* getIncompatiblePeers(); + // Returns and removes peers whose incompatible protocol versions have persisted long enough to report. + // Peers later recognized as multi-version connections are discarded without reporting. + std::vector consumeReportableIncompatiblePeers(); - // Returns when getIncompatiblePeers has at least one peer which is incompatible. + // Returns when an incompatible peer has persisted long enough to report. Future onIncompatibleChanged(); // Signal that a peer connection is being used, even if no messages are currently being sent to the peer diff --git a/fdbserver/worker/worker.cpp b/fdbserver/worker/worker.cpp index d3d65f908c9..6615decedcc 100644 --- a/fdbserver/worker/worker.cpp +++ b/fdbserver/worker/worker.cpp @@ -638,15 +638,7 @@ Future registrationClient(Referencebegin(); it != peers->end();) { - if (now() - it->second.second > FLOW_KNOBS->INCOMPATIBLE_PEER_DELAY_BEFORE_LOGGING) { - request.incompatiblePeers.push_back(it->first); - it = peers->erase(it); - } else { - it++; - } - } + request.incompatiblePeers = FlowTransport::transport().consumeReportableIncompatiblePeers(); bool ccInterfacePresent = ccInterface->get().present(); if (ccInterfacePresent) { From e7e39dab004f62586b7c6e327b43af6e37373b75 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 6 Sep 2026 11:54:33 -0700 Subject: [PATCH 035/170] Revert friend-based encapsulation changes --- fdbclient/DataDistributionConfig.cpp | 2 +- fdbclient/include/fdbclient/KeyBackedRangeMap.h | 8 +------- fdbclient/include/fdbclient/VersionVector.h | 7 ++----- 3 files changed, 4 insertions(+), 13 deletions(-) diff --git a/fdbclient/DataDistributionConfig.cpp b/fdbclient/DataDistributionConfig.cpp index fc301596685..b090a483f53 100644 --- a/fdbclient/DataDistributionConfig.cpp +++ b/fdbclient/DataDistributionConfig.cpp @@ -50,7 +50,7 @@ json_spirit::mValue DDConfiguration::toJSON(RangeConfigMapSnapshot const& config doc["numConfiguredRanges"] = configuredRanges; doc["numDefaultRanges"] = defaultRanges; - doc["numBoundaries"] = (int)config.boundaryCount(); + doc["numBoundaries"] = (int)config.map.size(); return doc; } diff --git a/fdbclient/include/fdbclient/KeyBackedRangeMap.h b/fdbclient/include/fdbclient/KeyBackedRangeMap.h index b82ad2e71eb..0d4b6649fb8 100644 --- a/fdbclient/include/fdbclient/KeyBackedRangeMap.h +++ b/fdbclient/include/fdbclient/KeyBackedRangeMap.h @@ -29,8 +29,7 @@ // This is ReferenceCounted as it can be large and there is no reason to copy it as // it should not be modified locally. template -class KeyRangeMapSnapshot : public ReferenceCounted> { -public: +struct KeyRangeMapSnapshot : public ReferenceCounted> { using Map = std::map; // A default constructed map snapshot can't be used to look anything up because no ranges are covered. @@ -98,11 +97,6 @@ class KeyRangeMapSnapshot : public ReferenceCounted - friend class KeyBackedRangeMap; Map map; }; diff --git a/fdbclient/include/fdbclient/VersionVector.h b/fdbclient/include/fdbclient/VersionVector.h index 5553749b6e1..57160149a0e 100644 --- a/fdbclient/include/fdbclient/VersionVector.h +++ b/fdbclient/include/fdbclient/VersionVector.h @@ -34,9 +34,6 @@ static const int InvalidEncodedSize = 0; struct VersionVector { constexpr static FileIdentifier file_identifier = 5253554; friend struct serializable_traits; - friend struct dynamic_size_traits; - -private: boost::container::flat_map versions; // An ordered map. (Note: // changing this to an unordered // map will break the @@ -45,7 +42,6 @@ struct VersionVector { // there may or may not be a corresponding entry for this // version in the "versions" map.) -public: VersionVector() : maxVersion(invalidVersion), cachedEncodedSize(InvalidEncodedSize) {} explicit VersionVector(Version version) : maxVersion(version), cachedEncodedSize(InvalidEncodedSize) {} @@ -58,7 +54,6 @@ struct VersionVector { } inline void invalidateCachedEncodedSize() { cachedEncodedSize = InvalidEncodedSize; } - void setMaxVersion(Version version) { maxVersion = version; } // Encoded version vector size. Introduced to help speed up serialization. // @note This encoded size is not meant to be kept in sync with the updates @@ -70,6 +65,8 @@ struct VersionVector { public: Version getMaxVersion() const { return maxVersion; } + void setMaxVersion(Version version) { maxVersion = version; } + int size() const { return versions.size(); } bool empty() const { return versions.empty(); } From 033a329485d7b2ad96a28b8230c7e4109231b5d9 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 6 Sep 2026 12:00:55 -0700 Subject: [PATCH 036/170] Separate RocksDB checkpoint implementation from server core --- fdbserver/CMakeLists.txt | 2 + fdbserver/checkpoint/BulkSstFiles.cpp | 181 ++++++++++++++++++ fdbserver/checkpoint/CMakeLists.txt | 26 +++ fdbserver/checkpoint/Checkpoint.cpp | 99 ++++++++++ .../RocksDBCheckpointUtils.cpp | 4 +- .../fdbserver/checkpoint/BulkSstFiles.h | 40 ++++ .../include/fdbserver/checkpoint/Checkpoint.h | 45 +++++ .../checkpoint}/RocksDBCheckpointUtils.h | 4 +- fdbserver/core/BulkDumpUtil.cpp | 92 +-------- fdbserver/core/BulkLoadUtil.cpp | 67 ------- fdbserver/core/CMakeLists.txt | 12 -- fdbserver/core/ServerCheckpoint.cpp | 80 +------- .../include/fdbserver/core/BulkDumpUtil.h | 15 -- .../include/fdbserver/core/BulkLoadUtil.h | 2 - .../include/fdbserver/core/ServerCheckpoint.h | 22 --- fdbserver/kvstore/CMakeLists.txt | 2 +- fdbserver/kvstore/KeyValueStoreRocksDB.cpp | 2 +- .../kvstore/KeyValueStoreShardedRocksDB.cpp | 2 +- fdbserver/storageserver/CMakeLists.txt | 2 +- fdbserver/storageserver/storageserver.cpp | 5 +- fdbserver/worker/CMakeLists.txt | 4 + fdbserver/workloads/BulkLoading.cpp | 2 +- fdbserver/workloads/CMakeLists.txt | 5 + fdbserver/workloads/PhysicalShardMove.cpp | 2 +- .../StorageServerCheckpointRestoreTest.cpp | 2 +- 25 files changed, 418 insertions(+), 301 deletions(-) create mode 100644 fdbserver/checkpoint/BulkSstFiles.cpp create mode 100644 fdbserver/checkpoint/CMakeLists.txt create mode 100644 fdbserver/checkpoint/Checkpoint.cpp rename fdbserver/{core => checkpoint}/RocksDBCheckpointUtils.cpp (99%) create mode 100644 fdbserver/checkpoint/include/fdbserver/checkpoint/BulkSstFiles.h create mode 100644 fdbserver/checkpoint/include/fdbserver/checkpoint/Checkpoint.h rename fdbserver/{core/include/fdbserver/core => checkpoint/include/fdbserver/checkpoint}/RocksDBCheckpointUtils.h (99%) diff --git a/fdbserver/CMakeLists.txt b/fdbserver/CMakeLists.txt index 704975e28d0..306b3d935da 100644 --- a/fdbserver/CMakeLists.txt +++ b/fdbserver/CMakeLists.txt @@ -26,6 +26,7 @@ if(WITH_ROCKSDB) endif() add_subdirectory(core) +add_subdirectory(checkpoint) add_subdirectory(kvstore) add_subdirectory(logsystem) add_subdirectory(mocks3) @@ -193,6 +194,7 @@ target_link_libraries(fdbserver PRIVATE fdbserver_storageserver fdbserver_tester fdbserver_tlog + fdbserver_checkpoint fdbserver_core) if (WITH_ROCKSDB) add_dependencies(fdbserver rocksdb) diff --git a/fdbserver/checkpoint/BulkSstFiles.cpp b/fdbserver/checkpoint/BulkSstFiles.cpp new file mode 100644 index 00000000000..c1c0f16ab1f --- /dev/null +++ b/fdbserver/checkpoint/BulkSstFiles.cpp @@ -0,0 +1,181 @@ +/* + * BulkSstFiles.cpp + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "fdbserver/checkpoint/BulkSstFiles.h" +#include "fdbserver/checkpoint/RocksDBCheckpointUtils.h" +#include "fdbserver/core/BulkLoadUtil.h" +#include "fdbserver/core/Knobs.h" +#include "fdbserver/core/StorageMetrics.h" +#include "flow/genericactors.h" + +// Generate SST file given the input sortedKVS to the input filePath. +// TODO(BulkDump): This copy of sortedKVS can be a slow task if data is large. +void writeKVSToSSTFile(std::string filePath, std::map& sortedKVS, UID logId) { + const std::string absFilePath = abspath(filePath); + // Check file + if (fileExists(absFilePath)) { + TraceEvent(SevWarn, "SSBulkDumpRetriableError", logId) + .detail("Reason", "exist old File when writeKVSToSSTFile") + .detail("DataFilePathLocal", absFilePath); + ASSERT_WE_THINK(false); + throw retry(); + } + // Dump data to file + std::unique_ptr sstWriter = newRocksDBSstFileWriter(); + sstWriter->open(absFilePath); + for (const auto& [key, value] : sortedKVS) { + sstWriter->write(key, value); // assuming sorted + } + if (!sstWriter->finish()) { + // Unexpected: having data but failed to finish + TraceEvent(SevWarn, "SSBulkDumpRetriableError", logId) + .detail("Reason", "failed to finish data sst writer when writeKVSToSSTFile") + .detail("DataFilePath", absFilePath); + ASSERT_WE_THINK(false); + throw retry(); + } + return; +} + +Future dumpDataFileToLocalDirectory(UID logId, + std::shared_ptr rangeDumpRawData, + BulkLoadFileSet localFileSet, + BulkLoadFileSet remoteFileSet, + BulkLoadByteSampleSetting byteSampleSetting, + Version dumpVersion, + KeyRange dumpRange, + BulkLoadType dumpType, + BulkLoadTransportMethod transportMethod) { + // Step 1: Clean up local folder + resetFileFolder((abspath(localFileSet.getFolder()))); + + // Step 2: Dump data to file + bool containDataFile = false; + if (!rangeDumpRawData->kvs.empty()) { + writeKVSToSSTFile(abspath(localFileSet.getDataFileFullPath()), rangeDumpRawData->kvs, logId); + containDataFile = true; + } else { + ASSERT(rangeDumpRawData->sampled.empty()); + containDataFile = false; + } + + // Step 3: Dump sample to file + bool containByteSampleFile = false; + if (!rangeDumpRawData->sampled.empty()) { + writeKVSToSSTFile(abspath(localFileSet.getBytesSampleFileFullPath()), rangeDumpRawData->sampled, logId); + containByteSampleFile = true; + } else { + containByteSampleFile = false; + } + + // Step 4: Generate manifest file + if (fileExists(abspath(localFileSet.getManifestFileFullPath()))) { + TraceEvent(SevWarn, "SSBulkDumpRetriableError", logId) + .detail("Reason", "exist old manifestFile") + .detail("ManifestFilePathLocal", abspath(localFileSet.getManifestFileFullPath())); + ASSERT_WE_THINK(false); + throw retry(); + } + BulkLoadFileSet fileSetRemote(remoteFileSet.getRootPath(), + remoteFileSet.getRelativePath(), + remoteFileSet.getManifestFileName(), + containDataFile ? remoteFileSet.getDataFileName() : std::string(), + containByteSampleFile ? remoteFileSet.getByteSampleFileName() : std::string(), + BulkLoadChecksum()); + BulkLoadManifest manifestMetadata(fileSetRemote, + dumpRange.begin, + dumpRange.end, + dumpVersion, + rangeDumpRawData->kvsBytes, + rangeDumpRawData->kvs.size(), + byteSampleSetting, + dumpType, + transportMethod); + std::string manifestStr = manifestMetadata.toString(); + std::shared_ptr manifest = std::make_shared(std::move(manifestStr)); + co_await writeBulkFileBytes(abspath(localFileSet.getManifestFileFullPath()), manifest); + co_return manifestMetadata; +} + + +// Return true if generated the byte sampling file. Otherwise, return false. +// TODO(BulkDump): directly read from special key space. +Future doBytesSamplingOnDataFile(std::string dataFileFullPath, // input file + std::string byteSampleFileFullPath, // output file + UID logId) { + int counter = 0; + bool res = false; + int retryCount = 0; + double startTime = now(); + while (true) { + Error err; + try { + std::unique_ptr sstWriter = newRocksDBSstFileWriter(); + sstWriter->open(abspath(byteSampleFileFullPath)); + bool anySampled = false; + std::unique_ptr reader = newRocksDBSstFileReader(); + reader->open(abspath(dataFileFullPath)); + while (reader->hasNext()) { + KeyValue kv = reader->next(); + ByteSampleInfo sampleInfo = isKeyValueInSample(kv); + if (sampleInfo.inSample) { + sstWriter->write(kv.key, BinaryWriter::toValue(sampleInfo.sampledSize, Unversioned())); + anySampled = true; + counter++; + if (counter > SERVER_KNOBS->BULKLOAD_BYTE_SAMPLE_BATCH_KEY_COUNT) { + co_await yield(); + counter = 0; + } + } + } + // It is possible that no key is sampled + // This can happen when the data to sample is small + // In this case, no SST sample byte file is generated + if (anySampled) { + ASSERT(sstWriter->finish()); + res = true; + } else { + ASSERT(!sstWriter->finish()); + deleteFile(abspath(byteSampleFileFullPath)); + } + break; + } catch (Error& e) { + err = e; + } + if (err.code() == error_code_actor_cancelled) { + throw err; + } + TraceEvent(SevWarn, "SSBulkLoadTaskSamplingError", logId) + .errorUnsuppressed(err) + .detail("DataFileFullPath", dataFileFullPath) + .detail("ByteSampleFileFullPath", byteSampleFileFullPath) + .detail("Duration", now() - startTime) + .detail("RetryCount", retryCount); + co_await delay(5.0); + deleteFile(abspath(byteSampleFileFullPath)); + retryCount++; + } + TraceEvent(bulkLoadVerboseEventSev(), "SSBulkLoadTaskSamplingComplete", logId) + .detail("DataFileFullPath", dataFileFullPath) + .detail("ByteSampleFileFullPath", byteSampleFileFullPath) + .detail("Duration", now() - startTime) + .detail("RetryCount", retryCount); + co_return res; +} diff --git a/fdbserver/checkpoint/CMakeLists.txt b/fdbserver/checkpoint/CMakeLists.txt new file mode 100644 index 00000000000..5ee1e12189f --- /dev/null +++ b/fdbserver/checkpoint/CMakeLists.txt @@ -0,0 +1,26 @@ +fdb_find_sources(FDBSERVER_CHECKPOINT_SRCS) + +add_flow_target(STATIC_LIBRARY NAME fdbserver_checkpoint SRCS ${FDBSERVER_CHECKPOINT_SRCS}) +add_fdbserver_link_test(fdbserver_checkpointlinktest + fdbserver_checkpoint + fdbserver_core) + +configure_fdbserver_common_includes(fdbserver_checkpoint) +target_include_directories(fdbserver_checkpoint + PUBLIC + ${CMAKE_CURRENT_SOURCE_DIR}/include + PRIVATE + ${CMAKE_SOURCE_DIR}/fdbserver/include) +target_link_libraries(fdbserver_checkpoint PUBLIC fdbserver_core) + +if(WITH_ROCKSDB) + add_dependencies(fdbserver_checkpoint rocksdb) + if(WITH_LIBURING) + target_include_directories(fdbserver_checkpoint PRIVATE ${ROCKSDB_INCLUDE_DIR} ${uring_INCLUDE_DIR}) + target_link_libraries(fdbserver_checkpoint PRIVATE ${ROCKSDB_LIBRARIES} ${uring_LIBRARIES} ${LZ4_LIBRARY}) + else() + target_include_directories(fdbserver_checkpoint PRIVATE ${ROCKSDB_INCLUDE_DIR}) + target_link_libraries(fdbserver_checkpoint PRIVATE ${ROCKSDB_LIBRARIES} ${LZ4_LIBRARY}) + endif() + target_compile_definitions(fdbserver_checkpoint PRIVATE WITH_ROCKSDB) +endif() diff --git a/fdbserver/checkpoint/Checkpoint.cpp b/fdbserver/checkpoint/Checkpoint.cpp new file mode 100644 index 00000000000..d84ac534f96 --- /dev/null +++ b/fdbserver/checkpoint/Checkpoint.cpp @@ -0,0 +1,99 @@ +/* + * Checkpoint.cpp + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "fdbserver/checkpoint/Checkpoint.h" +#include "fdbserver/checkpoint/RocksDBCheckpointUtils.h" + +ICheckpointReader* newCheckpointReader(const CheckpointMetaData& checkpoint, + const CheckpointAsKeyValues checkpointAsKeyValues, + UID logID) { + const CheckpointFormat format = checkpoint.getFormat(); + if (format == DataMoveRocksCF || format == RocksDB) { + return newRocksDBCheckpointReader(checkpoint, checkpointAsKeyValues, logID); + } else { + throw not_implemented(); + } + + return nullptr; +} + +Future deleteCheckpoint(CheckpointMetaData checkpoint) { + co_await delay(0, TaskPriority::FetchKeys); + const CheckpointFormat format = checkpoint.getFormat(); + if (format == DataMoveRocksCF || format == RocksDB || format == RocksDBKeyValues) { + if (!checkpoint.dir.empty()) { + platform::eraseDirectoryRecursive(checkpoint.dir); + } else { + TraceEvent(SevWarn, "CheckpointDirNotFound").detail("Checkpoint", checkpoint.toString()); + } + } else { + throw not_implemented(); + } +} + +Future fetchCheckpoint(Database cx, + CheckpointMetaData initialState, + std::string dir, + std::function(const CheckpointMetaData&)> cFun) { + TraceEvent("FetchCheckpointBegin", initialState.checkpointID).detail("CheckpointMetaData", initialState.toString()); + + CheckpointMetaData result; + const CheckpointFormat format = initialState.getFormat(); + ASSERT(format != RocksDBKeyValues); + if (format == DataMoveRocksCF || format == RocksDB) { + result = co_await fetchRocksDBCheckpoint(cx, initialState, dir, cFun); + } else { + throw not_implemented(); + } + + TraceEvent("FetchCheckpointEnd", initialState.checkpointID).detail("CheckpointMetaData", result.toString()); + co_return result; +} + +Future fetchCheckpointRanges(Database cx, + CheckpointMetaData initialState, + std::string dir, + std::vector ranges, + std::function(const CheckpointMetaData&)> cFun) { + TraceEvent(SevDebug, "FetchCheckpointRangesBegin", initialState.checkpointID) + .detail("CheckpointMetaData", initialState.toString()) + .detail("Ranges", describe(ranges)); + ASSERT(!ranges.empty()); + + CheckpointMetaData result; + const CheckpointFormat format = initialState.getFormat(); + if (format != RocksDBKeyValues) { + if (format != DataMoveRocksCF) { + throw not_implemented(); + } + initialState.setFormat(RocksDBKeyValues); + initialState.ranges = ranges; + initialState.dir = dir; + initialState.setSerializedCheckpoint( + ObjectWriter::toValue(RocksDBCheckpointKeyValues(ranges), IncludeVersion())); + } + + result = co_await fetchRocksDBCheckpoint(cx, initialState, dir, cFun); + + TraceEvent(SevDebug, "FetchCheckpointRangesEnd", initialState.checkpointID) + .detail("CheckpointMetaData", result.toString()) + .detail("Ranges", describe(ranges)); + co_return result; +} diff --git a/fdbserver/core/RocksDBCheckpointUtils.cpp b/fdbserver/checkpoint/RocksDBCheckpointUtils.cpp similarity index 99% rename from fdbserver/core/RocksDBCheckpointUtils.cpp rename to fdbserver/checkpoint/RocksDBCheckpointUtils.cpp index 118936d5fca..3fcb4b1ad08 100644 --- a/fdbserver/core/RocksDBCheckpointUtils.cpp +++ b/fdbserver/checkpoint/RocksDBCheckpointUtils.cpp @@ -1,5 +1,5 @@ /* - *RocksDBCheckpointUtils.cpp + * RocksDBCheckpointUtils.cpp * * This source file is part of the FoundationDB open source project * @@ -18,7 +18,7 @@ * limitations under the License. */ -#include "fdbserver/core/RocksDBCheckpointUtils.h" +#include "fdbserver/checkpoint/RocksDBCheckpointUtils.h" #ifdef WITH_ROCKSDB #include diff --git a/fdbserver/checkpoint/include/fdbserver/checkpoint/BulkSstFiles.h b/fdbserver/checkpoint/include/fdbserver/checkpoint/BulkSstFiles.h new file mode 100644 index 00000000000..76806bf0b36 --- /dev/null +++ b/fdbserver/checkpoint/include/fdbserver/checkpoint/BulkSstFiles.h @@ -0,0 +1,40 @@ +/* + * BulkSstFiles.h + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#pragma once + +#include "fdbserver/core/BulkDumpUtil.h" + +// Generate key-value data, byte sampling data, and manifest file. +// Return BulkLoadManifest metadata (equivalent to content of the manifest file). +// TODO(BulkDump): can cause slow tasks, do the task in a separate thread in the future. +// The size of sortedData is defined at the place of generating the data (getRangeDataToDump). +// The size is configured by MOVE_SHARD_KRM_ROW_LIMIT. +Future dumpDataFileToLocalDirectory(UID logId, + std::shared_ptr rangeDumpRawData, + BulkLoadFileSet localFileSet, + BulkLoadFileSet remoteFileSet, + BulkLoadByteSampleSetting byteSampleSetting, + Version dumpVersion, + KeyRange dumpRange, + BulkLoadType dumpType, + BulkLoadTransportMethod transportMethod); + +Future doBytesSamplingOnDataFile(std::string dataFileFullPath, std::string byteSampleFileFullPath, UID logId); diff --git a/fdbserver/checkpoint/include/fdbserver/checkpoint/Checkpoint.h b/fdbserver/checkpoint/include/fdbserver/checkpoint/Checkpoint.h new file mode 100644 index 00000000000..984fec7d359 --- /dev/null +++ b/fdbserver/checkpoint/include/fdbserver/checkpoint/Checkpoint.h @@ -0,0 +1,45 @@ +/* + * Checkpoint.h + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#pragma once + +#include "fdbserver/core/ServerCheckpoint.h" + +ICheckpointReader* newCheckpointReader(const CheckpointMetaData& checkpoint, + const CheckpointAsKeyValues checkpointAsKeyValues, + UID logID); + +// Delete a checkpoint. +Future deleteCheckpoint(CheckpointMetaData checkpoint); + +// Fetches checkpoint to a local `dir`, `initialState` provides the checkpoint formats, location, restart point, etc. +// If cFun is provided, the progress can be checkpointed. +// Returns a CheckpointMetaData, which could contain KVS-specific results, e.g., the list of fetched checkpoint files. +Future fetchCheckpoint(Database cx, + CheckpointMetaData initialState, + std::string dir, + std::function(const CheckpointMetaData&)> cFun = nullptr); + +// Same as above, except that the checkpoint is fetched as key-value pairs. +Future fetchCheckpointRanges(Database cx, + CheckpointMetaData initialState, + std::string dir, + std::vector ranges, + std::function(const CheckpointMetaData&)> cFun = nullptr); diff --git a/fdbserver/core/include/fdbserver/core/RocksDBCheckpointUtils.h b/fdbserver/checkpoint/include/fdbserver/checkpoint/RocksDBCheckpointUtils.h similarity index 99% rename from fdbserver/core/include/fdbserver/core/RocksDBCheckpointUtils.h rename to fdbserver/checkpoint/include/fdbserver/checkpoint/RocksDBCheckpointUtils.h index 9813caea9bf..6d966dff4bf 100644 --- a/fdbserver/core/include/fdbserver/core/RocksDBCheckpointUtils.h +++ b/fdbserver/checkpoint/include/fdbserver/checkpoint/RocksDBCheckpointUtils.h @@ -1,5 +1,5 @@ /* - *RocksDBCheckpointUtils.h + * RocksDBCheckpointUtils.h * * This source file is part of the FoundationDB open source project * @@ -21,7 +21,7 @@ #pragma once #include "fdbclient/NativeAPI.h" -#include "fdbserver/core/ServerCheckpoint.h" +#include "fdbserver/checkpoint/Checkpoint.h" #include "flow/flow.h" class ICheckpointByteSampleReader { diff --git a/fdbserver/core/BulkDumpUtil.cpp b/fdbserver/core/BulkDumpUtil.cpp index 254689ae87b..5012582d6a9 100644 --- a/fdbserver/core/BulkDumpUtil.cpp +++ b/fdbserver/core/BulkDumpUtil.cpp @@ -22,12 +22,11 @@ #include "fdbclient/BulkLoading.h" #include "fdbclient/FDBTypes.h" #include "fdbclient/KeyRangeMap.h" +#include "fdbclient/NativeAPI.h" #include "fdbclient/S3Client.h" #include "fdbserver/core/BulkDumpUtil.h" #include "fdbserver/core/BulkLoadUtil.h" #include "fdbserver/core/Knobs.h" -#include "fdbserver/core/RocksDBCheckpointUtils.h" -#include "fdbserver/core/StorageMetrics.h" SSBulkDumpTask getSSBulkDumpTask(const std::map>& locations, const BulkDumpState& bulkDumpState) { StorageServerInterface targetServer; @@ -81,95 +80,6 @@ std::pair getLocalRemoteFileSetSetting(Version return std::make_pair(fileSetLocal, fileSetRemote); } -// Generate SST file given the input sortedKVS to the input filePath. -// TODO(BulkDump): This copy of sortedKVS can be a slow task if data is large. -void writeKVSToSSTFile(std::string filePath, std::map& sortedKVS, UID logId) { - const std::string absFilePath = abspath(filePath); - // Check file - if (fileExists(absFilePath)) { - TraceEvent(SevWarn, "SSBulkDumpRetriableError", logId) - .detail("Reason", "exist old File when writeKVSToSSTFile") - .detail("DataFilePathLocal", absFilePath); - ASSERT_WE_THINK(false); - throw retry(); - } - // Dump data to file - std::unique_ptr sstWriter = newRocksDBSstFileWriter(); - sstWriter->open(absFilePath); - for (const auto& [key, value] : sortedKVS) { - sstWriter->write(key, value); // assuming sorted - } - if (!sstWriter->finish()) { - // Unexpected: having data but failed to finish - TraceEvent(SevWarn, "SSBulkDumpRetriableError", logId) - .detail("Reason", "failed to finish data sst writer when writeKVSToSSTFile") - .detail("DataFilePath", absFilePath); - ASSERT_WE_THINK(false); - throw retry(); - } - return; -} - -Future dumpDataFileToLocalDirectory(UID logId, - std::shared_ptr rangeDumpRawData, - BulkLoadFileSet localFileSet, - BulkLoadFileSet remoteFileSet, - BulkLoadByteSampleSetting byteSampleSetting, - Version dumpVersion, - KeyRange dumpRange, - BulkLoadType dumpType, - BulkLoadTransportMethod transportMethod) { - // Step 1: Clean up local folder - resetFileFolder((abspath(localFileSet.getFolder()))); - - // Step 2: Dump data to file - bool containDataFile = false; - if (!rangeDumpRawData->kvs.empty()) { - writeKVSToSSTFile(abspath(localFileSet.getDataFileFullPath()), rangeDumpRawData->kvs, logId); - containDataFile = true; - } else { - ASSERT(rangeDumpRawData->sampled.empty()); - containDataFile = false; - } - - // Step 3: Dump sample to file - bool containByteSampleFile = false; - if (!rangeDumpRawData->sampled.empty()) { - writeKVSToSSTFile(abspath(localFileSet.getBytesSampleFileFullPath()), rangeDumpRawData->sampled, logId); - containByteSampleFile = true; - } else { - containByteSampleFile = false; - } - - // Step 4: Generate manifest file - if (fileExists(abspath(localFileSet.getManifestFileFullPath()))) { - TraceEvent(SevWarn, "SSBulkDumpRetriableError", logId) - .detail("Reason", "exist old manifestFile") - .detail("ManifestFilePathLocal", abspath(localFileSet.getManifestFileFullPath())); - ASSERT_WE_THINK(false); - throw retry(); - } - BulkLoadFileSet fileSetRemote(remoteFileSet.getRootPath(), - remoteFileSet.getRelativePath(), - remoteFileSet.getManifestFileName(), - containDataFile ? remoteFileSet.getDataFileName() : std::string(), - containByteSampleFile ? remoteFileSet.getByteSampleFileName() : std::string(), - BulkLoadChecksum()); - BulkLoadManifest manifestMetadata(fileSetRemote, - dumpRange.begin, - dumpRange.end, - dumpVersion, - rangeDumpRawData->kvsBytes, - rangeDumpRawData->kvs.size(), - byteSampleSetting, - dumpType, - transportMethod); - std::string manifestStr = manifestMetadata.toString(); - std::shared_ptr manifest = std::make_shared(std::move(manifestStr)); - co_await writeBulkFileBytes(abspath(localFileSet.getManifestFileFullPath()), manifest); - co_return manifestMetadata; -} - // Validate the invariant of filenames. Source is the file stored locally. Destination is the file going to move to. bool validateSourceDestinationFileSets(const BulkLoadFileSet& source, const BulkLoadFileSet& destination) { // Manifest file must be present diff --git a/fdbserver/core/BulkLoadUtil.cpp b/fdbserver/core/BulkLoadUtil.cpp index de032445189..701ef7c13d2 100644 --- a/fdbserver/core/BulkLoadUtil.cpp +++ b/fdbserver/core/BulkLoadUtil.cpp @@ -24,8 +24,6 @@ #include "fdbclient/S3Client.h" #include "fdbserver/core/BulkLoadUtil.h" #include "fdbserver/core/Knobs.h" -#include "fdbserver/core/RocksDBCheckpointUtils.h" -#include "fdbserver/core/StorageMetrics.h" #include "flow/genericactors.h" #include "flow/UnitTest.h" @@ -182,71 +180,6 @@ Future getBulkLoadTaskStateFromDataMove(Database cx, } } -// Return true if generated the byte sampling file. Otherwise, return false. -// TODO(BulkDump): directly read from special key space. -Future doBytesSamplingOnDataFile(std::string dataFileFullPath, // input file - std::string byteSampleFileFullPath, // output file - UID logId) { - int counter = 0; - bool res = false; - int retryCount = 0; - double startTime = now(); - while (true) { - Error err; - try { - std::unique_ptr sstWriter = newRocksDBSstFileWriter(); - sstWriter->open(abspath(byteSampleFileFullPath)); - bool anySampled = false; - std::unique_ptr reader = newRocksDBSstFileReader(); - reader->open(abspath(dataFileFullPath)); - while (reader->hasNext()) { - KeyValue kv = reader->next(); - ByteSampleInfo sampleInfo = isKeyValueInSample(kv); - if (sampleInfo.inSample) { - sstWriter->write(kv.key, BinaryWriter::toValue(sampleInfo.sampledSize, Unversioned())); - anySampled = true; - counter++; - if (counter > SERVER_KNOBS->BULKLOAD_BYTE_SAMPLE_BATCH_KEY_COUNT) { - co_await yield(); - counter = 0; - } - } - } - // It is possible that no key is sampled - // This can happen when the data to sample is small - // In this case, no SST sample byte file is generated - if (anySampled) { - ASSERT(sstWriter->finish()); - res = true; - } else { - ASSERT(!sstWriter->finish()); - deleteFile(abspath(byteSampleFileFullPath)); - } - break; - } catch (Error& e) { - err = e; - } - if (err.code() == error_code_actor_cancelled) { - throw err; - } - TraceEvent(SevWarn, "SSBulkLoadTaskSamplingError", logId) - .errorUnsuppressed(err) - .detail("DataFileFullPath", dataFileFullPath) - .detail("ByteSampleFileFullPath", byteSampleFileFullPath) - .detail("Duration", now() - startTime) - .detail("RetryCount", retryCount); - co_await delay(5.0); - deleteFile(abspath(byteSampleFileFullPath)); - retryCount++; - } - TraceEvent(bulkLoadVerboseEventSev(), "SSBulkLoadTaskSamplingComplete", logId) - .detail("DataFileFullPath", dataFileFullPath) - .detail("ByteSampleFileFullPath", byteSampleFileFullPath) - .detail("Duration", now() - startTime) - .detail("RetryCount", retryCount); - co_return res; -} - // TODO(BulkLoad): slow task void clearFileFolder(const std::string& folderPath, const UID& logId, bool ignoreError) { try { diff --git a/fdbserver/core/CMakeLists.txt b/fdbserver/core/CMakeLists.txt index a94b1df9681..78af86ad3b5 100644 --- a/fdbserver/core/CMakeLists.txt +++ b/fdbserver/core/CMakeLists.txt @@ -21,15 +21,3 @@ target_include_directories(fdbserver_core PRIVATE ${CMAKE_SOURCE_DIR}/fdbserver/include) target_link_libraries(fdbserver_core PUBLIC fdbclient) - -if(WITH_ROCKSDB) - add_dependencies(fdbserver_core rocksdb) - if(WITH_LIBURING) - target_include_directories(fdbserver_core PRIVATE ${ROCKSDB_INCLUDE_DIR} ${uring_INCLUDE_DIR}) - target_link_libraries(fdbserver_core PRIVATE ${ROCKSDB_LIBRARIES} ${uring_LIBRARIES} ${LZ4_LIBRARY}) - else() - target_include_directories(fdbserver_core PRIVATE ${ROCKSDB_INCLUDE_DIR}) - target_link_libraries(fdbserver_core PRIVATE ${ROCKSDB_LIBRARIES} ${LZ4_LIBRARY}) - endif() - target_compile_definitions(fdbserver_core PUBLIC WITH_ROCKSDB) -endif() diff --git a/fdbserver/core/ServerCheckpoint.cpp b/fdbserver/core/ServerCheckpoint.cpp index 39bf743aa58..0869802bd0f 100644 --- a/fdbserver/core/ServerCheckpoint.cpp +++ b/fdbserver/core/ServerCheckpoint.cpp @@ -1,5 +1,5 @@ /* - *ServerCheckpoint.cpp + * ServerCheckpoint.cpp * * This source file is part of the FoundationDB open source project * @@ -19,84 +19,6 @@ */ #include "fdbserver/core/ServerCheckpoint.h" -#include "fdbserver/core/RocksDBCheckpointUtils.h" - -ICheckpointReader* newCheckpointReader(const CheckpointMetaData& checkpoint, - const CheckpointAsKeyValues checkpointAsKeyValues, - UID logID) { - const CheckpointFormat format = checkpoint.getFormat(); - if (format == DataMoveRocksCF || format == RocksDB) { - return newRocksDBCheckpointReader(checkpoint, checkpointAsKeyValues, logID); - } else { - throw not_implemented(); - } - - return nullptr; -} - -Future deleteCheckpoint(CheckpointMetaData checkpoint) { - co_await delay(0, TaskPriority::FetchKeys); - const CheckpointFormat format = checkpoint.getFormat(); - if (format == DataMoveRocksCF || format == RocksDB || format == RocksDBKeyValues) { - if (!checkpoint.dir.empty()) { - platform::eraseDirectoryRecursive(checkpoint.dir); - } else { - TraceEvent(SevWarn, "CheckpointDirNotFound").detail("Checkpoint", checkpoint.toString()); - } - } else { - throw not_implemented(); - } -} - -Future fetchCheckpoint(Database cx, - CheckpointMetaData initialState, - std::string dir, - std::function(const CheckpointMetaData&)> cFun) { - TraceEvent("FetchCheckpointBegin", initialState.checkpointID).detail("CheckpointMetaData", initialState.toString()); - - CheckpointMetaData result; - const CheckpointFormat format = initialState.getFormat(); - ASSERT(format != RocksDBKeyValues); - if (format == DataMoveRocksCF || format == RocksDB) { - result = co_await fetchRocksDBCheckpoint(cx, initialState, dir, cFun); - } else { - throw not_implemented(); - } - - TraceEvent("FetchCheckpointEnd", initialState.checkpointID).detail("CheckpointMetaData", result.toString()); - co_return result; -} - -Future fetchCheckpointRanges(Database cx, - CheckpointMetaData initialState, - std::string dir, - std::vector ranges, - std::function(const CheckpointMetaData&)> cFun) { - TraceEvent(SevDebug, "FetchCheckpointRangesBegin", initialState.checkpointID) - .detail("CheckpointMetaData", initialState.toString()) - .detail("Ranges", describe(ranges)); - ASSERT(!ranges.empty()); - - CheckpointMetaData result; - const CheckpointFormat format = initialState.getFormat(); - if (format != RocksDBKeyValues) { - if (format != DataMoveRocksCF) { - throw not_implemented(); - } - initialState.setFormat(RocksDBKeyValues); - initialState.ranges = ranges; - initialState.dir = dir; - initialState.setSerializedCheckpoint( - ObjectWriter::toValue(RocksDBCheckpointKeyValues(ranges), IncludeVersion())); - } - - result = co_await fetchRocksDBCheckpoint(cx, initialState, dir, cFun); - - TraceEvent(SevDebug, "FetchCheckpointRangesEnd", initialState.checkpointID) - .detail("CheckpointMetaData", result.toString()) - .detail("Ranges", describe(ranges)); - co_return result; -} std::string serverCheckpointDir(const std::string& baseDir, const UID& checkpointId) { return joinPath(baseDir, checkpointId.toString()); diff --git a/fdbserver/core/include/fdbserver/core/BulkDumpUtil.h b/fdbserver/core/include/fdbserver/core/BulkDumpUtil.h index 88f798f3061..1725f25d8e4 100644 --- a/fdbserver/core/include/fdbserver/core/BulkDumpUtil.h +++ b/fdbserver/core/include/fdbserver/core/BulkDumpUtil.h @@ -86,21 +86,6 @@ std::string generateBulkDumpJobFolder(const UID& jobId); // Define task folder name. std::string getBulkDumpJobTaskFolder(const UID& jobId, const UID& taskId); -// Generate key-value data, byte sampling data, and manifest file. -// Return BulkLoadManifest metadata (equivalent to content of the manifest file). -// TODO(BulkDump): can cause slow tasks, do the task in a separate thread in the future. -// The size of sortedData is defined at the place of generating the data (getRangeDataToDump). -// The size is configured by MOVE_SHARD_KRM_ROW_LIMIT. -Future dumpDataFileToLocalDirectory(UID logId, - std::shared_ptr rangeDumpRawData, - BulkLoadFileSet localFileSet, - BulkLoadFileSet remoteFileSet, - BulkLoadByteSampleSetting byteSampleSetting, - Version dumpVersion, - KeyRange dumpRange, - BulkLoadType dumpType, - BulkLoadTransportMethod transportMethod); - // Upload manifest file for bulkdump job // Each job has one manifest file including manifest paths of all tasks. // The local file path: /-manifest.txt diff --git a/fdbserver/core/include/fdbserver/core/BulkLoadUtil.h b/fdbserver/core/include/fdbserver/core/BulkLoadUtil.h index 3805ab09e71..4ad7887a79c 100644 --- a/fdbserver/core/include/fdbserver/core/BulkLoadUtil.h +++ b/fdbserver/core/include/fdbserver/core/BulkLoadUtil.h @@ -56,8 +56,6 @@ Future bulkLoadDownloadTaskFileSets(BulkLoadTransportMethod transportMetho std::string toLocalRoot, UID logId); -Future doBytesSamplingOnDataFile(std::string dataFileFullPath, std::string byteSampleFileFullPath, UID logId); - // Download job manifest file which is generated when dumping the data Future downloadBulkLoadJobManifestFile(BulkLoadTransportMethod transportMethod, std::string localJobManifestFilePath, diff --git a/fdbserver/core/include/fdbserver/core/ServerCheckpoint.h b/fdbserver/core/include/fdbserver/core/ServerCheckpoint.h index 1f0458e8121..3362e7d1c3b 100644 --- a/fdbserver/core/include/fdbserver/core/ServerCheckpoint.h +++ b/fdbserver/core/include/fdbserver/core/ServerCheckpoint.h @@ -57,27 +57,5 @@ class ICheckpointReader { virtual ~ICheckpointReader() = default; }; -ICheckpointReader* newCheckpointReader(const CheckpointMetaData& checkpoint, - const CheckpointAsKeyValues checkpointAsKeyValues, - UID logID); - -// Delete a checkpoint. -Future deleteCheckpoint(CheckpointMetaData checkpoint); - -// Fetches checkpoint to a local `dir`, `initialState` provides the checkpoint formats, location, restart point, etc. -// If cFun is provided, the progress can be checkpointed. -// Returns a CheckpointMetaData, which could contain KVS-specific results, e.g., the list of fetched checkpoint files. -Future fetchCheckpoint(Database cx, - CheckpointMetaData initialState, - std::string dir, - std::function(const CheckpointMetaData&)> cFun = nullptr); - -// Same as above, except that the checkpoint is fetched as key-value pairs. -Future fetchCheckpointRanges(Database cx, - CheckpointMetaData initialState, - std::string dir, - std::vector ranges, - std::function(const CheckpointMetaData&)> cFun = nullptr); - std::string serverCheckpointDir(const std::string& baseDir, const UID& checkpointId); std::string fetchedCheckpointDir(const std::string& baseDir, const UID& checkpointId); diff --git a/fdbserver/kvstore/CMakeLists.txt b/fdbserver/kvstore/CMakeLists.txt index 677fc092698..e3885a8240a 100644 --- a/fdbserver/kvstore/CMakeLists.txt +++ b/fdbserver/kvstore/CMakeLists.txt @@ -18,7 +18,7 @@ target_include_directories(fdbserver_kvstore ${CMAKE_CURRENT_SOURCE_DIR} ${CMAKE_SOURCE_DIR}/contrib/sqlite ${CMAKE_SOURCE_DIR}/fdbserver/include) -target_link_libraries(fdbserver_kvstore PUBLIC fdbserver_core sqlite) +target_link_libraries(fdbserver_kvstore PUBLIC fdbserver_core sqlite PRIVATE fdbserver_checkpoint) if(WITH_ROCKSDB) add_dependencies(fdbserver_kvstore rocksdb) diff --git a/fdbserver/kvstore/KeyValueStoreRocksDB.cpp b/fdbserver/kvstore/KeyValueStoreRocksDB.cpp index 048cb6c66cc..277e2bbd3ef 100644 --- a/fdbserver/kvstore/KeyValueStoreRocksDB.cpp +++ b/fdbserver/kvstore/KeyValueStoreRocksDB.cpp @@ -68,7 +68,7 @@ #include "fdbserver/core/Knobs.h" #include "fdbserver/kvstore/IKeyValueStore.h" -#include "fdbserver/core/RocksDBCheckpointUtils.h" +#include "fdbserver/checkpoint/RocksDBCheckpointUtils.h" #include "RocksDBCommon.h" #include "flow/CoroUtils.h" diff --git a/fdbserver/kvstore/KeyValueStoreShardedRocksDB.cpp b/fdbserver/kvstore/KeyValueStoreShardedRocksDB.cpp index 721e4983e98..23cfc3e5770 100644 --- a/fdbserver/kvstore/KeyValueStoreShardedRocksDB.cpp +++ b/fdbserver/kvstore/KeyValueStoreShardedRocksDB.cpp @@ -44,7 +44,7 @@ #include "fdbserver/core/Knobs.h" #include "fdbserver/kvstore/IKeyValueStore.h" -#include "fdbserver/core/RocksDBCheckpointUtils.h" +#include "fdbserver/checkpoint/RocksDBCheckpointUtils.h" #include "RocksDBCommon.h" #ifdef WITH_ROCKSDB diff --git a/fdbserver/storageserver/CMakeLists.txt b/fdbserver/storageserver/CMakeLists.txt index 9a7d8c76d36..056ff505dcc 100644 --- a/fdbserver/storageserver/CMakeLists.txt +++ b/fdbserver/storageserver/CMakeLists.txt @@ -16,4 +16,4 @@ target_include_directories(fdbserver_storageserver ${CMAKE_CURRENT_SOURCE_DIR}/include PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}) -target_link_libraries(fdbserver_storageserver PRIVATE fdbserver_core fdbserver_kvstore fdbserver_logsystem) +target_link_libraries(fdbserver_storageserver PRIVATE fdbserver_core fdbserver_checkpoint fdbserver_kvstore fdbserver_logsystem) diff --git a/fdbserver/storageserver/storageserver.cpp b/fdbserver/storageserver/storageserver.cpp index 5c7af2ea86a..3fdf09b0932 100644 --- a/fdbserver/storageserver/storageserver.cpp +++ b/fdbserver/storageserver/storageserver.cpp @@ -42,6 +42,8 @@ #include "fdbserver/core/AccumulativeChecksumUtil.h" #include "fdbserver/core/BulkDumpUtil.h" #include "fdbserver/core/BulkLoadUtil.h" +#include "fdbserver/checkpoint/BulkSstFiles.h" +#include "fdbserver/checkpoint/Checkpoint.h" #include "fdbserver/core/FDBSimulationPolicy.h" #include "fdbserver/core/FDBRocksDBVersion.h" #include "fdbserver/kvstore/IKeyValueStore.h" @@ -55,8 +57,7 @@ #include "MappedKeyPlan.h" #include "ReadLatencySamples.h" #include "fdbserver/core/RecoveryState.h" -#include "fdbserver/core/RocksDBCheckpointUtils.h" -#include "fdbserver/core/ServerCheckpoint.h" +#include "fdbserver/checkpoint/RocksDBCheckpointUtils.h" #include "fdbserver/core/SpanContextMessage.h" #include "fdbserver/storageserver/StorageCorruptionBug.h" #include "fdbserver/core/StorageMetrics.h" diff --git a/fdbserver/worker/CMakeLists.txt b/fdbserver/worker/CMakeLists.txt index 74b8e01d846..0fd50242475 100644 --- a/fdbserver/worker/CMakeLists.txt +++ b/fdbserver/worker/CMakeLists.txt @@ -70,3 +70,7 @@ target_link_libraries(fdbserver_worker fdbserver_tester fdbserver_tlog fdbserver_core) + +if(WITH_ROCKSDB) + target_compile_definitions(fdbserver_worker PRIVATE WITH_ROCKSDB) +endif() diff --git a/fdbserver/workloads/BulkLoading.cpp b/fdbserver/workloads/BulkLoading.cpp index 4a04b6930e0..3dd3d7fd28f 100644 --- a/fdbserver/workloads/BulkLoading.cpp +++ b/fdbserver/workloads/BulkLoading.cpp @@ -24,7 +24,7 @@ #include "fdbclient/RangeLock.h" #include "fdbclient/SystemData.h" #include "fdbserver/core/BulkLoadUtil.h" -#include "fdbserver/core/RocksDBCheckpointUtils.h" +#include "fdbserver/checkpoint/RocksDBCheckpointUtils.h" #include "fdbserver/core/StorageMetrics.h" #include "fdbserver/tester/workloads.h" diff --git a/fdbserver/workloads/CMakeLists.txt b/fdbserver/workloads/CMakeLists.txt index 76564bbaea6..aadaf00b858 100644 --- a/fdbserver/workloads/CMakeLists.txt +++ b/fdbserver/workloads/CMakeLists.txt @@ -14,6 +14,7 @@ target_include_directories(fdbserver_workloads PRIVATE target_link_libraries(fdbserver_workloads PRIVATE fdbserver_consistencyscan fdbserver_core + fdbserver_checkpoint fdbserver_kvstore fdbserver_worker fdbserver_tester @@ -21,3 +22,7 @@ target_link_libraries(fdbserver_workloads PRIVATE fdbserver_resolver fdbserver_storageserver fdbserver_mocks3) + +if(WITH_ROCKSDB) + target_compile_definitions(fdbserver_workloads PRIVATE WITH_ROCKSDB) +endif() diff --git a/fdbserver/workloads/PhysicalShardMove.cpp b/fdbserver/workloads/PhysicalShardMove.cpp index e23c9e7c642..b5fb9d3a296 100644 --- a/fdbserver/workloads/PhysicalShardMove.cpp +++ b/fdbserver/workloads/PhysicalShardMove.cpp @@ -25,7 +25,7 @@ #include "fdbserver/core/Knobs.h" #include "fdbserver/core/MoveKeys.h" #include "fdbserver/core/QuietDatabase.h" -#include "fdbserver/core/ServerCheckpoint.h" +#include "fdbserver/checkpoint/Checkpoint.h" #include "fdbserver/tester/workloads.h" #include "flow/Error.h" #include "flow/IRandom.h" diff --git a/fdbserver/workloads/StorageServerCheckpointRestoreTest.cpp b/fdbserver/workloads/StorageServerCheckpointRestoreTest.cpp index 965ff6fe4eb..a904b460047 100644 --- a/fdbserver/workloads/StorageServerCheckpointRestoreTest.cpp +++ b/fdbserver/workloads/StorageServerCheckpointRestoreTest.cpp @@ -23,7 +23,7 @@ #include "fdbrpc/simulator.h" #include "fdbserver/kvstore/IKeyValueStore.h" #include "fdbserver/core/Knobs.h" -#include "fdbserver/core/ServerCheckpoint.h" +#include "fdbserver/checkpoint/Checkpoint.h" #include "fdbserver/core/MoveKeys.h" #include "fdbserver/core/QuietDatabase.h" #include "fdbserver/tester/workloads.h" From 79314be3adccdd9ce079fd99ed3a98a9fa7cac14 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 6 Sep 2026 12:02:38 -0700 Subject: [PATCH 037/170] Capture request state explicitly in lagging factory --- fdbrpc/include/fdbrpc/LoadBalance.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fdbrpc/include/fdbrpc/LoadBalance.h b/fdbrpc/include/fdbrpc/LoadBalance.h index 4d51aa31afc..184f9e9a108 100644 --- a/fdbrpc/include/fdbrpc/LoadBalance.h +++ b/fdbrpc/include/fdbrpc/LoadBalance.h @@ -268,7 +268,7 @@ struct RequestData : NonCopyable { ASSERT(modelHolder->model); QueueModel* model = modelHolder->model; - model->addLaggingRequest([&] { + model->addLaggingRequest([this] { // We need to process the lagging request in order to update the queue model Reference holderCapture = std::move(modelHolder); auto triedAllOptionsCapture = triedAllOptions; From 5080fba175cecbd966dfb36117583d5fd03c7f48 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 6 Sep 2026 13:13:05 -0700 Subject: [PATCH 038/170] Keep checkpoint CI checks aligned with moved source --- .github/workflows/tidy.yml | 2 +- fdbserver/checkpoint/BulkSstFiles.cpp | 1 - 2 files changed, 1 insertion(+), 2 deletions(-) diff --git a/.github/workflows/tidy.yml b/.github/workflows/tidy.yml index ec8a3e6264d..13193cf8caf 100644 --- a/.github/workflows/tidy.yml +++ b/.github/workflows/tidy.yml @@ -99,7 +99,7 @@ jobs: fdbrpc/AsyncFileKAIO.h) continue ;; - fdbserver/core/RocksDBCheckpointUtils.cpp|fdbserver/kvstore/KeyValueStoreRocksDB.cpp|fdbserver/kvstore/KeyValueStoreShardedRocksDB.cpp) + fdbserver/checkpoint/RocksDBCheckpointUtils.cpp|fdbserver/kvstore/KeyValueStoreRocksDB.cpp|fdbserver/kvstore/KeyValueStoreShardedRocksDB.cpp) continue ;; flow/include/flow/Coroutines.h|flow/include/flow/CoroutinesImpl.h|flow/include/flow/FlowThread.h) diff --git a/fdbserver/checkpoint/BulkSstFiles.cpp b/fdbserver/checkpoint/BulkSstFiles.cpp index c1c0f16ab1f..7444d343f78 100644 --- a/fdbserver/checkpoint/BulkSstFiles.cpp +++ b/fdbserver/checkpoint/BulkSstFiles.cpp @@ -114,7 +114,6 @@ Future dumpDataFileToLocalDirectory(UID logId, co_return manifestMetadata; } - // Return true if generated the byte sampling file. Otherwise, return false. // TODO(BulkDump): directly read from special key space. Future doBytesSamplingOnDataFile(std::string dataFileFullPath, // input file From 43756f8b076947ab2aef5fc89a2d0684a2034b0d Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 6 Sep 2026 13:30:46 -0700 Subject: [PATCH 039/170] Name CDC reply mutation boolean parameter --- fdbserver/cdcproxy/CDCProxy.cpp | 38 +++++++++++++++++++-------------- 1 file changed, 22 insertions(+), 16 deletions(-) diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index c69a23e7d8f..4214fa7e454 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -74,6 +74,8 @@ struct CDCTagInterval { struct CDCBufferedTag; +FDB_BOOLEAN_PARAM(HasMutations); + // Speculative buffering never proves a client cursor and never creates another read-ahead credit. class CDCStreamReadAhead { enum class State { Idle, Armed, Claimed }; @@ -86,7 +88,7 @@ class CDCStreamReadAhead { bool provesCursor(Version cursor, Version minVersion) const { return cursor <= std::max(issuedReplyThrough, minVersion - 1); } - bool issueReply(Version through, Version bufferedThrough, Version minVersion, bool hasMutations) { + bool issueReply(Version through, Version bufferedThrough, Version minVersion, HasMutations hasMutations) { const bool advanced = !provesCursor(through, minVersion); if (state == State::Armed && creditThrough != bufferedThrough) { cancel(); @@ -1919,8 +1921,10 @@ Future CDCProxy::consume(CDCConsumeRequest request) { reply.lastConsumedVersion = selectedCDCConsumeReplyThrough(selection, stream->bufferedThrough); // Record proof before send(), whose callbacks may run synchronously. Empty or capped replies do not // extend the speculative horizon; neither does replaying an already issued cursor. - const bool armed = stream->readAhead.issueReply( - reply.lastConsumedVersion, stream->bufferedThrough, stream->minVersion, !reply.mutations.empty()); + const bool armed = stream->readAhead.issueReply(reply.lastConsumedVersion, + stream->bufferedThrough, + stream->minVersion, + HasMutations(!reply.mutations.empty())); request.reply.send(reply); if (armed) { refreshStreamTags(stream); @@ -2359,7 +2363,7 @@ class CDCProxyPrefetchTest { stream->keys = KeyRangeRef("a"_sr, "z"_sr); stream->tagIntervals.emplace_back(tag->tag, 1, 200); stream->tagIntervals.back().bufferedThrough = 99; - ASSERT(stream->readAhead.issueReply(99, 99, 1, true)); + ASSERT(stream->readAhead.issueReply(99, 99, 1, HasMutations::True)); proxy.streams[id] = stream; proxy.tags[tag->tag] = tag; tag->streamIds.insert(id); @@ -2589,7 +2593,7 @@ class CDCProxyPrefetchTest { second->bufferedThrough = 199; second->tagIntervals.back().bufferedThrough = 199; second->tagIntervals.back().end = 300; - ASSERT(second->readAhead.issueReply(199, 199, second->minVersion, true)); + ASSERT(second->readAhead.issueReply(199, 199, second->minVersion, HasMutations::True)); auto firstCursor = makeReference(Void()); ASSERT_EQ(test.proxy.nextTagPrefetchVersion(test.tag).get(), 100); @@ -2725,17 +2729,18 @@ TEST_CASE("/NativeCDC/PrefetchCreditLifecycle") { CDCStreamReadAhead credit; auto tag = makeReference(Tag(tagLocalityCDC, 0)); ASSERT(!credit.armedFor(99)); - ASSERT(!credit.issueReply(99, 99, 100, true)); // Already covered by the durable floor. - ASSERT(!credit.issueReply(100, 100, 100, false)); // Empty progress is proved, but grants no lookahead. + ASSERT(!credit.issueReply(99, 99, 100, HasMutations::True)); // Already covered by the durable floor. + ASSERT( + !credit.issueReply(100, 100, 100, HasMutations::False)); // Empty progress is proved, but grants no lookahead. ASSERT(credit.provesCursor(100, 100)); - ASSERT(!credit.issueReply(101, 102, 100, true)); // A capped reply has not drained the buffered tail. - ASSERT(credit.issueReply(102, 102, 100, true)); + ASSERT(!credit.issueReply(101, 102, 100, HasMutations::True)); // A capped reply has not drained the buffered tail. + ASSERT(credit.issueReply(102, 102, 100, HasMutations::True)); ASSERT(credit.claim(tag.getPtr(), 102)); - ASSERT(!credit.issueReply(103, 103, 100, true)); // No credit banking while a pass is active. + ASSERT(!credit.issueReply(103, 103, 100, HasMutations::True)); // No credit banking while a pass is active. credit.finish(tag.getPtr()); ASSERT(!credit.armedFor(103)); - ASSERT(!credit.issueReply(103, 103, 100, true)); // Replayed cursor. - ASSERT(credit.issueReply(104, 104, 100, true)); + ASSERT(!credit.issueReply(103, 103, 100, HasMutations::True)); // Replayed cursor. + ASSERT(credit.issueReply(104, 104, 100, HasMutations::True)); credit.cancel(); ASSERT(!credit.claim(tag.getPtr(), 104)); ASSERT(!credit.provesCursor(105, 100)); @@ -2753,7 +2758,7 @@ TEST_CASE("/NativeCDC/PrefetchCreditTailAndTags") { stream->tagIntervals.emplace_back(first->tag, 1, 101); stream->tagIntervals.emplace_back(second->tag, 101, 200); stream->tagIntervals[0].bufferedThrough = 99; - ASSERT(stream->readAhead.issueReply(99, 99, 1, true)); + ASSERT(stream->readAhead.issueReply(99, 99, 1, HasMutations::True)); ASSERT_EQ(nextCDCPrefetchVersion(stream, first).get(), 100); ASSERT(!nextCDCPrefetchVersion(stream, second).present()); { @@ -2764,14 +2769,15 @@ TEST_CASE("/NativeCDC/PrefetchCreditTailAndTags") { ASSERT(!nextCDCPrefetchVersion(stream, second).present()); } ASSERT(!stream->readAhead.claimedBy(first.getPtr())); - ASSERT(stream->readAhead.issueReply(100, 100, 1, true)); + ASSERT(stream->readAhead.issueReply(100, 100, 1, HasMutations::True)); stream->bufferedThrough = 102; // Real demand filled more data before the credit could start. stream->tagIntervals[0].bufferedThrough = 100; stream->tagIntervals[1].bufferedThrough = 102; ASSERT(!nextCDCPrefetchVersion(stream, second).present()); - ASSERT(!stream->readAhead.issueReply(101, 102, 1, true)); // Capped reply cannot revive the old credit. + ASSERT( + !stream->readAhead.issueReply(101, 102, 1, HasMutations::True)); // Capped reply cannot revive the old credit. ASSERT(!nextCDCPrefetchVersion(stream, second).present()); - ASSERT(stream->readAhead.issueReply(102, 102, 1, true)); + ASSERT(stream->readAhead.issueReply(102, 102, 1, HasMutations::True)); ASSERT_EQ(nextCDCPrefetchVersion(stream, second).get(), 103); return Void(); } From 5c16a58174de2eddcafb86232a26b4fc202ddcf9 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 6 Sep 2026 14:32:47 -0700 Subject: [PATCH 040/170] Balance native CDC tag groups across live proxies --- design/cdc.md | 28 ++ fdbclient/NativeCdc.cpp | 161 ++++++++++- fdbclient/NativeCdcInternal.h | 8 + .../clustercontroller/ClusterController.cpp | 49 ++++ fdbserver/core/ServerKnobs.cpp | 2 + fdbserver/core/include/fdbserver/core/Knobs.h | 2 + fdbserver/workloads/CMakeLists.txt | 3 +- fdbserver/workloads/NativeCdcEndToEnd.cpp | 254 ++++++++++++++++++ tests/CMakeLists.txt | 2 + tests/fast/NativeCdcProxyRebalance.toml | 33 +++ .../NativeCdcProxyRebalanceAutomatic.toml | 34 +++ 11 files changed, 573 insertions(+), 3 deletions(-) create mode 100644 tests/fast/NativeCdcProxyRebalance.toml create mode 100644 tests/fast/NativeCdcProxyRebalanceAutomatic.toml diff --git a/design/cdc.md b/design/cdc.md index d5bfe37da2e..2e95accbd52 100644 --- a/design/cdc.md +++ b/design/cdc.md @@ -412,6 +412,34 @@ validation also rejects stale representatives left by older metadata writers. The allocator's stream-count scan remains necessary, and ownership discovery still scans global metadata when the representative is absent or invalid. +### Balancing ownership between live CDC proxies + +`CDC_PROXY_REBALANCE_ENABLED` is disabled by default. When enabled, the cluster +controller makes at most one live ownership move per +`CDC_PROXY_REBALANCE_INTERVAL` (60 seconds by default) while fully recovered. +It groups active streams by their current CDC tag and moves a complete group +only when that strictly reduces the stream-count difference between two +published proxies. A group with mixed owners is never moved. The controller +skips a pass when the metadata exceeds 512 active streams or assignments, or +2,048 tag-history rows, or one MiB in any of those three ranges. Each pass has +a five-second transaction timeout and at most three attempts. It skips +individual groups larger than 64 streams or +with an in-progress versionstamped tag transition. These are conservative +balancer limits, not CDC registration or cluster capacity limits. + +The move validates durable streams, tags, and owners in one transaction, changes +every member's assignment, and signals the existing ownership monitor. Its +publication wakes the old proxy to drop buffered state and the new proxy to +reload from durable acknowledgement watermarks. Clients may replay delivered +but unacknowledged mutations, as with proxy replacement. Tag routing, stream +identities, acknowledgement, and safe-pop metadata do not change. Disabling the +balancer or CDC admission stops future moves without requiring a stream drain. + +This policy balances the number of streams, not producer bytes, filtering cost, +or consumer lag. One hot stream cannot be divided among proxies by moving its +whole tag group. Throughput-aware proxy placement needs measured load and a +separate policy; the opt-in tag-retagging controller can inform a later version. + ### Metadata lifecycle example Assume a client registers stream name `orders` for range diff --git a/fdbclient/NativeCdc.cpp b/fdbclient/NativeCdc.cpp index c1403266ac6..2f1154fec03 100644 --- a/fdbclient/NativeCdc.cpp +++ b/fdbclient/NativeCdc.cpp @@ -258,8 +258,8 @@ bool rewindUnacknowledgedCursorAfterProxyReplacement(CDCCursor* currentPosition, return true; } -// TODO: Have the cluster controller rebalance stream ownership using aggregate CDC proxy throughput and -// update cdcProxyKeys and ClientDBInfo assignments; registration currently chooses any available proxy. +// TODO: Use measured aggregate CDC proxy throughput instead of stream counts when balancing ownership; +// registration currently chooses any available proxy before the controller's opt-in balancing pass. Optional selectAvailableNativeCdcProxy(ClientDBInfo const& clientInfo, Optional previousProxy) { for (const auto& proxy : clientInfo.cdcProxies) { if (!previousProxy.present() || proxy.id() != previousProxy.get()) { @@ -724,6 +724,163 @@ Future reassignNativeCdcStreams(Database cx, UID oldProxyId, UID newProxyI } } +Future rebalanceNativeCdcProxyAssignments(Database cx, + std::vector availableProxies, + std::function stillEligible) { + // The metadata is read and the whole tag group is moved in one transaction. Do not split a shared tag + // across owners or make a partial move when the scan approaches the transaction size/lifetime limits. + constexpr int maxStreams = 512; + constexpr int maxHistoryRows = 2048; + constexpr int maxStreamsPerMove = 64; + constexpr int maxRangeBytes = 1 << 20; + const int64_t timeoutMs = 5000; + const std::set available(availableProxies.begin(), availableProxies.end()); + if (available.size() < 2) { + co_return false; + } + + Transaction tr(cx); + int attempts = 0; + while (true) { + if (++attempts > 3) { + co_return false; + } + Error err; + try { + tr.setOption(FDBTransactionOptions::LOCK_AWARE); + tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); + tr.setOption(FDBTransactionOptions::TIMEOUT, + StringRef(reinterpret_cast(&timeoutMs), sizeof(timeoutMs))); + const UID publishedInfoId = cx->clientInfo->get().id; + if (!cx->clientInfo->get().nativeCdcEnabled || !stillEligible()) { + co_return false; + } + + RangeResult active = co_await tr.getRange(cdcStreamKeys, GetRangeLimits(maxStreams + 1, maxRangeBytes)); + RangeResult assignments = co_await tr.getRange(cdcProxyKeys, GetRangeLimits(maxStreams + 1, maxRangeBytes)); + RangeResult histories = + co_await tr.getRange(cdcTagHistoryKeys, GetRangeLimits(maxHistoryRows + 1, maxRangeBytes)); + if (active.more || assignments.more || histories.more || active.size() > maxStreams || + assignments.size() > maxStreams || histories.size() > maxHistoryRows) { + CODE_PROBE(true, "Native CDC proxy rebalancing skips oversized metadata"); + co_return false; + } + + std::set activeIds; + for (const auto& stream : active) { + activeIds.insert(decodeCDCStreamKey(stream.key)); + } + std::map ownerByStream; + for (const auto& assignment : assignments) { + const auto [streamId, owner] = decodeCDCProxyKey(assignment.key); + if (activeIds.contains(streamId) && !ownerByStream.emplace(streamId, owner).second) { + co_return false; + } + } + std::map currentTagByStream; + std::set pendingHistories; + for (const auto& history : histories) { + const CDCTagHistoryEntry entry = decodeCDCTagHistoryKey(history.key); + if (activeIds.contains(entry.streamId)) { + currentTagByStream[entry.streamId] = entry.tag; + if (!history.value.empty()) { + pendingHistories.insert(entry.streamId); + } + } + } + + std::map ownerLoads; + for (const UID& proxyId : available) { + ownerLoads.emplace(proxyId, 0); + } + std::map> membersByTag; + for (const CDCStreamId streamId : activeIds) { + auto owner = ownerByStream.find(streamId); + auto tag = currentTagByStream.find(streamId); + if (owner == ownerByStream.end() || tag == currentTagByStream.end() || + !ownerLoads.contains(owner->second)) { + // Let the cluster controller repair missing or stale ownership before balancing. + co_return false; + } + const auto published = cx->clientInfo->get().streamToCDCProxyId.find(streamId); + if (published == cx->clientInfo->get().streamToCDCProxyId.end() || published->second != owner->second) { + co_return false; + } + ++ownerLoads[owner->second]; + membersByTag[tag->second].push_back(streamId); + } + + Optional selectedTag; + Optional selectedSource; + Optional selectedTarget; + int bestImprovement = 0; + size_t bestGroupSize = 0; + for (const auto& [tag, members] : membersByTag) { + const UID source = ownerByStream.at(members.front()); + for (const CDCStreamId streamId : members) { + if (ownerByStream.at(streamId) != source) { + // Registration relies on every stream sharing a current tag having one owner. + co_return false; + } + } + if (members.size() > maxStreamsPerMove || + std::any_of(members.begin(), members.end(), [&](CDCStreamId id) { + return pendingHistories.contains(id); + })) { + continue; + } + for (const UID& target : available) { + if (source == target) { + continue; + } + const int difference = ownerLoads.at(source) - ownerLoads.at(target); + const int moved = 2 * static_cast(members.size()); + const int after = difference >= moved ? difference - moved : moved - difference; + const int improvement = difference - after; + if (improvement > bestImprovement || + (improvement == bestImprovement && improvement > 0 && members.size() > bestGroupSize)) { + bestImprovement = improvement; + bestGroupSize = members.size(); + selectedTag = tag; + selectedSource = source; + selectedTarget = target; + } + } + } + if (!selectedTag.present()) { + co_return false; + } + if (!stillEligible() || !cx->clientInfo->get().nativeCdcEnabled || + cx->clientInfo->get().id != publishedInfoId || + !containsNativeCdcProxy(cx->clientInfo->get(), selectedSource.get()) || + !containsNativeCdcProxy(cx->clientInfo->get(), selectedTarget.get())) { + co_return false; + } + + for (const CDCStreamId streamId : membersByTag.at(selectedTag.get())) { + tr.clear(cdcProxyKeyFor(streamId, selectedSource.get())); + tr.set(cdcProxyKeyFor(streamId, selectedTarget.get()), Value()); + } + signalNativeCdcProxyAssignmentChange(&tr); + co_await tr.commit(); + CODE_PROBE(true, "Native CDC rebalances an entire shared tag across live proxies"); + TraceEvent("CDCProxyTagRebalanced") + .detail("Tag", selectedTag.get().toString()) + .detail("OldCDCProxyID", selectedSource.get()) + .detail("NewCDCProxyID", selectedTarget.get()) + .detail("StreamCount", bestGroupSize); + co_return true; + } catch (Error& e) { + // An ambiguous commit may already have moved a group. Reconcile on the next controller pass. + if (e.code() == error_code_commit_unknown_result) { + throw; + } + err = e; + } + co_await tr.onError(err); + } +} + Future acknowledgeNativeCdcStream(Database cx, CDCStreamId streamId, Version consumedThrough, diff --git a/fdbclient/NativeCdcInternal.h b/fdbclient/NativeCdcInternal.h index ae52ee32fe1..8ddb5fbd669 100644 --- a/fdbclient/NativeCdcInternal.h +++ b/fdbclient/NativeCdcInternal.h @@ -22,6 +22,8 @@ #define FDBCLIENT_NATIVECDCINTERNAL_H #pragma once +#include + #include "fdbclient/NativeCdc.h" // Durable metadata operations used by CDC server roles. Registration is @@ -33,6 +35,12 @@ Future removeNativeCdcStream(Database cx, Key name, CDCStreamId streamId, Future> listNativeCdcStreams(Database cx); // Atomically moves any streams assigned to a failed proxy to its replacement. Future reassignNativeCdcStreams(Database cx, UID oldProxyId, UID newProxyId); +// Moves at most one complete current-tag group between live proxies when doing so reduces stream-count skew. +// Returns false if the metadata is incomplete, too large to scan safely, or already balanced. +Future rebalanceNativeCdcProxyAssignments( + Database cx, + std::vector availableProxies, + std::function stillEligible = [] { return true; }); // Persists the exclusive unpopped watermark after consuming through a version. // knownAvailableThrough permits a consumer to acknowledge log data it has // already received before that version is visible at a transaction read version. diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index a94a0a30cbf..1ed6380711e 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -2479,6 +2479,54 @@ Future monitorCDCProxyAssignments(ClusterControllerData* self) { } } +Future rebalanceCDCProxyAssignments(ClusterControllerData* self) { + while (true) { + co_await delay(std::max(1.0, SERVER_KNOBS->CDC_PROXY_REBALANCE_INTERVAL)); + if (!SERVER_KNOBS->CDC_PROXY_REBALANCE_ENABLED || !self->db.recoveryData.isValid() || + self->db.serverInfo->get().recoveryState != RecoveryState::FULLY_RECOVERED || + !self->db.clientInfo->get().nativeCdcEnabled) { + continue; + } + const uint64_t expectedRecoveryCount = self->db.recoveryData->cstate.myDBState.recoveryCount; + const std::vector& published = self->db.clientInfo->get().cdcProxies; + if (published.size() < 2 || published.size() != self->db.cdcProxies.size()) { + continue; + } + std::vector available; + available.reserve(published.size()); + for (const auto& proxy : published) { + if (!containsCDCProxy(self->db.cdcProxies, proxy.id())) { + available.clear(); + break; + } + available.push_back(proxy.id()); + } + if (available.size() < 2) { + continue; + } + try { + const std::vector expectedProxies = available; + auto stillEligible = [self, expectedRecoveryCount, expectedProxies] { + return SERVER_KNOBS->CDC_PROXY_REBALANCE_ENABLED && self->db.recoveryData.isValid() && + self->db.recoveryData->cstate.myDBState.recoveryCount == expectedRecoveryCount && + self->db.serverInfo->get().recoveryState == RecoveryState::FULLY_RECOVERED && + self->db.clientInfo->get().nativeCdcEnabled && + self->db.cdcProxies.size() == expectedProxies.size() && + std::all_of(expectedProxies.begin(), expectedProxies.end(), [self](UID proxyId) { + return containsCDCProxy(self->db.cdcProxies, proxyId); + }); + }; + co_await rebalanceNativeCdcProxyAssignments(self->db.db, std::move(available), std::move(stillEligible)); + } catch (Error& e) { + if (e.code() == error_code_actor_cancelled) { + throw; + } + // An ambiguous commit is reconciled by the assignment monitor; the next scheduled pass may try again. + TraceEvent(SevWarn, "CDCProxyRebalanceError", self->id).error(e); + } + } +} + Future updatedChangingDatacenters(ClusterControllerData* self) { // do not change the cluster controller until all the processes have had a chance to register co_await delay(SERVER_KNOBS->WAIT_FOR_GOOD_RECRUITMENT_DELAY); @@ -3508,6 +3556,7 @@ Future clusterControllerCore(ClusterControllerFullInterface interf, self.addActor.send(monitorGlobalConfig(&self.db)); // These actors also drain durable CDC state when new stream registration is disabled. self.addActor.send(monitorCDCProxyAssignments(&self)); + self.addActor.send(rebalanceCDCProxyAssignments(&self)); self.addActor.send(monitorAndRecruitCDCProxies(&self)); self.addActor.send(updatedChangingDatacenters(&self)); self.addActor.send(updatedChangedDatacenters(&self)); diff --git a/fdbserver/core/ServerKnobs.cpp b/fdbserver/core/ServerKnobs.cpp index a3c68152b2f..4a1638458f5 100644 --- a/fdbserver/core/ServerKnobs.cpp +++ b/fdbserver/core/ServerKnobs.cpp @@ -189,6 +189,8 @@ void ServerKnobs::initialize(Randomize randomize, ClientKnobs* clientKnobs, IsSi init( CDC_PROXY_FAILURE_COALESCE_DELAY, 0.0 ); init( CDC_PROXY_POP_MIN_INTERVAL, 0.1 ); if( randomize && buggify() ) CDC_PROXY_POP_MIN_INTERVAL = 0.01; init( CDC_PROXY_POP_SCAN_INTERVAL, 5.0 ); if( randomize && buggify() ) CDC_PROXY_POP_SCAN_INTERVAL = 0.1; + init( CDC_PROXY_REBALANCE_ENABLED, false ); + init( CDC_PROXY_REBALANCE_INTERVAL, 60.0 ); init( APPLY_MUTATION_BYTES, 1e6 ); init( BUGGIFY_RECOVER_MEMORY_LIMIT, 1e6 ); init( BUGGIFY_WORKER_REMOVED_MAX_LAG, 30 ); diff --git a/fdbserver/core/include/fdbserver/core/Knobs.h b/fdbserver/core/include/fdbserver/core/Knobs.h index 2f65564bde0..707fac7b22b 100644 --- a/fdbserver/core/include/fdbserver/core/Knobs.h +++ b/fdbserver/core/include/fdbserver/core/Knobs.h @@ -80,6 +80,8 @@ class SWIFT_CXX_IMMORTAL_SINGLETON_TYPE ServerKnobs : public KnobsImpl addStream(Database cx) { return addStream(cx, randomOverlappingRange()); } Future initializeStreams(Database cx) { + if (testProxyRebalance || testProxyRebalanceAutomatic) { + for (int i = 0; i < initialStreamCount; ++i) { + co_await addStream(cx, KeyRange(KeyRangeRef(keyForIndex(0), keyForIndex(keyCount)))); + } + co_return; + } for (int i = 0; i < initialStreamCount; ++i) { co_await addStream(cx); } @@ -568,6 +577,239 @@ class NativeCdcEndToEndWorkload : public TestWorkload { co_await timeoutError(removeNativeCdcStreamClient(cx, name), operationTimeout); } + Future validateProxyRebalance(Database cx) { + ASSERT_EQ(streams.size(), 3); + const CDCStreamId firstId = streams[0].consumer->position().streamId; + const CDCStreamId otherTagId = streams[1].consumer->position().streamId; + const CDCStreamId sharedTagId = streams[2].consumer->position().streamId; + const auto proxies = cx->clientInfo->get().cdcProxies; + ASSERT_EQ(proxies.size(), 2); + const NativeCdcStatus initial = co_await timeoutError(getNativeCdcStatus(cx), operationTimeout); + ASSERT(initial.metadataComplete); + ASSERT_EQ(initial.tagCount, 2); + ASSERT_EQ(initial.streams.size(), 3); + const auto findStream = [](NativeCdcStatus const& status, CDCStreamId id) -> NativeCdcStreamStatus const& { + const auto stream = std::find_if(status.streams.begin(), status.streams.end(), [&](auto const& candidate) { + return candidate.info.streamId == id; + }); + ASSERT(stream != status.streams.end()); + return *stream; + }; + const auto& first = findStream(initial, firstId); + const auto& otherTag = findStream(initial, otherTagId); + const auto& shared = findStream(initial, sharedTagId); + ASSERT_EQ(first.tags.size(), 1); + ASSERT_EQ(otherTag.tags.size(), 1); + ASSERT_EQ(shared.tags.size(), 1); + const Tag tag = first.tags.front(); + ASSERT_EQ(shared.tags.front(), tag); + ASSERT_NE(otherTag.tags.front(), tag); + co_await checkTagOwner(cx, tag, firstId); + co_await checkTagOwner(cx, otherTag.tags.front(), otherTagId); + + const CDCProxyInterface source = co_await timeoutError(waitForAssignedProxy(cx, firstId), operationTimeout); + const CDCProxyInterface target = proxies[proxies.front().id() == source.id() ? 1 : 0]; + ASSERT_NE(source.id(), target.id()); + for (const auto& stream : initial.streams) { + ASSERT(stream.owner.present()); + ASSERT_EQ(stream.owner.get(), source.id()); + ASSERT(stream.ownerPublished); + } + + const Key key = keyForIndex(keyCount / 2); + const Value beforeMove = "native-cdc-rebalance-before"_sr; + const Version beforeVersion = co_await writeValue(cx, key, beforeMove); + co_await consumeThroughValue(streams[0].consumer, beforeVersion, key, beforeMove); + co_await consumeThroughValue(streams[1].consumer, beforeVersion, key, beforeMove); + const auto containsValue = [](CDCConsumeReply const& reply, Version version, KeyRef key, ValueRef value) { + return std::any_of(reply.mutations.begin(), reply.mutations.end(), [&](auto const& versioned) { + return versioned.version == version && + std::any_of(versioned.mutations.begin(), versioned.mutations.end(), [&](auto const& mutation) { + return mutation.type == MutationRef::SetValue && mutation.param1 == key && + mutation.param2 == value; + }); + }); + }; + bool primed = false; + const double primeDeadline = now() + operationTimeout; + while (streams[2].consumer->position().lastConsumedVersion < beforeVersion) { + CDCConsumeReply reply = co_await timeoutError(streams[2].consumer->consume(), operationTimeout); + primed |= containsValue(reply, beforeVersion, key, beforeMove); + ASSERT_LT(now(), primeDeadline); + } + ASSERT(primed); + // Leave this stream unacknowledged so its old tag data must remain readable by the new owner. + Future pending; + co_await timeoutError(startBlockedConsume(cx, firstId, streams[0].consumer, source, &pending), + operationTimeout); + + std::vector availableProxies{ proxies[0].id(), proxies[1].id() }; + ASSERT(co_await timeoutError(rebalanceNativeCdcProxyAssignments(cx, availableProxies), operationTimeout)); + const CDCProxyInterface moved = + co_await timeoutError(waitForAssignedProxy(cx, firstId, source.id()), operationTimeout); + ASSERT_EQ(moved.id(), target.id()); + const ClientDBInfo& published = cx->clientInfo->get(); + ASSERT_EQ(published.cdcProxies, proxies); + ASSERT_EQ(published.streamToCDCProxyId.at(firstId), target.id()); + ASSERT_EQ(published.streamToCDCProxyId.at(sharedTagId), target.id()); + ASSERT_EQ(published.streamToCDCProxyId.at(otherTagId), source.id()); + const NativeCdcStatus afterMove = co_await timeoutError(getNativeCdcStatus(cx), operationTimeout); + ASSERT(afterMove.metadataComplete); + for (CDCStreamId id : { firstId, sharedTagId }) { + const auto& stream = findStream(afterMove, id); + ASSERT(stream.owner.present()); + ASSERT_EQ(stream.owner.get(), target.id()); + ASSERT(stream.ownerPublished); + } + const auto& untouched = findStream(afterMove, otherTagId); + ASSERT(untouched.owner.present()); + ASSERT_EQ(untouched.owner.get(), source.id()); + ASSERT(untouched.ownerPublished); + co_await checkTagOwner(cx, tag, firstId); + const auto blockingTag = std::find_if( + afterMove.tags.begin(), afterMove.tags.end(), [&](auto const& state) { return state.tag == tag; }); + ASSERT(blockingTag != afterMove.tags.end()); + ASSERT_LE(blockingTag->safePopVersion, beforeVersion); + ASSERT(std::find(blockingTag->blockingStreams.begin(), blockingTag->blockingStreams.end(), sharedTagId) != + blockingTag->blockingStreams.end()); + ASSERT_EQ(afterMove.proxies.size(), 2); + for (const auto& proxy : afterMove.proxies) { + ASSERT(proxy.sample.present()); + } + ASSERT(!(co_await timeoutError(rebalanceNativeCdcProxyAssignments(cx, availableProxies), operationTimeout))); + + const ErrorOr staleAck = + co_await timeoutError(source.ack.tryGetReply(CDCAckRequest(firstId, beforeVersion)), operationTimeout); + ASSERT(!staleAck.present()); + ASSERT_EQ(staleAck.getError().code(), error_code_wrong_shard_server); + bool replayed = false; + const double replayDeadline = now() + operationTimeout; + do { + ASSERT_LT(now(), replayDeadline); + CDCConsumeReply replay = co_await timeoutError(streams[2].consumer->consume(), replayDeadline - now()); + replayed |= containsValue(replay, beforeVersion, key, beforeMove); + } while (streams[2].consumer->position().lastConsumedVersion < beforeVersion); + ASSERT(replayed); + co_await timeoutError(streams[2].consumer->acknowledge(), operationTimeout); + const NativeCdcStatus afterAck = co_await timeoutError(getNativeCdcStatus(cx), operationTimeout); + const auto advancedTag = std::find_if( + afterAck.tags.begin(), afterAck.tags.end(), [&](auto const& state) { return state.tag == tag; }); + ASSERT(advancedTag != afterAck.tags.end()); + ASSERT_GT(advancedTag->safePopVersion, beforeVersion); + + const Value afterMoveValue = "native-cdc-rebalance-after"_sr; + const Version afterVersion = co_await writeValue(cx, key, afterMoveValue); + bool pendingObserved = false; + const double deliveryDeadline = now() + operationTimeout; + while (!pendingObserved) { + CDCConsumeReply reply = co_await timeoutError(pending, operationTimeout); + pendingObserved = containsValue(reply, afterVersion, key, afterMoveValue); + co_await timeoutError(streams[0].consumer->acknowledge(), operationTimeout); + ASSERT_LT(now(), deliveryDeadline); + if (!pendingObserved) { + pending = streams[0].consumer->consume(); + } + } + co_await consumeThroughValue(streams[1].consumer, afterVersion, key, afterMoveValue); + co_await consumeThroughValue(streams[2].consumer, afterVersion, key, afterMoveValue); + for (const auto& stream : streams) { + co_await timeoutError(removeNativeCdcStreamClient(cx, stream.name), operationTimeout); + } + streams.clear(); + co_await timeoutError(waitForRetiredTagCleanup(cx), operationTimeout); + CODE_PROBE(true, "Native CDC rebalances a shared tag across live proxies without losing delivery"); + } + + Future validateAutomaticProxyRebalance(Database cx) { + ASSERT_EQ(streams.size(), 3); + const CDCStreamId firstId = streams[0].consumer->position().streamId; + const CDCStreamId otherTagId = streams[1].consumer->position().streamId; + const CDCStreamId sharedTagId = streams[2].consumer->position().streamId; + const auto proxies = cx->clientInfo->get().cdcProxies; + ASSERT_EQ(proxies.size(), 2); + const NativeCdcStatus initial = co_await timeoutError(getNativeCdcStatus(cx), operationTimeout); + ASSERT(initial.metadataComplete); + ASSERT_EQ(initial.tagCount, 2); + ASSERT_EQ(initial.streams.size(), 3); + const auto findStream = [](NativeCdcStatus const& status, CDCStreamId id) -> NativeCdcStreamStatus const& { + const auto stream = std::find_if(status.streams.begin(), status.streams.end(), [&](auto const& candidate) { + return candidate.info.streamId == id; + }); + ASSERT(stream != status.streams.end()); + return *stream; + }; + const auto& first = findStream(initial, firstId); + const auto& otherTag = findStream(initial, otherTagId); + const auto& shared = findStream(initial, sharedTagId); + ASSERT_EQ(first.tags.size(), 1); + ASSERT_EQ(otherTag.tags.size(), 1); + ASSERT_EQ(shared.tags.size(), 1); + ASSERT_EQ(first.tags.front(), shared.tags.front()); + ASSERT_NE(first.tags.front(), otherTag.tags.front()); + for (const auto& stream : initial.streams) { + ASSERT(stream.owner.present()); + ASSERT(std::any_of( + proxies.begin(), proxies.end(), [&](auto const& proxy) { return proxy.id() == stream.owner.get(); })); + } + ASSERT_EQ(first.owner.get(), shared.owner.get()); + + UID groupOwner; + UID otherOwner; + const double deadline = now() + operationTimeout; + while (true) { + Future changed = cx->clientInfo->onChange(); + const ClientDBInfo& published = cx->clientInfo->get(); + ASSERT_EQ(published.cdcProxies, proxies); + const auto group = published.streamToCDCProxyId.find(firstId); + const auto sharedGroup = published.streamToCDCProxyId.find(sharedTagId); + const auto other = published.streamToCDCProxyId.find(otherTagId); + const auto isLiveProxy = [&](UID id) { + return std::any_of(proxies.begin(), proxies.end(), [&](auto const& proxy) { return proxy.id() == id; }); + }; + if (group != published.streamToCDCProxyId.end() && sharedGroup != published.streamToCDCProxyId.end() && + other != published.streamToCDCProxyId.end() && group->second == sharedGroup->second && + group->second != other->second && isLiveProxy(group->second) && isLiveProxy(other->second)) { + groupOwner = group->second; + otherOwner = other->second; + break; + } + ASSERT_LT(now(), deadline); + co_await timeoutError(changed, deadline - now()); + } + if (first.owner.get() == otherTag.owner.get()) { + ASSERT_NE(groupOwner, first.owner.get()); + ASSERT_EQ(otherOwner, otherTag.owner.get()); + } else { + ASSERT_EQ(groupOwner, first.owner.get()); + ASSERT_EQ(otherOwner, otherTag.owner.get()); + } + const NativeCdcStatus afterMove = co_await timeoutError(getNativeCdcStatus(cx), operationTimeout); + ASSERT(afterMove.metadataComplete); + ASSERT_EQ(afterMove.streams.size(), 3); + for (const auto& stream : afterMove.streams) { + ASSERT(stream.owner.present()); + ASSERT_EQ(stream.owner.get(), stream.info.streamId == otherTagId ? otherOwner : groupOwner); + ASSERT(stream.ownerPublished); + } + co_await checkTagOwner(cx, first.tags.front(), firstId); + co_await checkTagOwner(cx, otherTag.tags.front(), otherTagId); + ASSERT_EQ(afterMove.proxies.size(), 2); + for (const auto& proxy : afterMove.proxies) { + ASSERT(proxy.sample.present()); + } + + const Key key = keyForIndex(keyCount / 2); + const Value value = "native-cdc-automatic-rebalance"_sr; + const Version committed = co_await writeValue(cx, key, value); + for (const auto& stream : streams) { + co_await consumeThroughValue(stream.consumer, committed, key, value); + co_await timeoutError(removeNativeCdcStreamClient(cx, stream.name), operationTimeout); + } + streams.clear(); + co_await timeoutError(waitForRetiredTagCleanup(cx), operationTimeout); + CODE_PROBE(true, "Native CDC controller rebalances a whole tag without proxy replacement"); + } + Future checkTagOwner(Database cx, Tag tag, Optional expected) { Transaction tr(cx); while (true) { @@ -1866,6 +2108,14 @@ class NativeCdcEndToEndWorkload : public TestWorkload { } Future run(Database cx) { + if (testProxyRebalanceAutomatic) { + co_await timeoutError(validateAutomaticProxyRebalance(cx), operationTimeout); + co_return; + } + if (testProxyRebalance) { + co_await timeoutError(validateProxyRebalance(cx), operationTimeout); + co_return; + } if (testRetiredSharedTagSnapshot) { co_await validateRetiredSharedTagSnapshot(cx); co_return; @@ -1961,6 +2211,8 @@ class NativeCdcEndToEndWorkload : public TestWorkload { rounds = getOption(options, "rounds"_sr, 30); assignmentPublicationChecks = getOption(options, "assignmentPublicationChecks"_sr, 0); testProxyReplacement = getOption(options, "testProxyReplacement"_sr, false); + testProxyRebalance = getOption(options, "testProxyRebalance"_sr, false); + testProxyRebalanceAutomatic = getOption(options, "testProxyRebalanceAutomatic"_sr, false); testTagOwnership = getOption(options, "testTagOwnership"_sr, false); injectUndeliveredProxyHalt = getOption(options, "injectUndeliveredProxyHalt"_sr, false); testMemoryBound = getOption(options, "testMemoryBound"_sr, false); @@ -1984,6 +2236,8 @@ class NativeCdcEndToEndWorkload : public TestWorkload { ASSERT_GE(writesPerRound, 1); ASSERT_LE(writesPerRound, keyCount); ASSERT_GE(assignmentPublicationChecks, 0); + ASSERT(!(testProxyRebalance && testProxyRebalanceAutomatic)); + ASSERT(!(testProxyRebalance || testProxyRebalanceAutomatic) || initialStreamCount == 3); ASSERT(!injectUndeliveredProxyHalt || testProxyReplacement); ASSERT_GT(memoryTestValueBytes, 0); ASSERT_GE(retentionValidationDelay, 0.0); diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 4b203deb592..e91719088c0 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -207,6 +207,8 @@ if(WITH_PYTHON) add_fdb_test(TEST_FILES fast/RangeLocking.toml) add_fdb_test(TEST_FILES fast/NativeCdcEndToEnd.toml) add_fdb_test(TEST_FILES fast/NativeCdcAssignmentPublication.toml) + add_fdb_test(TEST_FILES fast/NativeCdcProxyRebalance.toml) + add_fdb_test(TEST_FILES fast/NativeCdcProxyRebalanceAutomatic.toml) add_fdb_test(TEST_FILES fast/NativeCdcSharedTag.toml) add_fdb_test(TEST_FILES fast/NativeCdcRetiredSharedTagSnapshot.toml) add_fdb_test(TEST_FILES fast/NativeCdcMemoryBound.toml) diff --git a/tests/fast/NativeCdcProxyRebalance.toml b/tests/fast/NativeCdcProxyRebalance.toml new file mode 100644 index 00000000000..74b0b019f80 --- /dev/null +++ b/tests/fast/NativeCdcProxyRebalance.toml @@ -0,0 +1,33 @@ +[configuration] +config = 'single commit_proxies=1 grv_proxies=2' +singleRegion = true +datacenters = 1 +machineCount = 12 +statelessProcessClassesPerDC = 3 +buggify = false +faultInjection = false + +[[knobs]] +enable_native_cdc = true +native_cdc_tag_count = 2 +# Invoke the production policy from the workload after all three registrations are published. +cdc_proxy_rebalance_enabled = false + +[[test]] +testTitle = 'NativeCdcProxyRebalance' +useDB = true +waitForQuiescenceEnd = false +runFailureWorkloads = false +connectionFailuresDisableDuration = 1000000 +timeout = 180 + + [[test.workload]] + testName = 'NativeCdcEndToEnd' + initialStreamCount = 3 + minStreamCount = 3 + maxStreamCount = 3 + keyCount = 4 + writesPerRound = 1 + rounds = 0 + testProxyRebalance = true + operationTimeout = 90.0 diff --git a/tests/fast/NativeCdcProxyRebalanceAutomatic.toml b/tests/fast/NativeCdcProxyRebalanceAutomatic.toml new file mode 100644 index 00000000000..b2162522ad8 --- /dev/null +++ b/tests/fast/NativeCdcProxyRebalanceAutomatic.toml @@ -0,0 +1,34 @@ +[configuration] +config = 'single commit_proxies=1 grv_proxies=2' +singleRegion = true +datacenters = 1 +machineCount = 12 +statelessProcessClassesPerDC = 3 +buggify = false +faultInjection = false + +[[knobs]] +enable_native_cdc = true +native_cdc_tag_count = 2 +cdc_proxy_rebalance_enabled = true +# Leave time for the three registrations to establish one shared-tag group before the controller scans. +cdc_proxy_rebalance_interval = 20.0 + +[[test]] +testTitle = 'NativeCdcProxyRebalanceAutomatic' +useDB = true +waitForQuiescenceEnd = false +runFailureWorkloads = false +connectionFailuresDisableDuration = 1000000 +timeout = 240 + + [[test.workload]] + testName = 'NativeCdcEndToEnd' + initialStreamCount = 3 + minStreamCount = 3 + maxStreamCount = 3 + keyCount = 4 + writesPerRound = 1 + rounds = 0 + testProxyRebalanceAutomatic = true + operationTimeout = 150.0 From 98a7d75369312730b9feb70528332318efcca023 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 6 Sep 2026 15:26:17 -0700 Subject: [PATCH 041/170] Move CDC proxy balancing policy into server code --- fdbclient/NativeCdc.cpp | 157 -------------- fdbclient/NativeCdcInternal.h | 8 - .../clustercontroller/ClusterController.cpp | 1 + .../NativeCdcProxyBalancer.cpp | 204 ++++++++++++++++++ .../NativeCdcProxyBalancer.h | 32 +++ fdbserver/workloads/CMakeLists.txt | 4 +- fdbserver/workloads/NativeCdcEndToEnd.cpp | 8 +- 7 files changed, 244 insertions(+), 170 deletions(-) create mode 100644 fdbserver/clustercontroller/NativeCdcProxyBalancer.cpp create mode 100644 fdbserver/clustercontroller/include/fdbserver/clustercontroller/NativeCdcProxyBalancer.h diff --git a/fdbclient/NativeCdc.cpp b/fdbclient/NativeCdc.cpp index 2f1154fec03..9bc9780c450 100644 --- a/fdbclient/NativeCdc.cpp +++ b/fdbclient/NativeCdc.cpp @@ -724,163 +724,6 @@ Future reassignNativeCdcStreams(Database cx, UID oldProxyId, UID newProxyI } } -Future rebalanceNativeCdcProxyAssignments(Database cx, - std::vector availableProxies, - std::function stillEligible) { - // The metadata is read and the whole tag group is moved in one transaction. Do not split a shared tag - // across owners or make a partial move when the scan approaches the transaction size/lifetime limits. - constexpr int maxStreams = 512; - constexpr int maxHistoryRows = 2048; - constexpr int maxStreamsPerMove = 64; - constexpr int maxRangeBytes = 1 << 20; - const int64_t timeoutMs = 5000; - const std::set available(availableProxies.begin(), availableProxies.end()); - if (available.size() < 2) { - co_return false; - } - - Transaction tr(cx); - int attempts = 0; - while (true) { - if (++attempts > 3) { - co_return false; - } - Error err; - try { - tr.setOption(FDBTransactionOptions::LOCK_AWARE); - tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); - tr.setOption(FDBTransactionOptions::TIMEOUT, - StringRef(reinterpret_cast(&timeoutMs), sizeof(timeoutMs))); - const UID publishedInfoId = cx->clientInfo->get().id; - if (!cx->clientInfo->get().nativeCdcEnabled || !stillEligible()) { - co_return false; - } - - RangeResult active = co_await tr.getRange(cdcStreamKeys, GetRangeLimits(maxStreams + 1, maxRangeBytes)); - RangeResult assignments = co_await tr.getRange(cdcProxyKeys, GetRangeLimits(maxStreams + 1, maxRangeBytes)); - RangeResult histories = - co_await tr.getRange(cdcTagHistoryKeys, GetRangeLimits(maxHistoryRows + 1, maxRangeBytes)); - if (active.more || assignments.more || histories.more || active.size() > maxStreams || - assignments.size() > maxStreams || histories.size() > maxHistoryRows) { - CODE_PROBE(true, "Native CDC proxy rebalancing skips oversized metadata"); - co_return false; - } - - std::set activeIds; - for (const auto& stream : active) { - activeIds.insert(decodeCDCStreamKey(stream.key)); - } - std::map ownerByStream; - for (const auto& assignment : assignments) { - const auto [streamId, owner] = decodeCDCProxyKey(assignment.key); - if (activeIds.contains(streamId) && !ownerByStream.emplace(streamId, owner).second) { - co_return false; - } - } - std::map currentTagByStream; - std::set pendingHistories; - for (const auto& history : histories) { - const CDCTagHistoryEntry entry = decodeCDCTagHistoryKey(history.key); - if (activeIds.contains(entry.streamId)) { - currentTagByStream[entry.streamId] = entry.tag; - if (!history.value.empty()) { - pendingHistories.insert(entry.streamId); - } - } - } - - std::map ownerLoads; - for (const UID& proxyId : available) { - ownerLoads.emplace(proxyId, 0); - } - std::map> membersByTag; - for (const CDCStreamId streamId : activeIds) { - auto owner = ownerByStream.find(streamId); - auto tag = currentTagByStream.find(streamId); - if (owner == ownerByStream.end() || tag == currentTagByStream.end() || - !ownerLoads.contains(owner->second)) { - // Let the cluster controller repair missing or stale ownership before balancing. - co_return false; - } - const auto published = cx->clientInfo->get().streamToCDCProxyId.find(streamId); - if (published == cx->clientInfo->get().streamToCDCProxyId.end() || published->second != owner->second) { - co_return false; - } - ++ownerLoads[owner->second]; - membersByTag[tag->second].push_back(streamId); - } - - Optional selectedTag; - Optional selectedSource; - Optional selectedTarget; - int bestImprovement = 0; - size_t bestGroupSize = 0; - for (const auto& [tag, members] : membersByTag) { - const UID source = ownerByStream.at(members.front()); - for (const CDCStreamId streamId : members) { - if (ownerByStream.at(streamId) != source) { - // Registration relies on every stream sharing a current tag having one owner. - co_return false; - } - } - if (members.size() > maxStreamsPerMove || - std::any_of(members.begin(), members.end(), [&](CDCStreamId id) { - return pendingHistories.contains(id); - })) { - continue; - } - for (const UID& target : available) { - if (source == target) { - continue; - } - const int difference = ownerLoads.at(source) - ownerLoads.at(target); - const int moved = 2 * static_cast(members.size()); - const int after = difference >= moved ? difference - moved : moved - difference; - const int improvement = difference - after; - if (improvement > bestImprovement || - (improvement == bestImprovement && improvement > 0 && members.size() > bestGroupSize)) { - bestImprovement = improvement; - bestGroupSize = members.size(); - selectedTag = tag; - selectedSource = source; - selectedTarget = target; - } - } - } - if (!selectedTag.present()) { - co_return false; - } - if (!stillEligible() || !cx->clientInfo->get().nativeCdcEnabled || - cx->clientInfo->get().id != publishedInfoId || - !containsNativeCdcProxy(cx->clientInfo->get(), selectedSource.get()) || - !containsNativeCdcProxy(cx->clientInfo->get(), selectedTarget.get())) { - co_return false; - } - - for (const CDCStreamId streamId : membersByTag.at(selectedTag.get())) { - tr.clear(cdcProxyKeyFor(streamId, selectedSource.get())); - tr.set(cdcProxyKeyFor(streamId, selectedTarget.get()), Value()); - } - signalNativeCdcProxyAssignmentChange(&tr); - co_await tr.commit(); - CODE_PROBE(true, "Native CDC rebalances an entire shared tag across live proxies"); - TraceEvent("CDCProxyTagRebalanced") - .detail("Tag", selectedTag.get().toString()) - .detail("OldCDCProxyID", selectedSource.get()) - .detail("NewCDCProxyID", selectedTarget.get()) - .detail("StreamCount", bestGroupSize); - co_return true; - } catch (Error& e) { - // An ambiguous commit may already have moved a group. Reconcile on the next controller pass. - if (e.code() == error_code_commit_unknown_result) { - throw; - } - err = e; - } - co_await tr.onError(err); - } -} - Future acknowledgeNativeCdcStream(Database cx, CDCStreamId streamId, Version consumedThrough, diff --git a/fdbclient/NativeCdcInternal.h b/fdbclient/NativeCdcInternal.h index 8ddb5fbd669..ae52ee32fe1 100644 --- a/fdbclient/NativeCdcInternal.h +++ b/fdbclient/NativeCdcInternal.h @@ -22,8 +22,6 @@ #define FDBCLIENT_NATIVECDCINTERNAL_H #pragma once -#include - #include "fdbclient/NativeCdc.h" // Durable metadata operations used by CDC server roles. Registration is @@ -35,12 +33,6 @@ Future removeNativeCdcStream(Database cx, Key name, CDCStreamId streamId, Future> listNativeCdcStreams(Database cx); // Atomically moves any streams assigned to a failed proxy to its replacement. Future reassignNativeCdcStreams(Database cx, UID oldProxyId, UID newProxyId); -// Moves at most one complete current-tag group between live proxies when doing so reduces stream-count skew. -// Returns false if the metadata is incomplete, too large to scan safely, or already balanced. -Future rebalanceNativeCdcProxyAssignments( - Database cx, - std::vector availableProxies, - std::function stillEligible = [] { return true; }); // Persists the exclusive unpopped watermark after consuming through a version. // knownAvailableThrough permits a consumer to acknowledge log data it has // already received before that version is visible at a transaction read version. diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index 1ed6380711e..ce9e0960034 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -45,6 +45,7 @@ #include "fdbserver/core/CoordinatedState.h" #include "fdbserver/core/CoordinationInterface.h" // copy constructors for ServerCoordinators class #include "fdbserver/clustercontroller/ClusterController.h" +#include "fdbserver/clustercontroller/NativeCdcProxyBalancer.h" #include "ClusterController.h" #include "ClusterRecovery.h" #include "fdbserver/core/DataDistributorInterface.h" diff --git a/fdbserver/clustercontroller/NativeCdcProxyBalancer.cpp b/fdbserver/clustercontroller/NativeCdcProxyBalancer.cpp new file mode 100644 index 00000000000..9807ce2fe0d --- /dev/null +++ b/fdbserver/clustercontroller/NativeCdcProxyBalancer.cpp @@ -0,0 +1,204 @@ +/* + * NativeCdcProxyBalancer.cpp + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include +#include +#include +#include + +#include "fdbclient/DatabaseContext.h" +#include "fdbclient/SystemData.h" +#include "fdbserver/clustercontroller/NativeCdcProxyBalancer.h" +#include "flow/CodeProbe.h" +#include "flow/DeterministicRandom.h" +#include "flow/Trace.h" + +namespace { + +bool containsNativeCdcProxy(ClientDBInfo const& clientInfo, UID proxyId) { + return std::any_of(clientInfo.cdcProxies.begin(), + clientInfo.cdcProxies.end(), + [proxyId](CDCProxyInterface const& proxy) { return proxy.id() == proxyId; }); +} + +void signalNativeCdcProxyAssignmentChange(Transaction* tr) { + tr->set(cdcProxyAssignmentChangeKey, + BinaryWriter::toValue(deterministicRandom()->randomUniqueID(), + IncludeVersion(ProtocolVersion::withNativeCdc()))); +} + +} // namespace + +Future rebalanceNativeCdcProxyAssignments(Database cx, + std::vector availableProxies, + std::function stillEligible) { + // The metadata is read and the whole tag group is moved in one transaction. Do not split a shared tag + // across owners or make a partial move when the scan approaches the transaction size/lifetime limits. + constexpr int maxStreams = 512; + constexpr int maxHistoryRows = 2048; + constexpr int maxStreamsPerMove = 64; + constexpr int maxRangeBytes = 1 << 20; + const int64_t timeoutMs = 5000; + const std::set available(availableProxies.begin(), availableProxies.end()); + if (available.size() < 2) { + co_return false; + } + + Transaction tr(cx); + int attempts = 0; + while (true) { + if (++attempts > 3) { + co_return false; + } + Error err; + try { + tr.setOption(FDBTransactionOptions::LOCK_AWARE); + tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); + tr.setOption(FDBTransactionOptions::TIMEOUT, + StringRef(reinterpret_cast(&timeoutMs), sizeof(timeoutMs))); + const UID publishedInfoId = cx->clientInfo->get().id; + if (!cx->clientInfo->get().nativeCdcEnabled || !stillEligible()) { + co_return false; + } + + RangeResult active = co_await tr.getRange(cdcStreamKeys, GetRangeLimits(maxStreams + 1, maxRangeBytes)); + RangeResult assignments = co_await tr.getRange(cdcProxyKeys, GetRangeLimits(maxStreams + 1, maxRangeBytes)); + RangeResult histories = + co_await tr.getRange(cdcTagHistoryKeys, GetRangeLimits(maxHistoryRows + 1, maxRangeBytes)); + if (active.more || assignments.more || histories.more || active.size() > maxStreams || + assignments.size() > maxStreams || histories.size() > maxHistoryRows) { + CODE_PROBE(true, "Native CDC proxy rebalancing skips oversized metadata"); + co_return false; + } + + std::set activeIds; + for (const auto& stream : active) { + activeIds.insert(decodeCDCStreamKey(stream.key)); + } + std::map ownerByStream; + for (const auto& assignment : assignments) { + const auto [streamId, owner] = decodeCDCProxyKey(assignment.key); + if (activeIds.contains(streamId) && !ownerByStream.emplace(streamId, owner).second) { + co_return false; + } + } + std::map currentTagByStream; + std::set pendingHistories; + for (const auto& history : histories) { + const CDCTagHistoryEntry entry = decodeCDCTagHistoryKey(history.key); + if (activeIds.contains(entry.streamId)) { + currentTagByStream[entry.streamId] = entry.tag; + if (!history.value.empty()) { + pendingHistories.insert(entry.streamId); + } + } + } + + std::map ownerLoads; + for (const UID& proxyId : available) { + ownerLoads.emplace(proxyId, 0); + } + std::map> membersByTag; + for (const CDCStreamId streamId : activeIds) { + auto owner = ownerByStream.find(streamId); + auto tag = currentTagByStream.find(streamId); + if (owner == ownerByStream.end() || tag == currentTagByStream.end() || + !ownerLoads.contains(owner->second)) { + // Let the cluster controller repair missing or stale ownership before balancing. + co_return false; + } + const auto published = cx->clientInfo->get().streamToCDCProxyId.find(streamId); + if (published == cx->clientInfo->get().streamToCDCProxyId.end() || published->second != owner->second) { + co_return false; + } + ++ownerLoads[owner->second]; + membersByTag[tag->second].push_back(streamId); + } + + Optional selectedTag; + Optional selectedSource; + Optional selectedTarget; + int bestImprovement = 0; + size_t bestGroupSize = 0; + for (const auto& [tag, members] : membersByTag) { + const UID source = ownerByStream.at(members.front()); + for (const CDCStreamId streamId : members) { + if (ownerByStream.at(streamId) != source) { + // Registration relies on every stream sharing a current tag having one owner. + co_return false; + } + } + if (members.size() > maxStreamsPerMove || + std::any_of(members.begin(), members.end(), [&](CDCStreamId id) { + return pendingHistories.contains(id); + })) { + continue; + } + for (const UID& target : available) { + if (source == target) { + continue; + } + const int difference = ownerLoads.at(source) - ownerLoads.at(target); + const int moved = 2 * static_cast(members.size()); + const int after = difference >= moved ? difference - moved : moved - difference; + const int improvement = difference - after; + if (improvement > bestImprovement || + (improvement == bestImprovement && improvement > 0 && members.size() > bestGroupSize)) { + bestImprovement = improvement; + bestGroupSize = members.size(); + selectedTag = tag; + selectedSource = source; + selectedTarget = target; + } + } + } + if (!selectedTag.present()) { + co_return false; + } + if (!stillEligible() || !cx->clientInfo->get().nativeCdcEnabled || + cx->clientInfo->get().id != publishedInfoId || + !containsNativeCdcProxy(cx->clientInfo->get(), selectedSource.get()) || + !containsNativeCdcProxy(cx->clientInfo->get(), selectedTarget.get())) { + co_return false; + } + + for (const CDCStreamId streamId : membersByTag.at(selectedTag.get())) { + tr.clear(cdcProxyKeyFor(streamId, selectedSource.get())); + tr.set(cdcProxyKeyFor(streamId, selectedTarget.get()), Value()); + } + signalNativeCdcProxyAssignmentChange(&tr); + co_await tr.commit(); + CODE_PROBE(true, "Native CDC rebalances an entire shared tag across live proxies"); + TraceEvent("CDCProxyTagRebalanced") + .detail("Tag", selectedTag.get().toString()) + .detail("OldCDCProxyID", selectedSource.get()) + .detail("NewCDCProxyID", selectedTarget.get()) + .detail("StreamCount", bestGroupSize); + co_return true; + } catch (Error& e) { + // An ambiguous commit may already have moved a group. Reconcile on the next controller pass. + if (e.code() == error_code_commit_unknown_result) { + throw; + } + err = e; + } + co_await tr.onError(err); + } +} diff --git a/fdbserver/clustercontroller/include/fdbserver/clustercontroller/NativeCdcProxyBalancer.h b/fdbserver/clustercontroller/include/fdbserver/clustercontroller/NativeCdcProxyBalancer.h new file mode 100644 index 00000000000..d15441f4d16 --- /dev/null +++ b/fdbserver/clustercontroller/include/fdbserver/clustercontroller/NativeCdcProxyBalancer.h @@ -0,0 +1,32 @@ +/* + * NativeCdcProxyBalancer.h + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#pragma once + +#include +#include + +#include "fdbclient/NativeCdc.h" + +// Moves at most one complete current-tag group between live proxies when doing so reduces stream-count skew. +// Returns false if the metadata is incomplete, too large to scan safely, or already balanced. +Future rebalanceNativeCdcProxyAssignments(Database cx, + std::vector availableProxies, + std::function stillEligible); diff --git a/fdbserver/workloads/CMakeLists.txt b/fdbserver/workloads/CMakeLists.txt index aa7b813577d..ed712f84ba6 100644 --- a/fdbserver/workloads/CMakeLists.txt +++ b/fdbserver/workloads/CMakeLists.txt @@ -10,9 +10,9 @@ target_sources(fdbserver_workloads_test PRIVATE ../MemoryTrackerTest.cpp ../Glob configure_fdbserver_common_includes(fdbserver_workloads) target_include_directories(fdbserver_workloads PRIVATE - ${CMAKE_CURRENT_SOURCE_DIR} - ${CMAKE_SOURCE_DIR}/fdbclient) + ${CMAKE_CURRENT_SOURCE_DIR}) target_link_libraries(fdbserver_workloads PRIVATE + fdbserver_clustercontroller fdbserver_consistencyscan fdbserver_core fdbserver_kvstore diff --git a/fdbserver/workloads/NativeCdcEndToEnd.cpp b/fdbserver/workloads/NativeCdcEndToEnd.cpp index 68dc869cb89..ba986c1b57b 100644 --- a/fdbserver/workloads/NativeCdcEndToEnd.cpp +++ b/fdbserver/workloads/NativeCdcEndToEnd.cpp @@ -30,7 +30,7 @@ #include "fdbclient/DatabaseContext.h" #include "fdbclient/NativeCdc.h" #include "fdbclient/SystemData.h" -#include "NativeCdcInternal.h" +#include "fdbserver/clustercontroller/NativeCdcProxyBalancer.h" #include "fdbserver/core/RecoveryState.h" #include "fdbserver/core/ServerDBInfo.h" #include "fdbserver/tester/workloads.h" @@ -644,7 +644,8 @@ class NativeCdcEndToEndWorkload : public TestWorkload { operationTimeout); std::vector availableProxies{ proxies[0].id(), proxies[1].id() }; - ASSERT(co_await timeoutError(rebalanceNativeCdcProxyAssignments(cx, availableProxies), operationTimeout)); + ASSERT(co_await timeoutError(rebalanceNativeCdcProxyAssignments(cx, availableProxies, [] { return true; }), + operationTimeout)); const CDCProxyInterface moved = co_await timeoutError(waitForAssignedProxy(cx, firstId, source.id()), operationTimeout); ASSERT_EQ(moved.id(), target.id()); @@ -676,7 +677,8 @@ class NativeCdcEndToEndWorkload : public TestWorkload { for (const auto& proxy : afterMove.proxies) { ASSERT(proxy.sample.present()); } - ASSERT(!(co_await timeoutError(rebalanceNativeCdcProxyAssignments(cx, availableProxies), operationTimeout))); + ASSERT(!(co_await timeoutError(rebalanceNativeCdcProxyAssignments(cx, availableProxies, [] { return true; }), + operationTimeout))); const ErrorOr staleAck = co_await timeoutError(source.ack.tryGetReply(CDCAckRequest(firstId, beforeVersion)), operationTimeout); From b168e247e771784b0f55d53165508fd79e05c1df Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 6 Sep 2026 19:55:11 -0700 Subject: [PATCH 042/170] Wait for CDC assignment publication in rebalance test --- fdbserver/workloads/NativeCdcEndToEnd.cpp | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/fdbserver/workloads/NativeCdcEndToEnd.cpp b/fdbserver/workloads/NativeCdcEndToEnd.cpp index ba986c1b57b..18295b33bc9 100644 --- a/fdbserver/workloads/NativeCdcEndToEnd.cpp +++ b/fdbserver/workloads/NativeCdcEndToEnd.cpp @@ -584,6 +584,10 @@ class NativeCdcEndToEndWorkload : public TestWorkload { const CDCStreamId sharedTagId = streams[2].consumer->position().streamId; const auto proxies = cx->clientInfo->get().cdcProxies; ASSERT_EQ(proxies.size(), 2); + // Registration commits durable ownership before the controller publishes each assignment. + for (const auto& stream : streams) { + co_await timeoutError(waitForAssignedProxy(cx, stream.consumer->position().streamId), operationTimeout); + } const NativeCdcStatus initial = co_await timeoutError(getNativeCdcStatus(cx), operationTimeout); ASSERT(initial.metadataComplete); ASSERT_EQ(initial.tagCount, 2); From 5c39b3774f112d9b0de91a50b122eda8ed8d405c Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Wed, 9 Sep 2026 08:57:19 -0700 Subject: [PATCH 043/170] Retarget ratekeeper monitoring when the published role changes --- .../clustercontroller/ClusterController.cpp | 69 ++++++++++++++++++- 1 file changed, 68 insertions(+), 1 deletion(-) diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index ea93c5a6c90..732365836e0 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -3063,7 +3063,8 @@ Future monitorRatekeeper(ClusterControllerData* self) { const UID monitoredRatekeeperID = self->db.serverInfo->get().ratekeeper.get().id(); auto res = co_await race(waitFailureClient(self->db.serverInfo->get().ratekeeper.get().waitFailure, SERVER_KNOBS->RATEKEEPER_FAILURE_TIME), - self->recruitRatekeeper.onChange()); + self->recruitRatekeeper.onChange(), + self->db.serverInfo->onChange()); if (res.index() == 0) { const auto& ratekeeper = self->db.serverInfo->get().ratekeeper; if (!ratekeeper.present() || ratekeeper.get().id() != monitoredRatekeeperID) { @@ -3893,6 +3894,72 @@ TEST_CASE("/fdbserver/clustercontroller/replacedRatekeeperSurvivesPreviousFailur monitor.cancel(); } +TEST_CASE("/fdbserver/clustercontroller/ratekeeperReplacementRefreshesMonitors") { + LocalityData controllerLocality; + controllerLocality.set(LocalityData::keyDcId, "primary"_sr); + ClusterControllerData data(ClusterControllerFullInterface(), + controllerLocality, + ServerCoordinators(Reference( + new ClusterConnectionMemoryRecord(ClusterConnectionString()))), + makeReference>>()); + WorkerInterface oldWorker = addSingletonTestWorker(data, "old-ratekeeper"_sr, "primary"_sr); + WorkerInterface newWorker = addSingletonTestWorker(data, "new-ratekeeper"_sr, "primary"_sr); + RatekeeperInterface oldRatekeeper(oldWorker.locality, UID(1, 1)); + RatekeeperInterface newRatekeeper(newWorker.locality, UID(1, 2)); + FutureStream> oldFailures = oldRatekeeper.waitFailure.getFuture(); + FutureStream> newFailures = newRatekeeper.waitFailure.getFuture(); + FutureStream oldHalts = oldRatekeeper.haltRatekeeper.getFuture(); + FutureStream newHalts = newRatekeeper.haltRatekeeper.getFuture(); + FutureStream oldEvents = oldWorker.eventLogRequest.getFuture(); + FutureStream newEvents = newWorker.eventLogRequest.getFuture(); + + auto serverInfo = data.db.serverInfo->get(); + serverInfo.recoveryState = RecoveryState::ACCEPTING_COMMITS; + serverInfo.id = UID(3, 1); + data.db.serverInfo->set(serverInfo); + data.db.setRatekeeper(oldRatekeeper); + Future monitor = monitorRatekeeper(&data); + auto oldWaitOrTimeout = co_await race(oldFailures, delay(2.0)); + ASSERT_EQ(oldWaitOrTimeout.index(), 0); + ReplyPromise oldFailure = std::get<0>(std::move(oldWaitOrTimeout)); + + processRegisteredSingletons(&data, newWorker, {}, newRatekeeper, {}); + ASSERT(data.db.serverInfo->get().ratekeeper.get().id() == newRatekeeper.id()); + auto oldHaltOrTimeout = co_await race(oldHalts, delay(2.0)); + ASSERT_EQ(oldHaltOrTimeout.index(), 0); + HaltRatekeeperRequest oldHalt = std::get<0>(std::move(oldHaltOrTimeout)); + auto newWaitOrTimeout = co_await race(newFailures, delay(2.0)); + ASSERT_EQ(newWaitOrTimeout.index(), 0); + ReplyPromise newFailure = std::get<0>(std::move(newWaitOrTimeout)); + + auto latestEvents = data.clusterHealthWorkerEventProvider->getLatestRatekeeperEvents("RkUpdate"); + auto eventOrTimeout = co_await race(oldEvents, newEvents, delay(2.0)); + ASSERT_EQ(eventOrTimeout.index(), 1); + EventLogRequest request = std::get<1>(std::move(eventOrTimeout)); + ASSERT(request.eventName == "RkUpdate"_sr); + TraceEventFields fields; + fields.addField("ReleasedTPS", "100"); + fields.addField("TPSLimit", "125"); + request.reply.send(fields); + auto events = co_await latestEvents; + ASSERT(events.present()); + ASSERT_EQ(events.get().first.size(), 1); + ASSERT(events.get().second.empty()); + ASSERT_EQ(events.get().first.begin()->second.getDouble("TPSLimit"), 125.0); + + newFailure.sendError(connection_failed()); + auto newHaltOrTimeout = co_await race(newHalts, delay(2.0)); + ASSERT_EQ(newHaltOrTimeout.index(), 0); + HaltRatekeeperRequest newHalt = std::get<0>(std::move(newHaltOrTimeout)); + newHalt.reply.send(Void()); + ASSERT(!data.db.serverInfo->get().ratekeeper.present()); + auto clearedEvents = co_await data.clusterHealthWorkerEventProvider->getLatestRatekeeperEvents("RkUpdate"); + ASSERT(!clearedEvents.present()); + oldHalt.reply.send(Void()); + oldFailure.sendError(connection_failed()); + monitor.cancel(); +} + TEST_CASE("/fdbserver/clustercontroller/deferCrossDatacenterSingletonHaltsUntilRecovery") { LocalityData controllerLocality; controllerLocality.set(LocalityData::keyDcId, "new-primary"_sr); From 53d08ebcd81cfc393c7618f9140a88f2977fac79 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sat, 12 Sep 2026 16:43:48 -0700 Subject: [PATCH 044/170] Fix key selection and timing for immediate async Mako operations --- bindings/c/CMakeLists.txt | 4 + bindings/c/test/mako/async.cpp | 8 +- bindings/c/test/mako/async_test.cpp | 200 ++++++++++++++++++++++++++++ 3 files changed, 208 insertions(+), 4 deletions(-) create mode 100644 bindings/c/test/mako/async_test.cpp diff --git a/bindings/c/CMakeLists.txt b/bindings/c/CMakeLists.txt index e5b6a8ec0fb..bdfbef36512 100644 --- a/bindings/c/CMakeLists.txt +++ b/bindings/c/CMakeLists.txt @@ -180,6 +180,7 @@ if(NOT WIN32) add_library(fdb_c_client_memory_test OBJECT test/client_memory_test.cpp test/unit/fdb_api.cpp test/unit/fdb_api.hpp) if(BUILD_MAKO) add_library(mako OBJECT ${MAKO_SRCS}) + add_library(mako_async_test OBJECT test/mako/async_test.cpp test/mako/async.cpp) endif() add_library(fdb_c_setup_tests OBJECT test/unit/setup_tests.cpp) add_library(fdb_c_unit_tests_version_510 OBJECT ${UNIT_TEST_VERSION_510_SRCS}) @@ -193,6 +194,7 @@ if(NOT WIN32) add_executable(fdb_c_client_memory_test test/client_memory_test.cpp test/unit/fdb_api.cpp test/unit/fdb_api.hpp) if(BUILD_MAKO) add_executable(mako ${MAKO_SRCS}) + add_executable(mako_async_test test/mako/async_test.cpp test/mako/async.cpp) endif() add_executable(fdb_c_setup_tests test/unit/setup_tests.cpp) add_executable(fdb_c_unit_tests) @@ -235,7 +237,9 @@ if(NOT WIN32) # do not set RPATH for mako set_property(TARGET mako PROPERTY SKIP_BUILD_RPATH TRUE) target_link_libraries(mako PRIVATE fdb_c fdbclient fmt::fmt Threads::Threads fdb_cpp boost_target rapidjson) + target_link_libraries(mako_async_test PRIVATE fdb_c fdbclient fmt::fmt Threads::Threads fdb_cpp boost_target rapidjson doctest) if(NOT OPEN_FOR_IDE) + add_test(NAME mako_async_test COMMAND $) strip_debug_symbols(mako) if(NOT GENERATE_DEBUG_PACKAGES) add_custom_target(prepare_mako_install ALL DEPENDS strip_only_mako) diff --git a/bindings/c/test/mako/async.cpp b/bindings/c/test/mako/async.cpp index 25bd8640254..21682e87b94 100644 --- a/bindings/c/test/mako/async.cpp +++ b/bindings/c/test/mako/async.cpp @@ -101,15 +101,15 @@ void ResumableStateForRunWorkload::postNextTick() { void ResumableStateForRunWorkload::runOneTick() { assert(iter != OpEnd); + auto f = Future{}; + // to minimize context switch overhead, repeat immediately completed ops + // in a loop, not an async continuation. +repeat_immediate_steps: if (iter.step == 0 /* first step */) prepareKeys(iter.op, key1, key2, args); watch_step.start(); if (iter.step == 0) watch_op = Stopwatch(watch_step.getStart()); - auto f = Future{}; - // to minimize context switch overhead, repeat immediately completed ops - // in a loop, not an async continuation. -repeat_immediate_steps: f = opTable[iter.op].stepFunction(iter.step)(tx, args, key1, key2, val); if (!f) { // immediately completed client-side ops: e.g. set, setrange, clear, clearrange, ... diff --git a/bindings/c/test/mako/async_test.cpp b/bindings/c/test/mako/async_test.cpp new file mode 100644 index 00000000000..f1e3e7e043f --- /dev/null +++ b/bindings/c/test/mako/async_test.cpp @@ -0,0 +1,200 @@ +/* + * async_test.cpp + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#define DOCTEST_CONFIG_IMPLEMENT_WITH_MAIN +#include "doctest.h" + +#include "async.hpp" +#include "operations.hpp" +#include +#include + +thread_local mako::Logger logr{ mako::MainProcess{}, mako::VERBOSE_NONE }; + +namespace mako { + +// The driver tests do not link the benchmark's main or require a database. +Arguments::Arguments() + : rows(100), row_digits(3), sampling(1), key_length(16), value_length(16), zipf(0), commit_get(0), txnspec{}, + prefixpadding(0), transaction_timeout_db(0), transaction_timeout_tx(0), max_grv_queue_delay_ms(-1) {} + +bool Arguments::isAnyTimeoutEnabled() const { + return transaction_timeout_tx > 0 || transaction_timeout_db > 0; +} + +namespace { + +struct StopTick {}; +const auto staleStart = timepoint_t{} - std::chrono::seconds(1); + +class AsyncWorkloadTest { + static thread_local AsyncWorkloadTest* active; + Arguments args; + WorkflowStatistics stats; + boost::asio::io_context io; + std::atomic stopcount{ 0 }; + std::atomic signal{ SIGNAL_GREEN }; + RunWorkloadStateHandle state; + std::function observer; + +public: + AsyncWorkloadTest(WorkloadSpec spec, std::function observer) + : observer(std::move(observer)) { + assert(active == nullptr); + args.txnspec = spec; + state = std::make_shared( + logr, fdb::Database{}, fdb::Transaction{}, io, args, stats, stopcount, signal, -1, getOpBegin(args)); + active = this; + } + ~AsyncWorkloadTest() { active = nullptr; } + AsyncWorkloadTest(const AsyncWorkloadTest&) = delete; + AsyncWorkloadTest& operator=(const AsyncWorkloadTest&) = delete; + + ResumableStateForRunWorkload& workload() { return *state; } + WorkflowStatistics const& statistics() const { return stats; } + void runTick() { state->runOneTick(); } + void pollTick() { io.poll_one(); } + + static fdb::Future step(fdb::Transaction&, Arguments const&, fdb::ByteString&, fdb::ByteString&, fdb::ByteString&) { + active->observer(*active->state); + return {}; + } +}; + +thread_local AsyncWorkloadTest* AsyncWorkloadTest::active = nullptr; + +void checkPreparedKey(fdb::ByteString const& key) { + CHECK(std::equal(KEY_PREFIX.begin(), KEY_PREFIX.end(), key.begin())); +} + +void checkFreshTimers(ResumableStateForRunWorkload const& state) { + CHECK(state.watch_step.getStart() != staleStart); + CHECK(state.watch_op.getStart() == state.watch_step.getStart()); +} + +void poisonOperation(ResumableStateForRunWorkload& state) { + state.key1.assign(state.key1.size(), '!'); + state.key2.assign(state.key2.size(), '!'); + // Distinct starts make resets observable without waiting for the clock to advance. + state.watch_step = Stopwatch(staleStart); + state.watch_op = Stopwatch(staleStart); +} + +} // namespace + +// Synthetic immediate steps exercise the real driver while the observer stops before commit/reset. +const std::array opTable = [] { + std::array table{}; + const auto step = Step{ StepKind::IMM, &AsyncWorkloadTest::step }; + table[OP_OVERWRITE] = Operation{ "OVERWRITE", { step }, 1, true }; + table[OP_CLEAR] = Operation{ "CLEAR", { step }, 1, true }; + table[OP_SETCLEAR] = Operation{ "SETCLEAR", { step, step }, 2, true }; + table[OP_CLEARRANGE] = Operation{ "CLEARRANGE", { step }, 1, true }; + return table; +}(); + +TEST_CASE("mako async repeated immediate operations") { + WorkloadSpec spec{}; + spec.ops[OP_OVERWRITE][OP_COUNT] = 10; + spec.ops[OP_CLEAR][OP_COUNT] = 1; + int overwrites = 0; + AsyncWorkloadTest test(spec, [&](auto& state) { + checkPreparedKey(state.key1); + checkFreshTimers(state); + if (state.iter.op == OP_CLEAR) + throw StopTick{}; + CHECK(state.iter.count == overwrites++); + poisonOperation(state); + }); + CHECK_THROWS_AS(test.runTick(), StopTick); + CHECK(overwrites == 10); + CHECK(test.statistics().getOpCount(OP_OVERWRITE) == 10); + CHECK(test.statistics().getLatencySampleCount(OP_OVERWRITE) == 10); +} + +TEST_CASE("mako async immediate transition prepares range keys") { + WorkloadSpec spec{}; + spec.ops[OP_OVERWRITE][OP_COUNT] = 1; + spec.ops[OP_CLEARRANGE][OP_COUNT] = 1; + spec.ops[OP_CLEARRANGE][OP_RANGE] = 7; + AsyncWorkloadTest test(spec, [&](auto& state) { + checkPreparedKey(state.key1); + checkFreshTimers(state); + if (state.iter.op == OP_CLEARRANGE) { + checkPreparedKey(state.key2); + throw StopTick{}; + } + poisonOperation(state); + }); + CHECK_THROWS_AS(test.runTick(), StopTick); + CHECK(test.statistics().getOpCount(OP_OVERWRITE) == 1); +} + +TEST_CASE("mako async preserves keys and operation timer between steps") { + WorkloadSpec spec{}; + spec.ops[OP_SETCLEAR][OP_COUNT] = 2; + spec.ops[OP_CLEARRANGE][OP_COUNT] = 1; + AsyncWorkloadTest test(spec, [&](auto& state) { + CHECK(state.watch_step.getStart() != staleStart); + if (state.iter.step == 0) { + checkPreparedKey(state.key1); + checkFreshTimers(state); + } else { + CHECK(state.key1 == fdb::ByteString(state.key1.size(), '!')); + CHECK(state.key2 == fdb::ByteString(state.key2.size(), '!')); + CHECK(state.watch_op.getStart() == staleStart); + } + if (state.iter.op == OP_CLEARRANGE) + throw StopTick{}; + poisonOperation(state); + }); + SUBCASE("steps complete in the same tick") {} + SUBCASE("resume at a later step") { + test.workload().iter.step = 1; + poisonOperation(test.workload()); + } + CHECK_THROWS_AS(test.runTick(), StopTick); + CHECK(test.statistics().getOpCount(OP_SETCLEAR) == 2); +} + +TEST_CASE("mako async retry restarts operation preparation") { + WorkloadSpec spec{}; + spec.ops[OP_OVERWRITE][OP_COUNT] = 2; + spec.ops[OP_CLEAR][OP_COUNT] = 1; + int overwrites = 0; + AsyncWorkloadTest test(spec, [&](auto& state) { + checkPreparedKey(state.key1); + checkFreshTimers(state); + if (state.iter.op == OP_CLEAR) + throw StopTick{}; + CHECK(state.iter.count == overwrites++); + poisonOperation(state); + }); + test.workload().iter.count = 1; + test.workload().needs_commit = true; + poisonOperation(test.workload()); + test.workload().onIterationEnd(FutureRC::RETRY); + CHECK(test.workload().total_xacts == 0); + CHECK_FALSE(test.workload().needs_commit); + CHECK_THROWS_AS(test.pollTick(), StopTick); + CHECK(overwrites == 2); +} + +} // namespace mako From d1c109bf08548c7b409b741692adff4e9d0eb96b Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sat, 12 Sep 2026 17:05:02 -0700 Subject: [PATCH 045/170] Remove async Mako regression test target --- bindings/c/CMakeLists.txt | 4 - bindings/c/test/mako/async_test.cpp | 200 ---------------------------- 2 files changed, 204 deletions(-) delete mode 100644 bindings/c/test/mako/async_test.cpp diff --git a/bindings/c/CMakeLists.txt b/bindings/c/CMakeLists.txt index bdfbef36512..e5b6a8ec0fb 100644 --- a/bindings/c/CMakeLists.txt +++ b/bindings/c/CMakeLists.txt @@ -180,7 +180,6 @@ if(NOT WIN32) add_library(fdb_c_client_memory_test OBJECT test/client_memory_test.cpp test/unit/fdb_api.cpp test/unit/fdb_api.hpp) if(BUILD_MAKO) add_library(mako OBJECT ${MAKO_SRCS}) - add_library(mako_async_test OBJECT test/mako/async_test.cpp test/mako/async.cpp) endif() add_library(fdb_c_setup_tests OBJECT test/unit/setup_tests.cpp) add_library(fdb_c_unit_tests_version_510 OBJECT ${UNIT_TEST_VERSION_510_SRCS}) @@ -194,7 +193,6 @@ if(NOT WIN32) add_executable(fdb_c_client_memory_test test/client_memory_test.cpp test/unit/fdb_api.cpp test/unit/fdb_api.hpp) if(BUILD_MAKO) add_executable(mako ${MAKO_SRCS}) - add_executable(mako_async_test test/mako/async_test.cpp test/mako/async.cpp) endif() add_executable(fdb_c_setup_tests test/unit/setup_tests.cpp) add_executable(fdb_c_unit_tests) @@ -237,9 +235,7 @@ if(NOT WIN32) # do not set RPATH for mako set_property(TARGET mako PROPERTY SKIP_BUILD_RPATH TRUE) target_link_libraries(mako PRIVATE fdb_c fdbclient fmt::fmt Threads::Threads fdb_cpp boost_target rapidjson) - target_link_libraries(mako_async_test PRIVATE fdb_c fdbclient fmt::fmt Threads::Threads fdb_cpp boost_target rapidjson doctest) if(NOT OPEN_FOR_IDE) - add_test(NAME mako_async_test COMMAND $) strip_debug_symbols(mako) if(NOT GENERATE_DEBUG_PACKAGES) add_custom_target(prepare_mako_install ALL DEPENDS strip_only_mako) diff --git a/bindings/c/test/mako/async_test.cpp b/bindings/c/test/mako/async_test.cpp deleted file mode 100644 index f1e3e7e043f..00000000000 --- a/bindings/c/test/mako/async_test.cpp +++ /dev/null @@ -1,200 +0,0 @@ -/* - * async_test.cpp - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -#define DOCTEST_CONFIG_IMPLEMENT_WITH_MAIN -#include "doctest.h" - -#include "async.hpp" -#include "operations.hpp" -#include -#include - -thread_local mako::Logger logr{ mako::MainProcess{}, mako::VERBOSE_NONE }; - -namespace mako { - -// The driver tests do not link the benchmark's main or require a database. -Arguments::Arguments() - : rows(100), row_digits(3), sampling(1), key_length(16), value_length(16), zipf(0), commit_get(0), txnspec{}, - prefixpadding(0), transaction_timeout_db(0), transaction_timeout_tx(0), max_grv_queue_delay_ms(-1) {} - -bool Arguments::isAnyTimeoutEnabled() const { - return transaction_timeout_tx > 0 || transaction_timeout_db > 0; -} - -namespace { - -struct StopTick {}; -const auto staleStart = timepoint_t{} - std::chrono::seconds(1); - -class AsyncWorkloadTest { - static thread_local AsyncWorkloadTest* active; - Arguments args; - WorkflowStatistics stats; - boost::asio::io_context io; - std::atomic stopcount{ 0 }; - std::atomic signal{ SIGNAL_GREEN }; - RunWorkloadStateHandle state; - std::function observer; - -public: - AsyncWorkloadTest(WorkloadSpec spec, std::function observer) - : observer(std::move(observer)) { - assert(active == nullptr); - args.txnspec = spec; - state = std::make_shared( - logr, fdb::Database{}, fdb::Transaction{}, io, args, stats, stopcount, signal, -1, getOpBegin(args)); - active = this; - } - ~AsyncWorkloadTest() { active = nullptr; } - AsyncWorkloadTest(const AsyncWorkloadTest&) = delete; - AsyncWorkloadTest& operator=(const AsyncWorkloadTest&) = delete; - - ResumableStateForRunWorkload& workload() { return *state; } - WorkflowStatistics const& statistics() const { return stats; } - void runTick() { state->runOneTick(); } - void pollTick() { io.poll_one(); } - - static fdb::Future step(fdb::Transaction&, Arguments const&, fdb::ByteString&, fdb::ByteString&, fdb::ByteString&) { - active->observer(*active->state); - return {}; - } -}; - -thread_local AsyncWorkloadTest* AsyncWorkloadTest::active = nullptr; - -void checkPreparedKey(fdb::ByteString const& key) { - CHECK(std::equal(KEY_PREFIX.begin(), KEY_PREFIX.end(), key.begin())); -} - -void checkFreshTimers(ResumableStateForRunWorkload const& state) { - CHECK(state.watch_step.getStart() != staleStart); - CHECK(state.watch_op.getStart() == state.watch_step.getStart()); -} - -void poisonOperation(ResumableStateForRunWorkload& state) { - state.key1.assign(state.key1.size(), '!'); - state.key2.assign(state.key2.size(), '!'); - // Distinct starts make resets observable without waiting for the clock to advance. - state.watch_step = Stopwatch(staleStart); - state.watch_op = Stopwatch(staleStart); -} - -} // namespace - -// Synthetic immediate steps exercise the real driver while the observer stops before commit/reset. -const std::array opTable = [] { - std::array table{}; - const auto step = Step{ StepKind::IMM, &AsyncWorkloadTest::step }; - table[OP_OVERWRITE] = Operation{ "OVERWRITE", { step }, 1, true }; - table[OP_CLEAR] = Operation{ "CLEAR", { step }, 1, true }; - table[OP_SETCLEAR] = Operation{ "SETCLEAR", { step, step }, 2, true }; - table[OP_CLEARRANGE] = Operation{ "CLEARRANGE", { step }, 1, true }; - return table; -}(); - -TEST_CASE("mako async repeated immediate operations") { - WorkloadSpec spec{}; - spec.ops[OP_OVERWRITE][OP_COUNT] = 10; - spec.ops[OP_CLEAR][OP_COUNT] = 1; - int overwrites = 0; - AsyncWorkloadTest test(spec, [&](auto& state) { - checkPreparedKey(state.key1); - checkFreshTimers(state); - if (state.iter.op == OP_CLEAR) - throw StopTick{}; - CHECK(state.iter.count == overwrites++); - poisonOperation(state); - }); - CHECK_THROWS_AS(test.runTick(), StopTick); - CHECK(overwrites == 10); - CHECK(test.statistics().getOpCount(OP_OVERWRITE) == 10); - CHECK(test.statistics().getLatencySampleCount(OP_OVERWRITE) == 10); -} - -TEST_CASE("mako async immediate transition prepares range keys") { - WorkloadSpec spec{}; - spec.ops[OP_OVERWRITE][OP_COUNT] = 1; - spec.ops[OP_CLEARRANGE][OP_COUNT] = 1; - spec.ops[OP_CLEARRANGE][OP_RANGE] = 7; - AsyncWorkloadTest test(spec, [&](auto& state) { - checkPreparedKey(state.key1); - checkFreshTimers(state); - if (state.iter.op == OP_CLEARRANGE) { - checkPreparedKey(state.key2); - throw StopTick{}; - } - poisonOperation(state); - }); - CHECK_THROWS_AS(test.runTick(), StopTick); - CHECK(test.statistics().getOpCount(OP_OVERWRITE) == 1); -} - -TEST_CASE("mako async preserves keys and operation timer between steps") { - WorkloadSpec spec{}; - spec.ops[OP_SETCLEAR][OP_COUNT] = 2; - spec.ops[OP_CLEARRANGE][OP_COUNT] = 1; - AsyncWorkloadTest test(spec, [&](auto& state) { - CHECK(state.watch_step.getStart() != staleStart); - if (state.iter.step == 0) { - checkPreparedKey(state.key1); - checkFreshTimers(state); - } else { - CHECK(state.key1 == fdb::ByteString(state.key1.size(), '!')); - CHECK(state.key2 == fdb::ByteString(state.key2.size(), '!')); - CHECK(state.watch_op.getStart() == staleStart); - } - if (state.iter.op == OP_CLEARRANGE) - throw StopTick{}; - poisonOperation(state); - }); - SUBCASE("steps complete in the same tick") {} - SUBCASE("resume at a later step") { - test.workload().iter.step = 1; - poisonOperation(test.workload()); - } - CHECK_THROWS_AS(test.runTick(), StopTick); - CHECK(test.statistics().getOpCount(OP_SETCLEAR) == 2); -} - -TEST_CASE("mako async retry restarts operation preparation") { - WorkloadSpec spec{}; - spec.ops[OP_OVERWRITE][OP_COUNT] = 2; - spec.ops[OP_CLEAR][OP_COUNT] = 1; - int overwrites = 0; - AsyncWorkloadTest test(spec, [&](auto& state) { - checkPreparedKey(state.key1); - checkFreshTimers(state); - if (state.iter.op == OP_CLEAR) - throw StopTick{}; - CHECK(state.iter.count == overwrites++); - poisonOperation(state); - }); - test.workload().iter.count = 1; - test.workload().needs_commit = true; - poisonOperation(test.workload()); - test.workload().onIterationEnd(FutureRC::RETRY); - CHECK(test.workload().total_xacts == 0); - CHECK_FALSE(test.workload().needs_commit); - CHECK_THROWS_AS(test.pollTick(), StopTick); - CHECK(overwrites == 2); -} - -} // namespace mako From c1e672e8e89ad0cc1a4981622a4b2ae713a2ecc3 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sat, 12 Sep 2026 17:17:56 -0700 Subject: [PATCH 046/170] Prevent GRV master replies from being starved by request intake --- fdbserver/grvproxy/GrvProxyServer.cpp | 35 ++++++++++++++++++++++++++- 1 file changed, 34 insertions(+), 1 deletion(-) diff --git a/fdbserver/grvproxy/GrvProxyServer.cpp b/fdbserver/grvproxy/GrvProxyServer.cpp index d3b2775fe51..e9f6c994796 100644 --- a/fdbserver/grvproxy/GrvProxyServer.cpp +++ b/fdbserver/grvproxy/GrvProxyServer.cpp @@ -40,6 +40,7 @@ #include "flow/Buggify.h" #include "flow/IRandom.h" #include "flow/Trace.h" +#include "flow/UnitTest.h" #include "flow/flow.h" #include "flow/CoroUtils.h" #include "flow/genericactors.h" @@ -713,9 +714,10 @@ Future getLiveCommittedVersion(std::vector spa double grvStart = now(); Optional debugID = getDebugID(debugIDs); Future replyFromMasterFuture; + // Receive master replies at socket priority so incoming GRVs cannot starve an already-arrived reply. replyFromMasterFuture = grvProxyData->master.getLiveCommittedVersion.getReply( GetRawCommittedVersionRequest(span.context, debugID, grvProxyData->ssVersionVectorCache.getMaxVersion()), - TaskPriority::GetLiveCommittedVersionReply); + TaskPriority::ReadSocket); if (!SERVER_KNOBS->ALWAYS_CAUSAL_READ_RISKY && !(flags & GetReadVersionRequest::FLAG_CAUSAL_READ_RISKY)) { co_await transformError(updateLastCommit(grvProxyData, debugID), broken_promise(), tlog_failed()); @@ -1286,3 +1288,34 @@ Future grvProxyServer(GrvProxyInterface proxy, } } } + +TEST_CASE("noSim/fdbserver/grvproxy/masterReplyProgressAtSocketPriority") { + if (g_network->isSimulated()) { + co_return; + } + + auto db = makeReference>(); + MasterInterface master; + GrvProxyInterface proxy; + GrvProxyData data(deterministicRandom()->randomUniqueID(), master, proxy.getConsistentReadVersion, db); + // A fresh epoch confirmation isolates master-reply scheduling from TLog liveness. + data.lastCommitTime = now(); + Future pending = getLiveCommittedVersion( + {}, &data, GetReadVersionRequest::FLAG_CAUSAL_READ_RISKY, Optional(), 1, 0, 1, 0); + GetRawCommittedVersionRequest request = co_await master.getLiveCommittedVersion.getFuture(); + + // Serialize the reply through transport so delivery uses the endpoint priority selected by the proxy. + ReplyPromise sender; + sender.loadRemoteEndpoint( + Endpoint(FlowTransport::transport().getLocalAddresses(), request.reply.getEndpoint().token)); + GetRawCommittedVersionReply reply; + reply.version = 123; + reply.minKnownCommittedVersion = 123; + sender.send(reply); + + co_await delay(0, TaskPriority::ReadSocket); + bool completedBeforeNextSocketTask = pending.isReady(); + GetReadVersionReply result = co_await pending; + ASSERT(completedBeforeNextSocketTask); + ASSERT_EQ(result.version, reply.version); +} From e6bb9868b684792d5319e1a30fd741af4a3ee662 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sat, 12 Sep 2026 17:29:12 -0700 Subject: [PATCH 047/170] Remove GRV master reply priority unit test --- fdbserver/grvproxy/GrvProxyServer.cpp | 32 --------------------------- 1 file changed, 32 deletions(-) diff --git a/fdbserver/grvproxy/GrvProxyServer.cpp b/fdbserver/grvproxy/GrvProxyServer.cpp index e9f6c994796..8d5c87c939e 100644 --- a/fdbserver/grvproxy/GrvProxyServer.cpp +++ b/fdbserver/grvproxy/GrvProxyServer.cpp @@ -40,7 +40,6 @@ #include "flow/Buggify.h" #include "flow/IRandom.h" #include "flow/Trace.h" -#include "flow/UnitTest.h" #include "flow/flow.h" #include "flow/CoroUtils.h" #include "flow/genericactors.h" @@ -1288,34 +1287,3 @@ Future grvProxyServer(GrvProxyInterface proxy, } } } - -TEST_CASE("noSim/fdbserver/grvproxy/masterReplyProgressAtSocketPriority") { - if (g_network->isSimulated()) { - co_return; - } - - auto db = makeReference>(); - MasterInterface master; - GrvProxyInterface proxy; - GrvProxyData data(deterministicRandom()->randomUniqueID(), master, proxy.getConsistentReadVersion, db); - // A fresh epoch confirmation isolates master-reply scheduling from TLog liveness. - data.lastCommitTime = now(); - Future pending = getLiveCommittedVersion( - {}, &data, GetReadVersionRequest::FLAG_CAUSAL_READ_RISKY, Optional(), 1, 0, 1, 0); - GetRawCommittedVersionRequest request = co_await master.getLiveCommittedVersion.getFuture(); - - // Serialize the reply through transport so delivery uses the endpoint priority selected by the proxy. - ReplyPromise sender; - sender.loadRemoteEndpoint( - Endpoint(FlowTransport::transport().getLocalAddresses(), request.reply.getEndpoint().token)); - GetRawCommittedVersionReply reply; - reply.version = 123; - reply.minKnownCommittedVersion = 123; - sender.send(reply); - - co_await delay(0, TaskPriority::ReadSocket); - bool completedBeforeNextSocketTask = pending.isReady(); - GetReadVersionReply result = co_await pending; - ASSERT(completedBeforeNextSocketTask); - ASSERT_EQ(result.version, reply.version); -} From 99608a92301e7fefc89f1d3b1eeb5e6c2310e1ec Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sat, 12 Sep 2026 20:11:16 -0700 Subject: [PATCH 048/170] Honor simulated DD placement in consistency checks --- fdbserver/workloads/ConsistencyCheck.cpp | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/fdbserver/workloads/ConsistencyCheck.cpp b/fdbserver/workloads/ConsistencyCheck.cpp index 1c65ecead4e..23b0dbe1540 100644 --- a/fdbserver/workloads/ConsistencyCheck.cpp +++ b/fdbserver/workloads/ConsistencyCheck.cpp @@ -1026,10 +1026,13 @@ struct ConsistencyCheckWorkload : TestWorkload { // Check DataDistributor recruitment::Fitness fitnessLowerBound = recruitment::machineClassFitness( allWorkerProcessMap[db.master.address()].processClass, recruitment::DataDistributor); + // Bulk-load simulation can deliberately retain a DD with suboptimal fitness to let data moves finish. + const bool allowUnfitDistributor = g_network->isSimulated() && SERVER_KNOBS->CC_ENFORCE_USE_UNFIT_DD_IN_SIM; if (db.distributor.present() && (!nonExcludedWorkerProcessMap.contains(db.distributor.get().address()) || - recruitment::machineClassFitness(nonExcludedWorkerProcessMap[db.distributor.get().address()].processClass, - recruitment::DataDistributor) > fitnessLowerBound)) { + (!allowUnfitDistributor && + recruitment::machineClassFitness(nonExcludedWorkerProcessMap[db.distributor.get().address()].processClass, + recruitment::DataDistributor) > fitnessLowerBound))) { TraceEvent("ConsistencyCheck_DistributorNotBest") .detail("DataDistributorFitnessLowerBound", fitnessLowerBound) .detail("ExistingDistributorFitness", From 52e3dcb98287a6485028c719f0b7027c8f323135 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sat, 12 Sep 2026 20:13:28 -0700 Subject: [PATCH 049/170] Keep draining storage after exhausting the clear-range budget --- fdbserver/storageserver/storageserver.cpp | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/fdbserver/storageserver/storageserver.cpp b/fdbserver/storageserver/storageserver.cpp index 3cee29d3d32..0af10cf204d 100644 --- a/fdbserver/storageserver/storageserver.cpp +++ b/fdbserver/storageserver/storageserver.cpp @@ -11215,9 +11215,10 @@ Future updateStorage(StorageServer* data) { ++data->counters.kvCommits; recentCommitStats.back().seqId = data->counters.kvCommits.getValue(); - // If the mutation bytes budget was not fully used then wait some time before the next commit - durableDelay = - (bytesLeft > 0) ? delay(SERVER_KNOBS->STORAGE_COMMIT_INTERVAL, TaskPriority::UpdateStorage) : Void(); + // Batch only while both budgets have capacity; otherwise keep draining pending mutations. + durableDelay = (bytesLeft > 0 && clearRangesLeft > 0) + ? delay(SERVER_KNOBS->STORAGE_COMMIT_INTERVAL, TaskPriority::UpdateStorage) + : Void(); recentCommitStats.back().whenCommit = now(); try { From 7ed9647faff1e28a7dd84a49b5f3bcab1debb366 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sat, 12 Sep 2026 20:21:16 -0700 Subject: [PATCH 050/170] Remove added old log router unit tests --- .../logsystem/LogSystemRecoveryTests.cpp | 195 ------------------ fdbserver/worker/worker.cpp | 152 -------------- 2 files changed, 347 deletions(-) diff --git a/fdbserver/logsystem/LogSystemRecoveryTests.cpp b/fdbserver/logsystem/LogSystemRecoveryTests.cpp index 9e63c717da3..70baf43d283 100644 --- a/fdbserver/logsystem/LogSystemRecoveryTests.cpp +++ b/fdbserver/logsystem/LogSystemRecoveryTests.cpp @@ -20,13 +20,8 @@ #include "fdbserver/logsystem/LogSystem.h" #include "fdbserver/logsystem/LogSystemConsumer.h" -#include "fdbserver/core/Knobs.h" -#include "fdbrpc/FailureMonitor.h" -#include "flow/ScopeExit.h" #include "flow/UnitTest.h" -#include - namespace { Reference makeSingleLogSet(const std::vector& tlogs, bool isLocal = true) { @@ -59,200 +54,10 @@ std::tuple, bool> makeLogGroupResults( return std::make_tuple(replicationFactor, std::move(lockResults), nonAvailableTLogsCompletePolicy); } -class ScopedHealthyTestEndpoints { - IFailureMonitor& monitor = IFailureMonitor::failureMonitor(); - std::map savedStatus; - -public: - void add(Endpoint const& endpoint) { - const NetworkAddress address = endpoint.getPrimaryAddress(); - if (savedStatus.emplace(address, monitor.getState(address)).second) { - monitor.setStatus(address, FailureStatus(false)); - } - assertAvailable(endpoint); - } - void assertAvailable(Endpoint const& endpoint) const { ASSERT(monitor.getState(endpoint).isAvailable()); } - - ~ScopedHealthyTestEndpoints() { - for (const auto& [address, status] : savedStatus) { - monitor.setStatus(address, status); - } - } -}; - -class OldLogRouterRecruitmentFixture { - ScopedHealthyTestEndpoints healthyEndpoints; - Reference logSystem; - Reference logSet; - WorkerInterface worker; - TLogInterface router; - const bool forRemote; - Future recruitment; - -public: - static constexpr double retryDelay = 0.01; - - explicit OldLogRouterRecruitmentFixture(bool forRemote) - : logSystem(makeReference(UID(1, 1), LocalityData(), LogEpoch(1))), logSet(makeReference()), - router(UID(2, 2), UID(2, 2), LocalityData()), forRemote(forRemote) { - worker.logRouter.getEndpoint(TaskPriority::Worker); - router.initEndpoints(); - // These in-process RPCs have no connection that would mark their address healthy. - healthyEndpoints.add(worker.logRouter.getEndpoint()); - healthyEndpoints.add(router.waitFailure.getEndpoint()); - logSet->locality = 1; - logSet->startVersion = 100; - logSystem->logRouterTags = 1; - logSystem->recoverAt = 200; - logSystem->knownLockedTLogIds[1] = { 2, 4 }; - if (forRemote) { - OldLogData old; - old.logRouterTags = 1; - old.recoverAt = 200; - old.tLogs.push_back(logSet); - logSystem->oldLogData.push_back(std::move(old)); - } else { - logSystem->tLogs.push_back(logSet); - } - } - - ~OldLogRouterRecruitmentFixture() { recruitment.cancel(); } - - Future start() { - ASSERT(!recruitment.isValid()); - const double savedTimeout = SERVER_KNOBS->CC_RERECRUIT_LOG_ROUTER_TIMEOUT; - ScopeExit restoreTimeout( - [savedTimeout] { setServerKnob("cc_rerecruit_log_router_timeout", KnobValueRef::create(savedTimeout)); }); - setServerKnob("cc_rerecruit_log_router_timeout", KnobValueRef::create(retryDelay)); - // The eager coroutine captures its retry deadline before this synchronous scope restores the knob. - recruitment = logSystem->recruitOldLogRouters( - { worker }, LogEpoch(2), 1, 100, { LocalityData() }, makeReference(), forRemote); - return recruitment; - } - - FutureStream requests() const { return worker.logRouter.getFuture(); } - Future nextRequest() const { - return timeoutError(waitOrError(requests(), recruitment), 1.0); - } - FutureStream> failureRequests() const { return router.waitFailure.getFuture(); } - Future onConfigChange() { return logSystem->onLogSystemConfigChange(); } - void reply(InitializeLogRouterRequest const& request) const { request.reply.send(router); } - bool hasRouter() const { - return logSet->logRouters.size() == 1 && logSet->logRouters.front()->get().id() == router.id(); - } - bool hasNoRouters() const { return logSet->logRouters.empty(); } - void assertPending() const { - healthyEndpoints.assertAvailable(worker.logRouter.getEndpoint()); - healthyEndpoints.assertAvailable(router.waitFailure.getEndpoint()); - if (recruitment.isError()) { - throw recruitment.getError(); - } - ASSERT(!recruitment.isReady()); - } - void cancel() { recruitment.cancel(); } -}; - -void assertSameLogRouterRequest(InitializeLogRouterRequest const& first, InitializeLogRouterRequest const& retry) { - ASSERT(first.reqId == retry.reqId); - ASSERT_EQ(first.recoveryCount, retry.recoveryCount); - ASSERT(first.routerTag == retry.routerTag); - ASSERT_EQ(first.startVersion, retry.startVersion); - ASSERT(first.tLogLocalities == retry.tLogLocalities); - ASSERT(first.tLogPolicy == retry.tLogPolicy); - ASSERT_EQ(first.locality, retry.locality); - ASSERT(first.recoverAt == retry.recoverAt); - ASSERT(first.knownLockedTLogIds == retry.knownLockedTLogIds); - ASSERT_EQ(first.allowDropInSim, retry.allowDropInSim); - ASSERT_EQ(first.isReplacement, retry.isReplacement); - ASSERT(!(first.reply.getFuture() == retry.reply.getFuture())); -} - } // namespace void forceLinkLogSystemRecoveryTests() {} -TEST_CASE("/LogSystem/RecruitOldLogRouters/LocalRetryPreservesRequest") { - OldLogRouterRecruitmentFixture fixture(false); - Future changed = fixture.onConfigChange(); - Future recruitment = fixture.start(); - InitializeLogRouterRequest request = co_await fixture.nextRequest(); - InitializeLogRouterRequest retry = co_await fixture.nextRequest(); - assertSameLogRouterRequest(request, retry); - fixture.assertPending(); - ASSERT(fixture.hasNoRouters()); - fixture.reply(retry); - co_await timeoutError(changed, 1.0); - ASSERT(fixture.hasRouter()); - fixture.assertPending(); - fixture.cancel(); -} - -TEST_CASE("/LogSystem/RecruitOldLogRouters/LocalLateOriginalReply") { - OldLogRouterRecruitmentFixture fixture(false); - Future changed = fixture.onConfigChange(); - Future recruitment = fixture.start(); - InitializeLogRouterRequest request = co_await fixture.nextRequest(); - InitializeLogRouterRequest retry = co_await fixture.nextRequest(); - assertSameLogRouterRequest(request, retry); - fixture.reply(request); - co_await timeoutError(changed, 1.0); - ASSERT(fixture.hasRouter()); - fixture.assertPending(); - fixture.cancel(); -} - -TEST_CASE("/LogSystem/RecruitOldLogRouters/LocalReplySuccess") { - OldLogRouterRecruitmentFixture fixture(false); - Future changed = fixture.onConfigChange(); - Future recruitment = fixture.start(); - InitializeLogRouterRequest request = co_await fixture.nextRequest(); - fixture.reply(request); - co_await timeoutError(changed, 1.0); - ASSERT(fixture.hasRouter()); - ReplyPromise failureRequest = co_await fixture.failureRequests(); - - co_await delay(2 * OldLogRouterRecruitmentFixture::retryDelay); - fixture.assertPending(); - ASSERT(!fixture.requests().isReady()); - fixture.cancel(); - ASSERT(recruitment.isError()); - ASSERT_EQ(recruitment.getError().code(), error_code_actor_cancelled); - ASSERT_EQ(failureRequest.getFutureReferenceCount(), 0); -} - -TEST_CASE("/LogSystem/RecruitOldLogRouters/LocalRetryWaitAndCancellation") { - OldLogRouterRecruitmentFixture fixture(false); - Future recruitment = fixture.start(); - InitializeLogRouterRequest request = co_await fixture.nextRequest(); - InitializeLogRouterRequest retry = co_await fixture.nextRequest(); - assertSameLogRouterRequest(request, retry); - co_await delay(2 * OldLogRouterRecruitmentFixture::retryDelay); - fixture.assertPending(); - ASSERT(!fixture.requests().isReady()); - ASSERT(request.reply.getFutureReferenceCount() > 0); - ASSERT(retry.reply.getFutureReferenceCount() > 0); - fixture.cancel(); - ASSERT(recruitment.isError()); - ASSERT_EQ(recruitment.getError().code(), error_code_actor_cancelled); - ASSERT(fixture.hasNoRouters()); - ASSERT_EQ(request.reply.getFutureReferenceCount(), 0); - ASSERT_EQ(retry.reply.getFutureReferenceCount(), 0); -} - -TEST_CASE("/LogSystem/RecruitOldLogRouters/RemoteReplyWait") { - OldLogRouterRecruitmentFixture fixture(true); - Future recruitment = fixture.start(); - InitializeLogRouterRequest request = co_await fixture.nextRequest(); - ASSERT(!request.allowDropInSim); - co_await delay(2 * OldLogRouterRecruitmentFixture::retryDelay); - fixture.assertPending(); - ASSERT(!fixture.requests().isReady()); - ASSERT(fixture.hasNoRouters()); - fixture.reply(request); - co_await timeoutError(recruitment, 1.0); - ASSERT(fixture.hasRouter()); -} - TEST_CASE("/LogSystem/GetPseudoPopTag/LogRouterWithoutMappedLocality") { LocalityData locality; auto logSystem = makeReference(UID(), locality, LogEpoch(1)); diff --git a/fdbserver/worker/worker.cpp b/fdbserver/worker/worker.cpp index a5654de6765..5dd0bf244a1 100644 --- a/fdbserver/worker/worker.cpp +++ b/fdbserver/worker/worker.cpp @@ -47,7 +47,6 @@ #include "flow/ObjectSerializer.h" #include "flow/Platform.h" #include "flow/ProtocolVersion.h" -#include "flow/ScopeExit.h" #include "flow/SystemMonitor.h" #include "flow/TDMetric.h" #include "fdbrpc/simulator.h" @@ -62,7 +61,6 @@ #include "fdbserver/datadistributor/DataDistributor.h" #include "fdbserver/grvproxy/GrvProxyServer.h" #include "fdbserver/logrouter/LogRouter.h" -#include "fdbserver/logsystem/LogSystem.h" #include "fdbserver/core/BackupInterface.h" #include "RoleLineage.h" #include "fdbserver/core/WorkerInterface.h" @@ -1981,156 +1979,6 @@ bool replyToCachedLogRouter(WorkerCache& cache, InitializeLogRout return true; } -TEST_CASE("/fdbserver/worker/logRouterInitialization/cachedReplySurvivesDroppedResponse") { - WorkerCache cache; - InitializeLogRouterRequest first{}; - first.reqId = UID(1, 1); - Future firstReply = first.reply.getFuture(); - ASSERT(!replyToCachedLogRouter(cache, first)); - - TLogInterface router(UID(2, 2), UID(2, 2), LocalityData()); - Promise roleLifetime; - Promise initialized = cacheLogRouterInitialization(cache, first); - Future role = cache.removeOnReady(first.reqId, roleLifetime.getFuture()); - initialized.send(router); - ASSERT(!role.isReady()); - ASSERT(!firstReply.isReady()); - - InitializeLogRouterRequest duplicate = first; - duplicate.reply.reset(); - Future duplicateReply = duplicate.reply.getFuture(); - ASSERT(replyToCachedLogRouter(cache, duplicate)); - TLogInterface cachedRouter = co_await timeoutError(duplicateReply, 1.0); - ASSERT_EQ(cachedRouter.id(), router.id()); - ASSERT(!firstReply.isReady()); - ASSERT(cache.exists(first.reqId)); - ASSERT(!role.isReady()); - - roleLifetime.send(Void()); - co_await role; - ASSERT(!cache.exists(first.reqId)); - ASSERT(!replyToCachedLogRouter(cache, duplicate)); -} - -TEST_CASE("/fdbserver/worker/logRouterInitialization/recruitmentRetriesDroppedResponse") { - constexpr double retryDelay = 0.01; - IFailureMonitor& failureMonitor = IFailureMonitor::failureMonitor(); - std::map savedStatus; - ScopeExit restoreStatus([&] { - for (const auto& [address, status] : savedStatus) { - failureMonitor.setStatus(address, status); - } - }); - WorkerInterface worker; - worker.logRouter.getEndpoint(TaskPriority::Worker); - TLogInterface router(UID(2, 2), UID(2, 2), LocalityData()); - router.initEndpoints(); - const Endpoint endpoints[] = { worker.logRouter.getEndpoint(), router.waitFailure.getEndpoint() }; - for (const auto& endpoint : endpoints) { - const NetworkAddress address = endpoint.getPrimaryAddress(); - if (savedStatus.emplace(address, failureMonitor.getState(address)).second) { - failureMonitor.setStatus(address, FailureStatus(false)); - } - ASSERT(failureMonitor.getState(endpoint).isAvailable()); - } - - auto logSystem = makeReference(UID(1, 1), LocalityData(), LogEpoch(1)); - auto logSet = makeReference(); - logSet->locality = 1; - logSet->startVersion = 100; - logSystem->logRouterTags = 1; - logSystem->recoverAt = 200; - logSystem->knownLockedTLogIds[1] = { 2, 4 }; - logSystem->tLogs.push_back(logSet); - Future changed = logSystem->onLogSystemConfigChange(); - WorkerCache cache; - Promise roleLifetime; - Promise initialized; - InitializeLogRouterRequest first{}; - InitializeLogRouterRequest retry{}; - ReplyPromise failureRequest; - Future firstReply; - Future retryReply; - Future role; - Future recruitment; - ScopeExit cancelActors([&] { - recruitment.cancel(); - role.cancel(); - }); - { - const double savedTimeout = SERVER_KNOBS->CC_RERECRUIT_LOG_ROUTER_TIMEOUT; - ScopeExit restoreTimeout( - [savedTimeout] { setServerKnob("cc_rerecruit_log_router_timeout", KnobValueRef::create(savedTimeout)); }); - setServerKnob("cc_rerecruit_log_router_timeout", KnobValueRef::create(retryDelay)); - recruitment = logSystem->recruitOldLogRouters( - { worker }, LogEpoch(2), 1, 100, { LocalityData() }, makeReference(), false); - } - - const char* stage = "first request"; - try { - first = co_await timeoutError(waitOrError(worker.logRouter.getFuture(), recruitment), 1.0); - firstReply = first.reply.getFuture(); - ASSERT(!replyToCachedLogRouter(cache, first)); - initialized = cacheLogRouterInitialization(cache, first); - role = cache.removeOnReady(first.reqId, roleLifetime.getFuture()); - initialized.send(router); - ASSERT(!firstReply.isReady()); - - stage = "retry request"; - retry = co_await timeoutError(waitOrError(worker.logRouter.getFuture(), recruitment), 1.0); - ASSERT(retry.reqId == first.reqId); - ASSERT_EQ(retry.recoveryCount, first.recoveryCount); - ASSERT(retry.routerTag == first.routerTag); - ASSERT_EQ(retry.startVersion, first.startVersion); - ASSERT(retry.tLogLocalities == first.tLogLocalities); - ASSERT(retry.tLogPolicy == first.tLogPolicy); - ASSERT_EQ(retry.locality, first.locality); - ASSERT(retry.recoverAt == first.recoverAt); - ASSERT(retry.knownLockedTLogIds == first.knownLockedTLogIds); - ASSERT_EQ(retry.allowDropInSim, first.allowDropInSim); - ASSERT_EQ(retry.isReplacement, first.isReplacement); - ASSERT(!(first.reply.getFuture() == retry.reply.getFuture())); - retryReply = retry.reply.getFuture(); - ASSERT(replyToCachedLogRouter(cache, retry)); - stage = "cached reply"; - TLogInterface cachedRouter = co_await timeoutError(retryReply, 1.0); - ASSERT_EQ(cachedRouter.id(), router.id()); - stage = "router publication"; - co_await timeoutError(waitOrError(changed, recruitment), 1.0); - ASSERT_EQ(logSet->logRouters.size(), 1); - ASSERT_EQ(logSet->logRouters.front()->get().id(), router.id()); - ASSERT(!firstReply.isReady()); - ASSERT(cache.exists(first.reqId)); - ASSERT(!role.isReady()); - ASSERT(!recruitment.isReady()); - stage = "failure monitoring"; - failureRequest = co_await timeoutError(waitAndForward(router.waitFailure.getFuture()), 1.0); - co_await delay(2 * retryDelay); - ASSERT(!worker.logRouter.getFuture().isReady()); - ASSERT(!recruitment.isReady()); - - recruitment.cancel(); - ASSERT(recruitment.isError()); - ASSERT_EQ(recruitment.getError().code(), error_code_actor_cancelled); - ASSERT_EQ(failureRequest.getFutureReferenceCount(), 0); - firstReply = Future(); - ASSERT_EQ(first.reply.getFutureReferenceCount(), 0); - ASSERT(cache.exists(first.reqId)); - roleLifetime.send(Void()); - co_await role; - ASSERT(!cache.exists(first.reqId)); - } catch (Error& e) { - fprintf(stderr, - "Log router recruitment test failed at %s (cacheReady=%d, firstReplyReady=%d, configChanged=%d): %s\n", - stage, - cache.exists(first.reqId) && cache.get(first.reqId).isReady(), - firstReply.isValid() && firstReply.isReady(), - changed.isReady(), - e.what()); - throw; - } -} - #ifdef FLOW_GRPC_ENABLED Future registerWorkerGrpcServices(UID id, Reference ccr) { if (GrpcServer::instance() == nullptr) { From 21a7c391f24191a08629e69731a200bb30acda44 Mon Sep 17 00:00:00 2001 From: Ayush Kumar Anand <143122971+ayushk-1801@users.noreply.github.com> Date: Mon, 14 Sep 2026 22:40:24 +0530 Subject: [PATCH 051/170] Fix "versioned" client rpm package build-id conflicts (#14037) There are regular foundationdb-clients and foundationdb-server RPM packages defined and published, but like normal RPM packages only one version of each can be installed at one time. So we also define "versioned" packages like "foundationdbX.Y.Z-clients" so that multiple versions can be installed in parallel. But they conflict with build-id links from the main-package at the same version because they contain the exact same binaries/libraries. So, remove these build-id links from "versioned" packages so they cannot conflict. Also: * Add libfdb_c_shim.so to the fdbclients alternatives group. * Prevent the versioned client prerm script from removing alternatives during RPM upgrades. * Document the difference between regular RPM upgrades and versioned client package coexistence. --- cmake/InstallLayout.cmake | 6 +++++ .../sphinx/source/getting-started-linux.rst | 25 +++++++++++++++++++ packaging/multiversion/clients/postinst | 3 ++- packaging/multiversion/clients/prerm | 6 ++++- 4 files changed, 38 insertions(+), 2 deletions(-) diff --git a/cmake/InstallLayout.cmake b/cmake/InstallLayout.cmake index 02bb8971f4f..616d2253e9e 100644 --- a/cmake/InstallLayout.cmake +++ b/cmake/InstallLayout.cmake @@ -261,6 +261,12 @@ set(CPACK_RPM_SERVER-VERSIONED_PACKAGE_REQUIRES "${CPACK_COMPONENT_CL set(CPACK_RPM_SERVER-VERSIONED_POST_INSTALL_SCRIPT_FILE ${CMAKE_BINARY_DIR}/packaging/multiversion/server/postinst-rpm) set(CPACK_RPM_SERVER-VERSIONED_PRE_UNINSTALL_SCRIPT_FILE ${CMAKE_BINARY_DIR}/packaging/multiversion/server/prerm) +# Versioned packages must not own RPM's global build-id links. Two otherwise +# independent client packages can contain identical binaries, and those links +# would make the packages conflict outside their versioned install tree. +set(CPACK_RPM_SPEC_MORE_DEFINE + "%if \\\"%{name}\\\" == \\\"${CPACK_RPM_CLIENTS-VERSIONED_PACKAGE_NAME}\\\"\n%define _build_id_links none\n%endif") + file(MAKE_DIRECTORY "${CMAKE_BINARY_DIR}/packaging/emptydir") fdb_install(DIRECTORY "${CMAKE_BINARY_DIR}/packaging/emptydir/" DESTINATION data COMPONENT server) fdb_install(DIRECTORY "${CMAKE_BINARY_DIR}/packaging/emptydir/" DESTINATION log COMPONENT server) diff --git a/documentation/sphinx/source/getting-started-linux.rst b/documentation/sphinx/source/getting-started-linux.rst index 25ee0628e74..01add5f1300 100644 --- a/documentation/sphinx/source/getting-started-linux.rst +++ b/documentation/sphinx/source/getting-started-linux.rst @@ -48,6 +48,31 @@ To install on **RHEL/CentOS 7** use the rpm command: |simple-installation-mode-warnings| |networking-clarification| + +Installing multiple client versions +==================================== + +The regular ``foundationdb-clients`` RPM installs files in shared system paths +and is intended for a normal install or upgrade. Do not install two regular +client RPMs with ``rpm -ivh``; they own the same files and RPM will report file +conflicts. + +For multi-version client support, use the separately named CPack versioned +client RPMs. Their names have the form +``foundationdb-clients-1.versioned..rpm`` and their +files are installed below ``/usr/lib/foundationdb-/``. Non-release +builds may include build-time and prerelease text in the package name and +directory. Install each versioned package with ``rpm -ivh`` so the packages +can remain installed simultaneously. If a release does not publish its +versioned RPM, build the +``clients-versioned`` component from source with CPack (run ``cpack -G RPM`` +from the configured build directory). + +The active client is selected through the ``fdbclients`` alternatives group. +Use ``update-alternatives --display fdbclients`` to inspect it and +``update-alternatives --config fdbclients`` to select a version. The selected +alternative applies consistently to the client tools, library, headers, +pkg-config metadata, and CMake package metadata. Testing your FoundationDB installation ====================================== diff --git a/packaging/multiversion/clients/postinst b/packaging/multiversion/clients/postinst index 7b2e9936fa7..f5997392005 100644 --- a/packaging/multiversion/clients/postinst +++ b/packaging/multiversion/clients/postinst @@ -9,7 +9,7 @@ then fi if [ ! -d "${PKG_CONFIG_DIR}" ] then - mkdir ${PKG_CONFIG_DIR} + mkdir "${PKG_CONFIG_DIR}" fi update-alternatives --install /usr/bin/fdbcli fdbclients /usr/lib/foundationdb-@FDB_VERSION@@FDB_BUILDTIME_STRING@/bin/fdbcli @ALTERNATIVES_PRIORITY@ \ @@ -19,6 +19,7 @@ update-alternatives --install /usr/bin/fdbcli fdbclients /usr/lib/foundationdb-@ --slave /usr/bin/fdbdr fdbdr /usr/lib/foundationdb-@FDB_VERSION@@FDB_BUILDTIME_STRING@/bin/fdbbackup \ --slave /usr/bin/backup_agent backup_agent /usr/lib/foundationdb-@FDB_VERSION@@FDB_BUILDTIME_STRING@/bin/fdbbackup \ --slave /usr/@LIB_DIR@/libfdb_c.so libfdb_c /usr/lib/foundationdb-@FDB_VERSION@@FDB_BUILDTIME_STRING@/lib/libfdb_c.so \ + --slave /usr/@LIB_DIR@/libfdb_c_shim.so libfdb_c_shim /usr/lib/foundationdb-@FDB_VERSION@@FDB_BUILDTIME_STRING@/lib/libfdb_c_shim.so \ --slave /usr/@LIB_DIR@/pkgconfig/foundationdb-client.pc foundationdb-client.pc /usr/lib/foundationdb-@FDB_VERSION@@FDB_BUILDTIME_STRING@/lib/pkgconfig/foundationdb-client.pc \ --slave /usr/@LIB_DIR@/cmake/FoundationDB-Client FoundationDB-ClientConfig /usr/lib/foundationdb-@FDB_VERSION@@FDB_BUILDTIME_STRING@/lib/cmake/FoundationDB-Client \ --slave /usr/include/foundationdb fdb-client-headers /usr/lib/foundationdb-@FDB_VERSION@@FDB_BUILDTIME_STRING@/include/foundationdb diff --git a/packaging/multiversion/clients/prerm b/packaging/multiversion/clients/prerm index 0a87f237544..bc09201eca9 100644 --- a/packaging/multiversion/clients/prerm +++ b/packaging/multiversion/clients/prerm @@ -1,3 +1,7 @@ #!/usr/bin/env bash -update-alternatives --remove fdbclients /usr/lib/foundationdb-@FDB_VERSION@@FDB_BUILDTIME_STRING@/bin/fdbcli +case "${1:-remove}" in + 0|remove) + update-alternatives --remove fdbclients /usr/lib/foundationdb-@FDB_VERSION@@FDB_BUILDTIME_STRING@/bin/fdbcli + ;; +esac From d34596ab1f9f5b52b7b53a07235a727f726bc389 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Mon, 14 Sep 2026 12:24:55 -0700 Subject: [PATCH 052/170] Reset native CDC pop validation baseline after proxy replacement --- fdbserver/workloads/NativeCdcEndToEnd.cpp | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/fdbserver/workloads/NativeCdcEndToEnd.cpp b/fdbserver/workloads/NativeCdcEndToEnd.cpp index f74e7b0fe7d..160aa5ae2d6 100644 --- a/fdbserver/workloads/NativeCdcEndToEnd.cpp +++ b/fdbserver/workloads/NativeCdcEndToEnd.cpp @@ -1142,14 +1142,18 @@ class NativeCdcEndToEndWorkload : public TestWorkload { auto initialProxyStatus = co_await timeoutError(getAssignedProxyStatus(cx, streamId), operationTimeout); bool followedProxyReplacement = proxy->id() != initialProxyStatus.first.id(); updateObservedProxy(*proxy, initialProxyStatus.first); - const CDCProxyBufferStatus initial = initialProxyStatus.second; + CDCProxyBufferStatus initial = initialProxyStatus.second; auto stopped = makeReference>(false); Future requester = requestPopsUntilStopped(cx, stopped); const double deadline = now() + operationTimeout; while (true) { const UID previousProxy = proxy->id(); const CDCProxyBufferStatus status = co_await getCurrentProxyStatus(cx, streamId, proxy); - followedProxyReplacement |= previousProxy != proxy->id(); + if (previousProxy != proxy->id()) { + // Pop counters belong to one proxy instance; require fresh progress after replacement. + initial = status; + followedProxyReplacement = true; + } if (status.popCompletions > initial.popCompletions) { ASSERT_GT(status.popRequests, initial.popRequests); break; From 9ef77b9a991aa40489562940819da08c4b6db11c Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Mon, 14 Sep 2026 13:50:10 -0700 Subject: [PATCH 053/170] Enable integer-division checks and preserve fractional timing --- .clang-tidy | 1 + bindings/c/test/mako/mako.cpp | 6 ++-- bindings/c/test/ryw_benchmark.c | 6 ++-- documentation/sphinx/source/clang-tidy.rst | 4 +-- fdbclient/AsyncFileBlobStore.cpp | 3 +- fdbclient/ClientKnobs.cpp | 12 +++++++- fdbclient/include/fdbclient/NativeAPI.h | 4 +-- fdbrpc/sim2.cpp | 36 +++++++++++++++++++++- fdbserver/clustercontroller/Status.cpp | 2 +- 9 files changed, 61 insertions(+), 13 deletions(-) diff --git a/.clang-tidy b/.clang-tidy index 6f24252a8d0..50f4ab64fb7 100644 --- a/.clang-tidy +++ b/.clang-tidy @@ -9,6 +9,7 @@ Checks: > bugprone-implicit-widening-of-multiplication-result, bugprone-inaccurate-erase, bugprone-infinite-loop, + bugprone-integer-division, bugprone-macro-repeated-side-effects, bugprone-misplaced-widening-cast, bugprone-move-forwarding-reference, diff --git a/bindings/c/test/mako/mako.cpp b/bindings/c/test/mako/mako.cpp index 01859e5300f..a602c88e981 100644 --- a/bindings/c/test/mako/mako.cpp +++ b/bindings/c/test/mako/mako.cpp @@ -2040,12 +2040,12 @@ void printReport(Arguments const& args, double cpu_time_worker_threads = std::accumulate(thread_stats, - thread_stats + args.num_processes * args.num_threads, + thread_stats + static_cast(args.num_processes) * args.num_threads, 0.0, [](double x, const ThreadStatistics& s) { return x + s.getCPUTime(); }); double total_duration_worker_threads = std::accumulate(thread_stats, - thread_stats + args.num_processes * args.num_threads, + thread_stats + static_cast(args.num_processes) * args.num_threads, 0.0, [](double x, const ThreadStatistics& s) { return x + s.getTotalDuration(); }) / (args.num_processes * args.num_threads); // average @@ -2348,7 +2348,7 @@ int statsProcessMain(Arguments const& args, throttle_factor = 1 - (sin_factor * (1.0 - (tpsmin / tpsmax))); break; case TPS_SQUARE: - if (pos < (args.tpsinterval / 2)) { + if (pos < tpsinterval / 2.0) { /* set to max */ throttle_factor = 1.0; } else { diff --git a/bindings/c/test/ryw_benchmark.c b/bindings/c/test/ryw_benchmark.c index 2a20bcd46a8..ba9ac4915cd 100644 --- a/bindings/c/test/ryw_benchmark.c +++ b/bindings/c/test/ryw_benchmark.c @@ -179,7 +179,8 @@ int singleClearGetRange(FDBTransaction* tr, struct ResultSet* rs) { double end = getTime(); insertData(tr); - return 100 * numKeys / 2 / (end - start); + const int keysRead = 100 * numKeys / 2; + return keysRead / (end - start); } int clearRangeGetRange(FDBTransaction* tr, struct ResultSet* rs) { @@ -221,7 +222,8 @@ int clearRangeGetRange(FDBTransaction* tr, struct ResultSet* rs) { double end = getTime(); insertData(tr); - return 100 * numKeys * 3 / 4 / (end - start); + const int keysRead = 100 * numKeys * 3 / 4; + return keysRead / (end - start); } int interleavedSetsGets(FDBTransaction* tr, struct ResultSet* rs) { diff --git a/documentation/sphinx/source/clang-tidy.rst b/documentation/sphinx/source/clang-tidy.rst index c710e440e49..f3494befb78 100644 --- a/documentation/sphinx/source/clang-tidy.rst +++ b/documentation/sphinx/source/clang-tidy.rst @@ -10,11 +10,11 @@ This guide explains how to run ``clang-tidy`` locally so you can fix issues befo What clang-tidy checks ====================== -FoundationDB configures 50 named checks in the ``.clang-tidy`` file at the repository root. The +FoundationDB configures 51 named checks in the ``.clang-tidy`` file at the repository root. The active set depends on the clang-tidy version and can be inspected with ``clang-tidy --list-checks``. The intent is to enable more as we go forward. Here are some example rules: -* **31 Bugprone rules** -- catch potential runtime errors, including mismatched argument comments, obvious infinite loops, chained comparisons, swapped arguments, missed base-class copy construction, repeated macro argument evaluation, near-miss virtual overrides, dangling returned references, and incorrect erase/remove calls +* **32 Bugprone rules** -- catch potential runtime errors, including mismatched argument comments, obvious infinite loops, chained comparisons, swapped arguments, integer division in floating-point calculations, missed base-class copy construction, repeated macro argument evaluation, near-miss virtual overrides, dangling returned references, and incorrect erase/remove calls * **1 C++ Core Guidelines rule** -- catch unsafe captures in coroutine lambdas (``cppcoreguidelines-avoid-capturing-lambda-coroutines``) * **2 Misc rules** -- catch redundant expressions and RAII objects held across coroutine suspension points * **4 Modernize rules** -- encourage modern C++ practices (e.g., ``modernize-use-auto``, ``modernize-use-override``) diff --git a/fdbclient/AsyncFileBlobStore.cpp b/fdbclient/AsyncFileBlobStore.cpp index d2705b7e2b3..e1a1084d1b2 100644 --- a/fdbclient/AsyncFileBlobStore.cpp +++ b/fdbclient/AsyncFileBlobStore.cpp @@ -74,5 +74,6 @@ TEST_CASE("/backup/throttling") { double dur = timer() - ts; int speed = int(total / dur); printf("Speed limit was %d, measured speed was %d\n", limit, speed); - ASSERT(abs(speed - limit) / limit < .01); + // Host scheduling can delay completions; only exceeding the rate limit is an error. + ASSERT(speed <= limit * 1.01); } diff --git a/fdbclient/ClientKnobs.cpp b/fdbclient/ClientKnobs.cpp index c2ccfb7e807..85c69c875e2 100644 --- a/fdbclient/ClientKnobs.cpp +++ b/fdbclient/ClientKnobs.cpp @@ -264,7 +264,7 @@ void ClientKnobs::initialize(Randomize randomize, IsSimulated isSimulated) { init( BACKUP_SIMULATED_LIMIT_BYTES, 1e6 ); if( randomize && buggify() ) BACKUP_SIMULATED_LIMIT_BYTES = 1000; init( BACKUP_GET_RANGE_LIMIT_BYTES, 1e6 ); init( BACKUP_LOCK_BYTES, 1e8 ); - init( BACKUP_RANGE_TIMEOUT, TASKBUCKET_TIMEOUT_VERSIONS/CORE_VERSIONSPERSECOND/2.0 ); + init( BACKUP_RANGE_TIMEOUT, double(TASKBUCKET_TIMEOUT_VERSIONS)/CORE_VERSIONSPERSECOND/2.0 ); init( BACKUP_RANGE_MINWAIT, std::max(1.0, BACKUP_RANGE_TIMEOUT/2.0)); init( BULKDUMP_JOB_TIMEOUT, 3600 * 24 ); // 24 hours - large DBs may take days init( BULKLOAD_JOB_TIMEOUT, 3600 * 24 ); // 24 hours - large DBs may take days @@ -457,3 +457,13 @@ TEST_CASE("/fdbclient/knobs/initialize") { ASSERT_EQ(clientKnobs.TASKBUCKET_TIMEOUT_VERSIONS, initialTaskBucketTimeoutVersions * 2); return Void(); } + +TEST_CASE("/fdbclient/knobs/backupRangeFractionalTimeout") { + ClientKnobs clientKnobs(Randomize::False, IsSimulated::False); + clientKnobs.setKnob("core_versionspersecond", int64_t(1000000)); + clientKnobs.setKnob("taskbucket_timeout_versions", 1500000); + clientKnobs.initialize(Randomize::False, IsSimulated::False); + ASSERT_EQ(clientKnobs.BACKUP_RANGE_TIMEOUT, 0.75); + ASSERT_EQ(clientKnobs.BACKUP_RANGE_MINWAIT, 1.0); + return Void(); +} diff --git a/fdbclient/include/fdbclient/NativeAPI.h b/fdbclient/include/fdbclient/NativeAPI.h index 39daa8c67c6..33a9775818c 100644 --- a/fdbclient/include/fdbclient/NativeAPI.h +++ b/fdbclient/include/fdbclient/NativeAPI.h @@ -576,8 +576,8 @@ inline uint64_t getWriteOperationCost(uint64_t bytes) { if (bytes == 0) { return CLIENT_KNOBS->TAG_THROTTLING_RW_FUNGIBILITY_RATIO * CLIENT_KNOBS->TAG_THROTTLING_PAGE_SIZE; } else { - return CLIENT_KNOBS->TAG_THROTTLING_RW_FUNGIBILITY_RATIO * CLIENT_KNOBS->TAG_THROTTLING_PAGE_SIZE * - ((bytes - 1) / CLIENT_KNOBS->TAG_THROTTLING_PAGE_SIZE + 1); + const uint64_t pages = (bytes - 1) / CLIENT_KNOBS->TAG_THROTTLING_PAGE_SIZE + 1; + return CLIENT_KNOBS->TAG_THROTTLING_RW_FUNGIBILITY_RATIO * CLIENT_KNOBS->TAG_THROTTLING_PAGE_SIZE * pages; } } diff --git a/fdbrpc/sim2.cpp b/fdbrpc/sim2.cpp index 2f8a15a1f0e..4b42adfa9a6 100644 --- a/fdbrpc/sim2.cpp +++ b/fdbrpc/sim2.cpp @@ -38,6 +38,7 @@ #include "flow/IRandom.h" #include "flow/CodeProbe.h" #include "flow/ProtocolVersion.h" +#include "flow/ScopeExit.h" #include "flow/Util.h" #include "flow/IAsyncFile.h" #include "fdbrpc/AsyncFileCached.h" @@ -2968,7 +2969,7 @@ Future waitUntilDiskReady(Reference diskParameters, int64_ if (diskParameters->nextOperation < now()) diskParameters->nextOperation = now(); - diskParameters->nextOperation += (1.0 / diskParameters->iops) + (size / diskParameters->bandwidth); + diskParameters->nextOperation += (1.0 / diskParameters->iops) + (double(size) / diskParameters->bandwidth); double randomLatency; if (sync) { @@ -2980,6 +2981,39 @@ Future waitUntilDiskReady(Reference diskParameters, int64_ return delayUntil(diskParameters->nextOperation + randomLatency); } +TEST_CASE("/fdbrpc/Sim2Disk/fractionalTransferDelay") { + if (!g_network->isSimulated()) { + co_return; + } + + auto diskParameters = makeReference(4, 1024); + const double start = now(); + Future first; + Future second; + { + auto* process = g_simulator->getCurrentProcess(); + const bool failedDisk = process->failedDisk; + const double disabledDuration = g_simulator->connectionFailuresDisableDuration; + auto restore = ScopeExit([process, failedDisk, disabledDuration]() { + process->failedDisk = failedDisk; + g_simulator->connectionFailuresDisableDuration = disabledDuration; + }); + // Exercise disk queueing without changing the surrounding simulation across a yield. + process->failedDisk = false; + g_simulator->connectionFailuresDisableDuration = 0; + first = waitUntilDiskReady(diskParameters, 256); + ASSERT(std::abs(diskParameters->nextOperation - (start + 0.5)) < 1e-8); + second = waitUntilDiskReady(diskParameters, 1536); + ASSERT(std::abs(diskParameters->nextOperation - (start + 2.25)) < 1e-8); + } + ASSERT(!first.isReady()); + ASSERT(!second.isReady()); + co_await first; + ASSERT(now() >= start + 0.5); + co_await second; + ASSERT(now() >= start + 2.25); +} + void enableConnectionFailures(std::string const& context, double duration) { if (g_network->isSimulated()) { g_simulator->connectionFailuresDisableDuration = 0; diff --git a/fdbserver/clustercontroller/Status.cpp b/fdbserver/clustercontroller/Status.cpp index 71af46f3cf8..b0705510584 100644 --- a/fdbserver/clustercontroller/Status.cpp +++ b/fdbserver/clustercontroller/Status.cpp @@ -1739,7 +1739,7 @@ loadConfiguration(Database cx, JsonBuilderArray* messages, std::set } else if (healthyZone.second > tr.getReadVersion().get()) { res.healthyZone = healthyZone.first; res.healthyZoneSeconds = - (healthyZone.second - tr.getReadVersion().get()) / CLIENT_KNOBS->CORE_VERSIONSPERSECOND; + double(healthyZone.second - tr.getReadVersion().get()) / CLIENT_KNOBS->CORE_VERSIONSPERSECOND; } } res.rebalanceDDIgnored = rebalanceDDIgnored.get().present(); From 7a1a0d070f9dccef616570576fe27d4f53937791 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Mon, 14 Sep 2026 13:59:43 -0700 Subject: [PATCH 054/170] Remove added tests and clarify numeric conversions --- bindings/c/test/mako/mako.cpp | 12 ++++----- fdbclient/ClientKnobs.cpp | 12 +-------- fdbrpc/sim2.cpp | 37 ++------------------------ fdbserver/clustercontroller/Status.cpp | 4 +-- 4 files changed, 11 insertions(+), 54 deletions(-) diff --git a/bindings/c/test/mako/mako.cpp b/bindings/c/test/mako/mako.cpp index a602c88e981..0d8f50b0914 100644 --- a/bindings/c/test/mako/mako.cpp +++ b/bindings/c/test/mako/mako.cpp @@ -2037,18 +2037,18 @@ void printReport(Arguments const& args, } const auto warmup_duration_sec = warmup_snapshot.has_value() ? warmup_snapshot->duration_sec : 0.0; const auto measurement_duration_sec = std::max(run_duration_sec - warmup_duration_sec, 1e-9); + const auto num_worker_threads = static_cast(args.num_processes) * args.num_threads; double cpu_time_worker_threads = - std::accumulate(thread_stats, - thread_stats + static_cast(args.num_processes) * args.num_threads, - 0.0, - [](double x, const ThreadStatistics& s) { return x + s.getCPUTime(); }); + std::accumulate(thread_stats, thread_stats + num_worker_threads, 0.0, [](double x, const ThreadStatistics& s) { + return x + s.getCPUTime(); + }); double total_duration_worker_threads = std::accumulate(thread_stats, - thread_stats + static_cast(args.num_processes) * args.num_threads, + thread_stats + num_worker_threads, 0.0, [](double x, const ThreadStatistics& s) { return x + s.getTotalDuration(); }) / - (args.num_processes * args.num_threads); // average + num_worker_threads; // average double cpu_util_worker_threads = 100. * cpu_time_worker_threads / total_duration_worker_threads; diff --git a/fdbclient/ClientKnobs.cpp b/fdbclient/ClientKnobs.cpp index 85c69c875e2..0d08fdbf8fe 100644 --- a/fdbclient/ClientKnobs.cpp +++ b/fdbclient/ClientKnobs.cpp @@ -264,7 +264,7 @@ void ClientKnobs::initialize(Randomize randomize, IsSimulated isSimulated) { init( BACKUP_SIMULATED_LIMIT_BYTES, 1e6 ); if( randomize && buggify() ) BACKUP_SIMULATED_LIMIT_BYTES = 1000; init( BACKUP_GET_RANGE_LIMIT_BYTES, 1e6 ); init( BACKUP_LOCK_BYTES, 1e8 ); - init( BACKUP_RANGE_TIMEOUT, double(TASKBUCKET_TIMEOUT_VERSIONS)/CORE_VERSIONSPERSECOND/2.0 ); + init( BACKUP_RANGE_TIMEOUT, static_cast(TASKBUCKET_TIMEOUT_VERSIONS)/CORE_VERSIONSPERSECOND/2.0 ); init( BACKUP_RANGE_MINWAIT, std::max(1.0, BACKUP_RANGE_TIMEOUT/2.0)); init( BULKDUMP_JOB_TIMEOUT, 3600 * 24 ); // 24 hours - large DBs may take days init( BULKLOAD_JOB_TIMEOUT, 3600 * 24 ); // 24 hours - large DBs may take days @@ -457,13 +457,3 @@ TEST_CASE("/fdbclient/knobs/initialize") { ASSERT_EQ(clientKnobs.TASKBUCKET_TIMEOUT_VERSIONS, initialTaskBucketTimeoutVersions * 2); return Void(); } - -TEST_CASE("/fdbclient/knobs/backupRangeFractionalTimeout") { - ClientKnobs clientKnobs(Randomize::False, IsSimulated::False); - clientKnobs.setKnob("core_versionspersecond", int64_t(1000000)); - clientKnobs.setKnob("taskbucket_timeout_versions", 1500000); - clientKnobs.initialize(Randomize::False, IsSimulated::False); - ASSERT_EQ(clientKnobs.BACKUP_RANGE_TIMEOUT, 0.75); - ASSERT_EQ(clientKnobs.BACKUP_RANGE_MINWAIT, 1.0); - return Void(); -} diff --git a/fdbrpc/sim2.cpp b/fdbrpc/sim2.cpp index 4b42adfa9a6..4eb6d8f4108 100644 --- a/fdbrpc/sim2.cpp +++ b/fdbrpc/sim2.cpp @@ -38,7 +38,6 @@ #include "flow/IRandom.h" #include "flow/CodeProbe.h" #include "flow/ProtocolVersion.h" -#include "flow/ScopeExit.h" #include "flow/Util.h" #include "flow/IAsyncFile.h" #include "fdbrpc/AsyncFileCached.h" @@ -2969,7 +2968,8 @@ Future waitUntilDiskReady(Reference diskParameters, int64_ if (diskParameters->nextOperation < now()) diskParameters->nextOperation = now(); - diskParameters->nextOperation += (1.0 / diskParameters->iops) + (double(size) / diskParameters->bandwidth); + diskParameters->nextOperation += + (1.0 / diskParameters->iops) + (static_cast(size) / diskParameters->bandwidth); double randomLatency; if (sync) { @@ -2981,39 +2981,6 @@ Future waitUntilDiskReady(Reference diskParameters, int64_ return delayUntil(diskParameters->nextOperation + randomLatency); } -TEST_CASE("/fdbrpc/Sim2Disk/fractionalTransferDelay") { - if (!g_network->isSimulated()) { - co_return; - } - - auto diskParameters = makeReference(4, 1024); - const double start = now(); - Future first; - Future second; - { - auto* process = g_simulator->getCurrentProcess(); - const bool failedDisk = process->failedDisk; - const double disabledDuration = g_simulator->connectionFailuresDisableDuration; - auto restore = ScopeExit([process, failedDisk, disabledDuration]() { - process->failedDisk = failedDisk; - g_simulator->connectionFailuresDisableDuration = disabledDuration; - }); - // Exercise disk queueing without changing the surrounding simulation across a yield. - process->failedDisk = false; - g_simulator->connectionFailuresDisableDuration = 0; - first = waitUntilDiskReady(diskParameters, 256); - ASSERT(std::abs(diskParameters->nextOperation - (start + 0.5)) < 1e-8); - second = waitUntilDiskReady(diskParameters, 1536); - ASSERT(std::abs(diskParameters->nextOperation - (start + 2.25)) < 1e-8); - } - ASSERT(!first.isReady()); - ASSERT(!second.isReady()); - co_await first; - ASSERT(now() >= start + 0.5); - co_await second; - ASSERT(now() >= start + 2.25); -} - void enableConnectionFailures(std::string const& context, double duration) { if (g_network->isSimulated()) { g_simulator->connectionFailuresDisableDuration = 0; diff --git a/fdbserver/clustercontroller/Status.cpp b/fdbserver/clustercontroller/Status.cpp index b0705510584..e4e52c0c6b5 100644 --- a/fdbserver/clustercontroller/Status.cpp +++ b/fdbserver/clustercontroller/Status.cpp @@ -1738,8 +1738,8 @@ loadConfiguration(Database cx, JsonBuilderArray* messages, std::set res.healthyZone = healthyZone.first; } else if (healthyZone.second > tr.getReadVersion().get()) { res.healthyZone = healthyZone.first; - res.healthyZoneSeconds = - double(healthyZone.second - tr.getReadVersion().get()) / CLIENT_KNOBS->CORE_VERSIONSPERSECOND; + res.healthyZoneSeconds = static_cast(healthyZone.second - tr.getReadVersion().get()) / + CLIENT_KNOBS->CORE_VERSIONSPERSECOND; } } res.rebalanceDDIgnored = rebalanceDDIgnored.get().present(); From 49dcc1e98fb4a1e6bb56ff20938abe0960a02ed2 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Mon, 14 Sep 2026 14:09:15 -0700 Subject: [PATCH 055/170] Encapsulate coupled client and data distribution state --- fdbbackup/FileDecoder.cpp | 5 +- fdbclient/FileBackupAgent.cpp | 85 ++++++++++++++++++- fdbclient/GlobalConfig.cpp | 10 +-- fdbclient/SpecialKeySpace.cpp | 36 ++++---- fdbclient/include/fdbclient/BackupContainer.h | 11 ++- fdbclient/include/fdbclient/GlobalConfig.h | 15 ++-- .../datadistributor/DDTeamCollection.cpp | 8 +- fdbserver/datadistributor/DDTeamCollection.h | 5 +- fdbserver/datadistributor/StorageWiggler.h | 4 +- 9 files changed, 138 insertions(+), 41 deletions(-) diff --git a/fdbbackup/FileDecoder.cpp b/fdbbackup/FileDecoder.cpp index 8e3a3cab57e..0fa7f94cf20 100644 --- a/fdbbackup/FileDecoder.cpp +++ b/fdbbackup/FileDecoder.cpp @@ -545,10 +545,11 @@ class DecodeProgress { // version batch data that are in the next file. Optional getNextBatch() { for (auto& [version, m] : mutationBlocksByVersion) { - if (m.isComplete()) { + Optional completeMutations = m.getCompleteMutations(); + if (completeMutations.present()) { VersionedMutations vms; vms.version = version; - vms.serializedMutations = m.serializedMutations; + vms.serializedMutations = completeMutations.get().toString(); vms.mutations = fileBackup::decodeMutationLogValue(vms.serializedMutations); TraceEvent("Decode").detail("Version", vms.version).detail("N", vms.mutations.size()); mutationBlocksByVersion.erase(version); diff --git a/fdbclient/FileBackupAgent.cpp b/fdbclient/FileBackupAgent.cpp index d07cc508d1b..80c1b044f88 100644 --- a/fdbclient/FileBackupAgent.cpp +++ b/fdbclient/FileBackupAgent.cpp @@ -47,6 +47,7 @@ #include "FileBackupAgentFileFormat.h" #include "flow/network.h" #include "flow/Trace.h" +#include "flow/UnitTest.h" #include "flow/Util.h" #include @@ -5032,6 +5033,10 @@ void AccumulatedMutations::addChunk(int chunkNumber, const KeyValueRef& kv) { } bool AccumulatedMutations::isComplete() const { + return getCompleteMutations().present(); +} + +Optional AccumulatedMutations::getCompleteMutations() const { if (lastChunkNumber >= 0) { StringRefReader reader(serializedMutations, restore_corrupted_data()); @@ -5041,10 +5046,12 @@ bool AccumulatedMutations::isComplete() const { } uint32_t vLen = reader.consume(); - return vLen == reader.remainder().size(); + if (vLen == reader.remainder().size()) { + return StringRef(serializedMutations); + } } - return false; + return {}; } // Returns true if a complete chunk contains any MutationRefs which intersect with any @@ -5062,6 +5069,77 @@ bool AccumulatedMutations::matchesAnyRange(const RangeMapFilters& filters) const return false; } +TEST_CASE("/backup/AccumulatedMutations") { + BinaryWriter writer(Unversioned()); + writer << uint64_t(0x0FDB00A200090002) << uint32_t(MutationRef::OVERHEAD_BYTES + 8); + writer << uint32_t(MutationRef::SetValue) << uint32_t(3) << uint32_t(5); + writer.serializeBytes("keyvalue"_sr); + Value serialized = writer.toValue(); + const int split = sizeof(uint64_t) + sizeof(uint32_t) + 1; + KeyValueRef firstChunk("chunk0"_sr, serialized.substr(0, split)); + KeyValueRef secondChunk("chunk1"_sr, serialized.substr(split)); + + AccumulatedMutations mutations; + ASSERT(!mutations.getCompleteMutations().present()); + ASSERT(mutations.getChunks().empty()); + mutations.addChunk(0, firstChunk); + ASSERT(!mutations.isComplete()); + ASSERT(!mutations.getCompleteMutations().present()); + ASSERT(mutations.getChunks().size() == 1); + ASSERT(mutations.getChunks().front() == firstChunk); + + mutations.addChunk(1, secondChunk); + Optional complete = mutations.getCompleteMutations(); + ASSERT(complete.present()); + ASSERT(complete.get() == serialized); + ASSERT(mutations.isComplete()); + ASSERT(mutations.getChunks().size() == 2); + ASSERT(mutations.getChunks().back() == secondChunk); + std::vector decoded = decodeMutationLogValue(complete.get()); + ASSERT(decoded.size() == 1); + ASSERT(decoded.front().type == MutationRef::SetValue); + ASSERT(decoded.front().param1 == "key"_sr); + ASSERT(decoded.front().param2 == "value"_sr); + ASSERT(mutations.matchesAnyRange(RangeMapFilters({ singleKeyRange("key"_sr) }))); + ASSERT(!mutations.matchesAnyRange(RangeMapFilters({ singleKeyRange("other"_sr) }))); + + mutations.addChunk(1, secondChunk); + ASSERT(!mutations.getCompleteMutations().present()); + ASSERT(mutations.getChunks().size() == 3); + ASSERT(mutations.getChunks().back() == secondChunk); + + AccumulatedMutations outOfOrder; + outOfOrder.addChunk(1, secondChunk); + outOfOrder.addChunk(0, firstChunk); + outOfOrder.addChunk(1, secondChunk); + ASSERT(!outOfOrder.getCompleteMutations().present()); + ASSERT(outOfOrder.getChunks().size() == 3); + ASSERT(outOfOrder.getChunks().front() == secondChunk); + ASSERT(outOfOrder.getChunks()[1] == firstChunk); + + return Void(); +} + +TEST_CASE("/backup/AccumulatedMutations/invalidHeader") { + auto expectError = [](StringRef payload, int errorCode) { + AccumulatedMutations mutations; + mutations.addChunk(0, KeyValueRef("chunk0"_sr, payload)); + bool caughtError = false; + try { + mutations.getCompleteMutations(); + } catch (Error& e) { + ASSERT(e.code() == errorCode); + caughtError = true; + } + ASSERT(caughtError); + }; + expectError("short"_sr, error_code_restore_corrupted_data); + BinaryWriter writer(Unversioned()); + writer << uint64_t(0x0FDB00A200090001) << uint32_t(0); + expectError(writer.toValue(), error_code_incompatible_protocol_version); + return Void(); +} + bool RangeMapFilters::match(const MutationRef& m) const { if (isSingleKeyMutation((MutationRef::Type)m.type)) { if (match(singleKeyRange(m.param1))) { @@ -5112,7 +5190,8 @@ std::vector filterLogMutationKVPairs(VectorRef data, c // If the mutations are incomplete or match one of the ranges, include in results. if (!m.isComplete() || m.matchesAnyRange(filters)) { - output.insert(output.end(), m.kvs.begin(), m.kvs.end()); + const auto& chunks = m.getChunks(); + output.insert(output.end(), chunks.begin(), chunks.end()); } } diff --git a/fdbclient/GlobalConfig.cpp b/fdbclient/GlobalConfig.cpp index c4df4018c09..590f178f452 100644 --- a/fdbclient/GlobalConfig.cpp +++ b/fdbclient/GlobalConfig.cpp @@ -71,16 +71,16 @@ Key GlobalConfig::prefixedKey(KeyRef key) { return key.withPrefix(SpecialKeySpace::getModuleRange(SpecialKeySpace::MODULE::GLOBALCONFIG).begin); } -Reference GlobalConfig::get(KeyRef name) { +Reference GlobalConfig::get(KeyRef name) { auto it = data.find(name); if (it == data.end()) { - return Reference(); + return Reference(); } return it->second; } -std::map> GlobalConfig::get(KeyRangeRef range) { - std::map> results; +std::map> GlobalConfig::get(KeyRangeRef range) { + std::map> results; for (const auto& [key, value] : data) { if (range.contains(key)) { results[key] = value; @@ -128,7 +128,7 @@ void GlobalConfig::insert(KeyRef key, ValueRef value) { data[stableKey] = makeReference(std::move(arena), std::move(any)); if (callbacks.find(stableKey) != callbacks.end()) { - callbacks[stableKey](data[stableKey]->value); + callbacks[stableKey](data[stableKey]->getValue()); } } catch (Error& e) { TraceEvent(SevWarn, "GlobalConfigTupleParseError").detail("What", e.what()); diff --git a/fdbclient/SpecialKeySpace.cpp b/fdbclient/SpecialKeySpace.cpp index 151113592ae..44f548afae8 100644 --- a/fdbclient/SpecialKeySpace.cpp +++ b/fdbclient/SpecialKeySpace.cpp @@ -1628,25 +1628,27 @@ Future GlobalConfigImpl::getRange(ReadYourWritesTransaction* ryw, RangeResult result; KeyRangeRef modified = KeyRangeRef(kr.begin.removePrefix(getKeyRange().begin), kr.end.removePrefix(getKeyRange().begin)); - std::map> values = ryw->getDatabase()->globalConfig->get(modified); + std::map> values = ryw->getDatabase()->globalConfig->get(modified); for (const auto& [key, config] : values) { Key prefixedKey = key.withPrefix(getKeyRange().begin); - if (config.isValid() && config->value.has_value()) { - if (config->value.type() == typeid(StringRef)) { - result.push_back_deep(result.arena(), - KeyValueRef(prefixedKey, std::any_cast(config->value).toString())); - } else if (config->value.type() == typeid(int64_t)) { - result.push_back_deep(result.arena(), - KeyValueRef(prefixedKey, std::to_string(std::any_cast(config->value)))); - } else if (config->value.type() == typeid(bool)) { - result.push_back_deep(result.arena(), - KeyValueRef(prefixedKey, std::to_string(std::any_cast(config->value)))); - } else if (config->value.type() == typeid(float)) { - result.push_back_deep(result.arena(), - KeyValueRef(prefixedKey, std::to_string(std::any_cast(config->value)))); - } else if (config->value.type() == typeid(double)) { - result.push_back_deep(result.arena(), - KeyValueRef(prefixedKey, std::to_string(std::any_cast(config->value)))); + if (config.isValid() && config->getValue().has_value()) { + if (config->getValue().type() == typeid(StringRef)) { + result.push_back_deep( + result.arena(), KeyValueRef(prefixedKey, std::any_cast(config->getValue()).toString())); + } else if (config->getValue().type() == typeid(int64_t)) { + result.push_back_deep( + result.arena(), + KeyValueRef(prefixedKey, std::to_string(std::any_cast(config->getValue())))); + } else if (config->getValue().type() == typeid(bool)) { + result.push_back_deep( + result.arena(), KeyValueRef(prefixedKey, std::to_string(std::any_cast(config->getValue())))); + } else if (config->getValue().type() == typeid(float)) { + result.push_back_deep( + result.arena(), KeyValueRef(prefixedKey, std::to_string(std::any_cast(config->getValue())))); + } else if (config->getValue().type() == typeid(double)) { + result.push_back_deep( + result.arena(), + KeyValueRef(prefixedKey, std::to_string(std::any_cast(config->getValue())))); } else { ASSERT(false); } diff --git a/fdbclient/include/fdbclient/BackupContainer.h b/fdbclient/include/fdbclient/BackupContainer.h index 1a7c3173440..b8d56d49148 100644 --- a/fdbclient/include/fdbclient/BackupContainer.h +++ b/fdbclient/include/fdbclient/BackupContainer.h @@ -436,7 +436,8 @@ class RangeMapFilters { // Accumulates mutation log value chunks, as both a vector of chunks and as a combined chunk, // in chunk order, and can check the chunk set for completion or intersection with a set // of ranges. -struct AccumulatedMutations { +class AccumulatedMutations { +public: AccumulatedMutations() : lastChunkNumber(-1) {} // Add a KV pair for this mutation chunk set @@ -450,11 +451,19 @@ struct AccumulatedMutations { // that matches the bytes after the header in the combined value in serializedMutations bool isComplete() const; + // Returns the complete serialized payload, or no value if the chunk set is incomplete. + // The returned bytes remain valid until this accumulator is modified or destroyed. + Optional getCompleteMutations() const; + + // The key and value bytes remain owned by the inputs passed to addChunk(). + const std::vector& getChunks() const { return kvs; } + // Returns true if a complete chunk contains any MutationRefs which intersect with any // range in ranges. // It is undefined behavior to run this if isComplete() does not return true. bool matchesAnyRange(const RangeMapFilters& rangeMap) const; +private: std::vector kvs; std::string serializedMutations; int lastChunkNumber; diff --git a/fdbclient/include/fdbclient/GlobalConfig.h b/fdbclient/include/fdbclient/GlobalConfig.h index 3d15d0f1dd9..bd81f8c75e3 100644 --- a/fdbclient/include/fdbclient/GlobalConfig.h +++ b/fdbclient/include/fdbclient/GlobalConfig.h @@ -75,12 +75,15 @@ extern const KeyRef samplingWindow; // Structure used to hold the values stored by global configuration. The arena // is used as memory to store both the key and the value (the value is only // stored in the arena if it is an object; primitives are just copied). -struct ConfigValue : ReferenceCounted { +class ConfigValue : public ReferenceCounted { Arena arena; std::any value; +public: ConfigValue() = default; ConfigValue(Arena&& a, std::any&& v) : arena(a), value(v) {} + + const std::any& getValue() const { return value; } }; class GlobalConfig : NonCopyable { @@ -122,8 +125,8 @@ class GlobalConfig : NonCopyable { // reference which also contains the arena holding the object. As long as // the caller keeps the ConfigValue reference, the value is guaranteed to // be readable. An empty reference is returned if the value does not exist. - Reference get(KeyRef name); - std::map> get(KeyRangeRef range); + Reference get(KeyRef name); + std::map> get(KeyRangeRef range); // For arithmetic value types, returns a copy of the value for the given // key, or the supplied default value if the framework does not know about @@ -134,8 +137,8 @@ class GlobalConfig : NonCopyable { try { auto configValue = get(name); if (configValue.isValid()) { - if (configValue->value.has_value()) { - return std::any_cast(configValue->value); + if (configValue->getValue().has_value()) { + return std::any_cast(configValue->getValue()); } } @@ -195,7 +198,7 @@ class GlobalConfig : NonCopyable { Future _updater; Promise initialized; AsyncTrigger configChanged; - std::unordered_map> data; + std::unordered_map> data; Version lastUpdate; // The key should be a global config string literal key (see the top of this file). std::unordered_map)>> callbacks; diff --git a/fdbserver/datadistributor/DDTeamCollection.cpp b/fdbserver/datadistributor/DDTeamCollection.cpp index dbcec8a9e48..b93baab5d29 100644 --- a/fdbserver/datadistributor/DDTeamCollection.cpp +++ b/fdbserver/datadistributor/DDTeamCollection.cpp @@ -3202,7 +3202,7 @@ class DDTeamCollectionImpl { .detail("NumExistingSS", numExistingSS); } - if (hasHealthyTeam && !tssState->active && tssToRecruit > 0) { + if (hasHealthyTeam && !tssState->isActive() && tssToRecruit > 0) { TraceEvent("TSS_Recruit", self->distributorId) .detail("Stage", "HoldTSS") .detail("Addr", candidateSSAddr.toString()) @@ -3218,7 +3218,7 @@ class DDTeamCollectionImpl { initializeStorage(self, candidateWorker, ddEnabledState, true, tssState)); checkTss = self->initialFailureReactionDelay; } else { - if (tssState->active && tssState->inDataZone(candidateWorker.worker.locality)) { + if (tssState->isActive() && tssState->inDataZone(candidateWorker.worker.locality)) { CODE_PROBE(true, "TSS recruits pair in same dc/datahall"); self->isTssRecruiting = false; TraceEvent("TSS_Recruit", self->distributorId) @@ -3232,7 +3232,7 @@ class DDTeamCollectionImpl { tssState = makeReference(); } else { CODE_PROBE( - tssState->active, + tssState->isActive(), "TSS recruitment skipped potential pair because it's in a different dc/datahall"); self->addActor.send(initializeStorage( self, candidateWorker, ddEnabledState, false, makeReference())); @@ -6430,7 +6430,7 @@ bool DDTeamCollection::exclusionSafetyCheck(std::vector& excludeServerIDs) std::pair DDTeamCollection::getStorageWigglerState() const { if (storageWiggler) { - return { storageWiggler->getWiggleState(), storageWiggler->lastStateChangeTs }; + return storageWiggler->getWiggleStateSnapshot(); } return { StorageWiggler::INVALID, 0.0 }; } diff --git a/fdbserver/datadistributor/DDTeamCollection.h b/fdbserver/datadistributor/DDTeamCollection.h index 03faea63d43..c0a85957623 100644 --- a/fdbserver/datadistributor/DDTeamCollection.h +++ b/fdbserver/datadistributor/DDTeamCollection.h @@ -54,7 +54,7 @@ class TCMachineInfo; class TCMachineTeamInfo; // All state that represents an ongoing tss pair recruitment -struct TSSPairState : ReferenceCounted, NonCopyable { +class TSSPairState : public ReferenceCounted, NonCopyable { Promise>> ssPairInfo; // if set, for ss to pass its id to tss pair once it is successfully recruited Promise tssPairDone; // if set, for tss to pass ss that it was successfully recruited @@ -65,11 +65,14 @@ struct TSSPairState : ReferenceCounted, NonCopyable { bool active; +public: TSSPairState() : active(false) {} explicit TSSPairState(const LocalityData& locality) : dcId(locality.dcId()), dataHallId(locality.dataHallId()), active(true) {} + bool isActive() const { return active; } + bool inDataZone(const LocalityData& locality) const { return locality.dcId() == dcId && locality.dataHallId() == dataHallId; } diff --git a/fdbserver/datadistributor/StorageWiggler.h b/fdbserver/datadistributor/StorageWiggler.h index 5a424d21e0d..f1dbc7b1fe9 100644 --- a/fdbserver/datadistributor/StorageWiggler.h +++ b/fdbserver/datadistributor/StorageWiggler.h @@ -56,10 +56,10 @@ class StorageWiggler : public ReferenceCounted { wiggle_pq; std::unordered_map pq_handles; -public: State wiggleState = INVALID; double lastStateChangeTs = 0.0; // timestamp describes when did the state change +public: explicit StorageWiggler(DDTeamCollection* collection) : teamCollection(collection), stopWiggleSignal(true) {}; // wiggle related actors will quit when this signal is set to true void setStopSignal(bool value) { stopWiggleSignal.set(value); } @@ -80,7 +80,7 @@ class StorageWiggler : public ReferenceCounted { Optional getNextServerId(bool necessaryOnly = true); // next check time to avoid busy loop Future onCheck() const; - State getWiggleState() const { return wiggleState; } + std::pair getWiggleStateSnapshot() const { return { wiggleState, lastStateChangeTs }; } void setWiggleState(State s) { if (wiggleState != s) { wiggleState = s; From 57e5f7103ec3c72a954f5862ac2c41f03df3f691 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Mon, 14 Sep 2026 14:10:38 -0700 Subject: [PATCH 056/170] Generate C API options before clang-tidy checks --- .github/workflows/tidy.yml | 1 + 1 file changed, 1 insertion(+) diff --git a/.github/workflows/tidy.yml b/.github/workflows/tidy.yml index 62cbf2cdf00..eccacfd0a8b 100644 --- a/.github/workflows/tidy.yml +++ b/.github/workflows/tidy.yml @@ -46,6 +46,7 @@ jobs: fdboptions \ ProtocolVersion \ fdb_c_generated \ + fdb_c_options \ fdb-java # all protobuf headers From d339768600ace9826aaf061fe41d114bc05a5670 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Mon, 14 Sep 2026 14:50:05 -0700 Subject: [PATCH 057/170] Remove redundant backup accumulator unit tests --- fdbclient/FileBackupAgent.cpp | 72 ----------------------------------- 1 file changed, 72 deletions(-) diff --git a/fdbclient/FileBackupAgent.cpp b/fdbclient/FileBackupAgent.cpp index 80c1b044f88..de44442f40d 100644 --- a/fdbclient/FileBackupAgent.cpp +++ b/fdbclient/FileBackupAgent.cpp @@ -47,7 +47,6 @@ #include "FileBackupAgentFileFormat.h" #include "flow/network.h" #include "flow/Trace.h" -#include "flow/UnitTest.h" #include "flow/Util.h" #include @@ -5069,77 +5068,6 @@ bool AccumulatedMutations::matchesAnyRange(const RangeMapFilters& filters) const return false; } -TEST_CASE("/backup/AccumulatedMutations") { - BinaryWriter writer(Unversioned()); - writer << uint64_t(0x0FDB00A200090002) << uint32_t(MutationRef::OVERHEAD_BYTES + 8); - writer << uint32_t(MutationRef::SetValue) << uint32_t(3) << uint32_t(5); - writer.serializeBytes("keyvalue"_sr); - Value serialized = writer.toValue(); - const int split = sizeof(uint64_t) + sizeof(uint32_t) + 1; - KeyValueRef firstChunk("chunk0"_sr, serialized.substr(0, split)); - KeyValueRef secondChunk("chunk1"_sr, serialized.substr(split)); - - AccumulatedMutations mutations; - ASSERT(!mutations.getCompleteMutations().present()); - ASSERT(mutations.getChunks().empty()); - mutations.addChunk(0, firstChunk); - ASSERT(!mutations.isComplete()); - ASSERT(!mutations.getCompleteMutations().present()); - ASSERT(mutations.getChunks().size() == 1); - ASSERT(mutations.getChunks().front() == firstChunk); - - mutations.addChunk(1, secondChunk); - Optional complete = mutations.getCompleteMutations(); - ASSERT(complete.present()); - ASSERT(complete.get() == serialized); - ASSERT(mutations.isComplete()); - ASSERT(mutations.getChunks().size() == 2); - ASSERT(mutations.getChunks().back() == secondChunk); - std::vector decoded = decodeMutationLogValue(complete.get()); - ASSERT(decoded.size() == 1); - ASSERT(decoded.front().type == MutationRef::SetValue); - ASSERT(decoded.front().param1 == "key"_sr); - ASSERT(decoded.front().param2 == "value"_sr); - ASSERT(mutations.matchesAnyRange(RangeMapFilters({ singleKeyRange("key"_sr) }))); - ASSERT(!mutations.matchesAnyRange(RangeMapFilters({ singleKeyRange("other"_sr) }))); - - mutations.addChunk(1, secondChunk); - ASSERT(!mutations.getCompleteMutations().present()); - ASSERT(mutations.getChunks().size() == 3); - ASSERT(mutations.getChunks().back() == secondChunk); - - AccumulatedMutations outOfOrder; - outOfOrder.addChunk(1, secondChunk); - outOfOrder.addChunk(0, firstChunk); - outOfOrder.addChunk(1, secondChunk); - ASSERT(!outOfOrder.getCompleteMutations().present()); - ASSERT(outOfOrder.getChunks().size() == 3); - ASSERT(outOfOrder.getChunks().front() == secondChunk); - ASSERT(outOfOrder.getChunks()[1] == firstChunk); - - return Void(); -} - -TEST_CASE("/backup/AccumulatedMutations/invalidHeader") { - auto expectError = [](StringRef payload, int errorCode) { - AccumulatedMutations mutations; - mutations.addChunk(0, KeyValueRef("chunk0"_sr, payload)); - bool caughtError = false; - try { - mutations.getCompleteMutations(); - } catch (Error& e) { - ASSERT(e.code() == errorCode); - caughtError = true; - } - ASSERT(caughtError); - }; - expectError("short"_sr, error_code_restore_corrupted_data); - BinaryWriter writer(Unversioned()); - writer << uint64_t(0x0FDB00A200090001) << uint32_t(0); - expectError(writer.toValue(), error_code_incompatible_protocol_version); - return Void(); -} - bool RangeMapFilters::match(const MutationRef& m) const { if (isSingleKeyMutation((MutationRef::Type)m.type)) { if (match(singleKeyRange(m.param1))) { From 10623f1f47b57d954ab0ab5b8d2f081c19a3dd6a Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Mon, 14 Sep 2026 16:36:42 -0700 Subject: [PATCH 058/170] Add buggified native CDC coverage and fix consume retries --- design/cdc.md | 6 +- fdbclient/NativeCdc.cpp | 42 +++- .../include/fdbclient/CDCProxyInterface.h | 7 +- fdbclient/include/fdbclient/NativeCdc.h | 1 + fdbserver/cdcproxy/CDCProxy.cpp | 238 ++++++++++++------ fdbserver/workloads/NativeCdcEndToEnd.cpp | 21 ++ tests/CMakeLists.txt | 1 + tests/fast/NativeCdcBuggify.toml | 47 ++++ 8 files changed, 277 insertions(+), 86 deletions(-) create mode 100644 tests/fast/NativeCdcBuggify.toml diff --git a/design/cdc.md b/design/cdc.md index d5bfe37da2e..9466d21ecd0 100644 --- a/design/cdc.md +++ b/design/cdc.md @@ -568,7 +568,11 @@ consumer must resume from its last acknowledged checkpoint. An existing native consumer detects the replacement and automatically rewinds an unacknowledged later position to that durable checkpoint. Only one consume RPC may be active for a stream because all consumers would share the same durable acknowledgement -frontier; overlapping logical consumers are rejected. +frontier; overlapping logical consumers are rejected. Native consumers attach a +stable identity to consume RPCs. A transport retry with that identity cancels +the preceding server request before starting another, so a lost connection +does not turn one consumer into two. The identity is an optional trailing RPC +field; requests from older clients retain the strict overlap rejection. When no later version is available, `consume()` is intentionally a client-side long poll. Each server request has a bounded diff --git a/fdbclient/NativeCdc.cpp b/fdbclient/NativeCdc.cpp index c1403266ac6..c6ece7ce0aa 100644 --- a/fdbclient/NativeCdc.cpp +++ b/fdbclient/NativeCdc.cpp @@ -1021,8 +1021,8 @@ Future NativeCdcConsumer::consumeImpl(ReferencecurrentPosition))); + CDCConsumeReply reply = co_await throwErrorOr( + proxy.consume.tryGetReply(CDCConsumeRequest(self->currentPosition, self->consumerId))); if (reply.lastConsumedVersion == self->currentPosition.lastConsumedVersion && reply.mutations.empty()) { // The server lease bounds abandoned long polls. Renew it transparently so the public consume // operation remains a long poll without accumulating server actors after client cancellation. @@ -1103,6 +1103,44 @@ Future NativeCdcConsumer::acknowledge() { return acknowledgeImpl(Reference::addRef(this)); } +namespace { +struct LegacyCDCConsumeRequest { + constexpr static FileIdentifier file_identifier = CDCConsumeRequest::file_identifier; + CDCCursor cursor; + ReplyPromise reply; + + template + void serialize(Ar& ar) { + serializer(ar, cursor, reply); + } +}; +} // namespace + +TEST_CASE("/NativeCDC/ConsumeIdentityCompatibility") { + const Endpoint endpoint({ NetworkAddress(IPAddress(0x01020304), 4500) }, UID(1, 2)); + LegacyCDCConsumeRequest legacy; + legacy.cursor = CDCCursor(7, 100); + legacy.reply = ReplyPromise(endpoint); + const auto oldBytes = ObjectWriter::toValue(legacy, Unversioned()); + auto upgraded = ObjectReader::fromStringRef(oldBytes, Unversioned()); + ASSERT(!upgraded.consumerId.present()); + ASSERT_EQ(upgraded.cursor.streamId, 7); + ASSERT_EQ(upgraded.cursor.lastConsumedVersion, 100); + ASSERT_EQ(upgraded.reply.getEndpoint().token, endpoint.token); + + CDCConsumeRequest request(legacy.cursor, UID(3, 4)); + request.reply = legacy.reply; + const auto bytes = ObjectWriter::toValue(request, Unversioned()); + auto decoded = ObjectReader::fromStringRef(bytes, Unversioned()); + ASSERT_EQ(decoded.consumerId, request.consumerId); + ASSERT_EQ(decoded.cursor.lastConsumedVersion, 100); + auto downgraded = ObjectReader::fromStringRef(bytes, Unversioned()); + ASSERT_EQ(downgraded.cursor.streamId, 7); + ASSERT_EQ(downgraded.cursor.lastConsumedVersion, 100); + ASSERT_EQ(downgraded.reply.getEndpoint().token, endpoint.token); + return Void(); +} + TEST_CASE("/NativeCDC/LifecycleAllocation") { ASSERT(!validNativeCdcTagCount(-1)); ASSERT(!validNativeCdcTagCount(0)); diff --git a/fdbclient/include/fdbclient/CDCProxyInterface.h b/fdbclient/include/fdbclient/CDCProxyInterface.h index 8ed419d0dd6..4b2726c2a36 100644 --- a/fdbclient/include/fdbclient/CDCProxyInterface.h +++ b/fdbclient/include/fdbclient/CDCProxyInterface.h @@ -119,15 +119,18 @@ struct CDCConsumeRequest { constexpr static FileIdentifier file_identifier = 8178243; CDCCursor cursor; ReplyPromise reply; + // Stable across one consumer's RPC retries; absent for legacy or direct callers. + Optional consumerId; CDCConsumeRequest() = default; - explicit CDCConsumeRequest(CDCCursor cursor) : cursor(cursor) {} + explicit CDCConsumeRequest(CDCCursor cursor, Optional consumerId = {}) + : cursor(cursor), consumerId(consumerId) {} bool verify() const { return true; } template void serialize(Ar& ar) { - serializer(ar, cursor, reply); + serializer(ar, cursor, reply, consumerId); } }; diff --git a/fdbclient/include/fdbclient/NativeCdc.h b/fdbclient/include/fdbclient/NativeCdc.h index e63f6ccf1b7..ebde522c30a 100644 --- a/fdbclient/include/fdbclient/NativeCdc.h +++ b/fdbclient/include/fdbclient/NativeCdc.h @@ -35,6 +35,7 @@ class NativeCdcConsumer : public ReferenceCounted { Version knownAvailableThrough = invalidVersion; Version lastAcknowledgedVersion; Optional deliveryProxyId; + UID consumerId = deterministicRandom()->randomUniqueID(); bool operationOutstanding = false; public: diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index ea906b165fb..8d90464fdf7 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -72,6 +72,30 @@ struct CDCTagInterval { : tag(tag), begin(begin), end(end), bufferedThrough(begin - 1) {} }; +// A transport retry supersedes only requests from the same logical consumer. +class CDCConsumeLease : public ReferenceCounted { + Optional consumerId; + Promise superseded; + +public: + explicit CDCConsumeLease(Optional consumerId) : consumerId(consumerId) {} + + bool belongsTo(Optional other) const { + return consumerId.present() && consumerId.get().isValid() && consumerId == other; + } + void supersede() { superseded.send(Void()); } + + Future waitForReply(Future reply) { + // Coroutine parameters can outlive completion while the caller retains its result future. + ScopeExit cancelReply([&reply]() { reply.cancel(); }); + auto result = co_await race(reply, superseded.getFuture()); + if (result.index() == 1) { + throw request_maybe_delivered(); + } + co_return std::get<0>(std::move(result)); + } +}; + // Proxy-owned state for one assigned stream. In-flight actors may retain it after active becomes false. struct CDCBufferedStream : ReferenceCounted { CDCStreamId streamId; @@ -85,7 +109,7 @@ struct CDCBufferedStream : ReferenceCounted { Version bufferedThrough = invalidVersion; int64_t bufferedBytes = 0; int readDemand = 0; - int activeConsumes = 0; + Reference activeConsume; std::vector tagIntervals; std::deque> mutations; AsyncTrigger changed; @@ -454,6 +478,7 @@ class CDCProxy { Future monitorAcknowledgedDataPops(); void reconcileStreams(); Future consume(CDCConsumeRequest request); + Future consumeReply(Reference stream, CDCCursor cursor); Future acknowledge(CDCAckRequest request); Future registerStream(CDCRegisterStreamRequest request); Future removeStream(CDCRemoveStreamRequest request); @@ -1589,95 +1614,102 @@ Future CDCProxy::consume(CDCConsumeRequest request) { if (!stream->active) { throw wrong_shard_server(); } - if (stream->activeConsumes > 0) { - // A stream has one durable acknowledgement frontier, so concurrent logical consumers cannot be - // isolated. Reject overlapping server requests rather than duplicating an entire reply arena. - CODE_PROBE(true, "CDC proxy rejects concurrent consumers for one stream"); - throw client_invalid_operation(); - } - ++stream->activeConsumes; - ScopeExit releaseStreamConsume([stream]() { - ASSERT_GT(stream->activeConsumes, 0); - --stream->activeConsumes; - }); - const CDCStreamReadState metadata = - co_await readCDCStreamState(cx, request.cursor.streamId, id, true, PrioritizeConsume::True); - CODE_PROBE(stream->minVersion < metadata.minVersion, "Native CDC consume reconciles a durable acknowledgement"); - reconcileStreamMinVersion(stream, metadata.minVersion); - if (request.cursor.lastConsumedVersion > stream->bufferedThrough) { - // A cursor is trusted only when this owner has delivered through it or when it is covered by the durable - // acknowledgement watermark used to initialize bufferedThrough. This prevents a fabricated cursor from - // making the proxy retain every intervening tagged mutation while trying to reach an unproven position. - if (request.cursor.lastConsumedVersion > metadata.readVersion) { - CODE_PROBE(true, "CDC proxy rejects a consume cursor beyond its transaction read version"); - } else { - CODE_PROBE(true, "CDC proxy rejects an unproven consume cursor"); + if (stream->activeConsume.isValid()) { + if (!stream->activeConsume->belongsTo(request.consumerId)) { + CODE_PROBE(true, "CDC proxy rejects concurrent consumers for one stream"); + throw client_invalid_operation(); } - throw client_invalid_operation(); + CODE_PROBE(true, "CDC proxy supersedes a consume after transport retry"); + auto previous = stream->activeConsume; + previous->supersede(); + } + auto lease = makeReference(request.consumerId); + stream->activeConsume = lease; + ScopeExit releaseStreamConsume([stream, lease]() { + if (stream->activeConsume == lease) { + stream->activeConsume.clear(); + } + }); + CDCConsumeReply reply = co_await lease->waitForReply(consumeReply(stream, request.cursor)); + request.reply.send(reply); + } catch (Error& e) { + if (e.code() == error_code_actor_cancelled) { + throw; } + request.reply.sendError(e); + } +} - Version begin = request.cursor.lastConsumedVersion == invalidVersion ? stream->minVersion - : request.cursor.lastConsumedVersion + 1; - if (begin < stream->minVersion) { - throw transaction_too_old(); +Future CDCProxy::consumeReply(Reference stream, CDCCursor cursor) { + const CDCStreamReadState metadata = + co_await readCDCStreamState(cx, cursor.streamId, id, true, PrioritizeConsume::True); + CODE_PROBE(stream->minVersion < metadata.minVersion, "Native CDC consume reconciles a durable acknowledgement"); + reconcileStreamMinVersion(stream, metadata.minVersion); + if (cursor.lastConsumedVersion > stream->bufferedThrough) { + // A cursor is trusted only when this owner has delivered through it or when it is covered by the durable + // acknowledgement watermark used to initialize bufferedThrough. This prevents a fabricated cursor from + // making the proxy retain every intervening tagged mutation while trying to reach an unproven position. + if (cursor.lastConsumedVersion > metadata.readVersion) { + CODE_PROBE(true, "CDC proxy rejects a consume cursor beyond its transaction read version"); + } else { + CODE_PROBE(true, "CDC proxy rejects an unproven consume cursor"); } + throw client_invalid_operation(); + } - auto buffered = - co_await race(waitForBufferedVersion(stream, begin), delay(SERVER_KNOBS->CDC_PROXY_CONSUME_POLL_TIMEOUT)); - if (buffered.index() == 1) { - CODE_PROBE(true, "CDC proxy expires an idle consume lease"); - CDCConsumeReply reply; - reply.lastConsumedVersion = request.cursor.lastConsumedVersion; - request.reply.send(reply); - co_return; - } - if (stream->tooOld) { - throw transaction_too_old(); - } - if (stream->bufferLimitExceeded) { - throw server_overloaded(); - } - if (!stream->active) { - throw wrong_shard_server(); - } + Version begin = cursor.lastConsumedVersion == invalidVersion ? stream->minVersion : cursor.lastConsumedVersion + 1; + if (begin < stream->minVersion) { + throw transaction_too_old(); + } + auto buffered = + co_await race(waitForBufferedVersion(stream, begin), delay(SERVER_KNOBS->CDC_PROXY_CONSUME_POLL_TIMEOUT)); + if (buffered.index() == 1) { + CODE_PROBE(true, "CDC proxy expires an idle consume lease"); CDCConsumeReply reply; - CDCConsumeReplySelection selection; - for (const auto& versioned : stream->mutations) { - if (versioned.version < begin) { - continue; - } - if (versioned.version > stream->bufferedThrough) { - break; - } - if (!selectCDCConsumeReplyVersion(&selection, - begin, - versioned.version, - estimatedCDCConsumeVersionBytes(versioned), - SERVER_KNOBS->CDC_PROXY_CONSUME_REPLY_BYTES)) { - break; - } - // Retain the already-accounted stream arena instead of copying mutation payloads for every reply. - reply.arena.dependsOn(versioned.arena()); - reply.mutations.push_back(reply.arena, VersionedMutationsRef(versioned.version, versioned.mutations)); + reply.lastConsumedVersion = cursor.lastConsumedVersion; + co_return reply; + } + if (stream->tooOld) { + throw transaction_too_old(); + } + if (stream->bufferLimitExceeded) { + throw server_overloaded(); + } + if (!stream->active) { + throw wrong_shard_server(); + } + + CDCConsumeReply reply; + CDCConsumeReplySelection selection; + for (const auto& versioned : stream->mutations) { + if (versioned.version < begin) { + continue; } - if (selection.firstVersionTooLarge) { - CODE_PROBE( - true, "CDC proxy rejects one consume version larger than its reply budget", probe::decoration::rare); - TraceEvent(SevWarn, "CDCProxyConsumeVersionExceedsReplyLimit", id) - .detail("StreamId", stream->streamId) - .detail("Version", begin) - .detail("ReplyLimit", SERVER_KNOBS->CDC_PROXY_CONSUME_REPLY_BYTES); - throw server_overloaded(); - } - reply.lastConsumedVersion = selectedCDCConsumeReplyThrough(selection, stream->bufferedThrough); - request.reply.send(reply); - } catch (Error& e) { - if (e.code() == error_code_actor_cancelled) { - throw; + if (versioned.version > stream->bufferedThrough) { + break; } - request.reply.sendError(e); + if (!selectCDCConsumeReplyVersion(&selection, + begin, + versioned.version, + estimatedCDCConsumeVersionBytes(versioned), + SERVER_KNOBS->CDC_PROXY_CONSUME_REPLY_BYTES)) { + break; + } + // Retain the already-accounted stream arena instead of copying mutation payloads for every reply. + reply.arena.dependsOn(versioned.arena()); + reply.mutations.push_back(reply.arena, VersionedMutationsRef(versioned.version, versioned.mutations)); + } + if (selection.firstVersionTooLarge) { + CODE_PROBE(true, "CDC proxy rejects one consume version larger than its reply budget", probe::decoration::rare); + TraceEvent(SevWarn, "CDCProxyConsumeVersionExceedsReplyLimit", id) + .detail("StreamId", stream->streamId) + .detail("Version", begin) + .detail("ReplyLimit", SERVER_KNOBS->CDC_PROXY_CONSUME_REPLY_BYTES); + throw server_overloaded(); } + reply.lastConsumedVersion = selectedCDCConsumeReplyThrough(selection, stream->bufferedThrough); + co_return reply; } Future CDCProxy::acknowledge(CDCAckRequest request) { @@ -1809,7 +1841,7 @@ Future CDCProxy::serveStatusRequests(FutureStreambufferedThrough; streamStatus.bufferedBytes = stream->bufferedBytes; streamStatus.readDemand = stream->readDemand; - streamStatus.activeConsumeRequests = stream->activeConsumes; + streamStatus.activeConsumeRequests = stream->activeConsume.isValid() ? 1 : 0; streamStatus.tooOld = stream->tooOld; streamStatus.bufferLimitExceeded = stream->bufferLimitExceeded; } @@ -2021,6 +2053,50 @@ Future cdcProxyServer(CDCProxyInterface proxy, } } +namespace { +Future blockedConsumeReply(int* active) { + ++*active; + ScopeExit release([active]() { --*active; }); + co_await Future(Never()); + co_return CDCConsumeReply(); +} +} // namespace + +TEST_CASE("/NativeCDC/ConsumeRetryLease") { + const UID consumer(1, 2); + auto lease = makeReference(consumer); + ASSERT(lease->belongsTo(consumer)); + ASSERT(!lease->belongsTo(UID(3, 4))); + ASSERT(!lease->belongsTo({})); + ASSERT(!makeReference(Optional())->belongsTo({})); + ASSERT(!makeReference(UID())->belongsTo(UID())); + + int active = 0; + Future first = lease->waitForReply(blockedConsumeReply(&active)); + ASSERT_EQ(active, 1); + ASSERT(!first.isReady()); + lease->supersede(); + ASSERT(first.isReady() && first.isError()); + ASSERT_EQ(first.getError().code(), error_code_request_maybe_delivered); + ASSERT_EQ(active, 0); + + auto replacement = makeReference(consumer); + Promise delivered; + Future second = replacement->waitForReply(delivered.getFuture()); + CDCConsumeReply reply; + reply.lastConsumedVersion = 123; + delivered.send(reply); + ASSERT(second.isReady() && !second.isError()); + ASSERT_EQ(second.get().lastConsumedVersion, 123); + + auto cancelled = makeReference(consumer); + Future pending = cancelled->waitForReply(blockedConsumeReply(&active)); + ASSERT_EQ(active, 1); + pending.cancel(); + ASSERT_EQ(active, 0); + return Void(); +} + TEST_CASE("/NativeCDC/ProxyMutationFiltering") { const KeyRangeRef keys("c"_sr, "m"_sr); diff --git a/fdbserver/workloads/NativeCdcEndToEnd.cpp b/fdbserver/workloads/NativeCdcEndToEnd.cpp index f74e7b0fe7d..d4549447926 100644 --- a/fdbserver/workloads/NativeCdcEndToEnd.cpp +++ b/fdbserver/workloads/NativeCdcEndToEnd.cpp @@ -1091,6 +1091,14 @@ class NativeCdcEndToEndWorkload : public TestWorkload { ASSERT(overlappingAcknowledgement.isReady() && overlappingAcknowledgement.isError()); ASSERT_EQ(overlappingAcknowledgement.getError().code(), error_code_client_invalid_operation); + // Lose the reply channel while the server still owns the long poll. The public consumer must retry without + // becoming a competing consumer or leaving multiple requests reading the stream. + FlowTransport::transport().resetConnection(proxy->consume.getEndpoint().getPrimaryAddress()); + co_await delay(1.0); + if (idleConsume.isReady()) { + co_await idleConsume; + } + // Assignment publications for unrelated streams used to abandon the client reply without canceling the // corresponding server actor. The active request and read demand must remain bounded at one. for (int i = 0; i < 4; ++i) { @@ -1127,6 +1135,19 @@ class NativeCdcEndToEndWorkload : public TestWorkload { co_await expectConcurrentConsumeRejected(*proxy, currentCursor); first.cancel(); co_await waitForNoActiveConsumes(cx, streamId, proxy); + + // Reproduce the server-side overlap without relying on the timing of a socket reset: both requests belong + // to one consumer, so the retry must supersede the pending metadata read instead of failing exclusivity. + const UID consumerId = deterministicRandom()->randomUniqueID(); + Future> original = + proxy->consume.tryGetReply(CDCConsumeRequest(currentCursor, consumerId)); + Future> retry = + proxy->consume.tryGetReply(CDCConsumeRequest(currentCursor, consumerId)); + const ErrorOr superseded = co_await timeoutError(original, operationTimeout); + ASSERT(superseded.isError()); + ASSERT_EQ(superseded.getError().code(), error_code_request_maybe_delivered); + co_await timeoutError(throwErrorOr(retry), operationTimeout); + co_await waitForNoActiveConsumes(cx, streamId, proxy); } Future requestPopsUntilStopped(Database cx, Reference> stopped) { diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index bf6b1785cac..b54bc98461b 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -225,6 +225,7 @@ if(WITH_PYTHON) add_fdb_test(TEST_FILES fast/RandomUnitTests.toml) add_fdb_test(TEST_FILES fast/RangeLocking.toml) add_fdb_test(TEST_FILES fast/NativeCdcEndToEnd.toml) + add_fdb_test(TEST_FILES fast/NativeCdcBuggify.toml) add_fdb_test(TEST_FILES fast/NativeCdcAssignmentPublication.toml) add_fdb_test(TEST_FILES fast/NativeCdcSharedTag.toml) add_fdb_test(TEST_FILES fast/NativeCdcRetiredSharedTagSnapshot.toml) diff --git a/tests/fast/NativeCdcBuggify.toml b/tests/fast/NativeCdcBuggify.toml new file mode 100644 index 00000000000..733d5d2e996 --- /dev/null +++ b/tests/fast/NativeCdcBuggify.toml @@ -0,0 +1,47 @@ +[configuration] +config = 'double' +singleRegion = true +datacenters = 1 +machineCount = 16 +processesPerMachine = 1 +statelessProcessClassesPerDC = 3 +buggify = true +faultInjection = true + +[[knobs]] +enable_native_cdc = true +# More active streams than tags exercise retention shared with a lagging consumer. +native_cdc_tag_count = 2 +# Up to six TLog replies plus filtered mutations must fit the buggified 10 KB proxy buffer. +maximum_peek_bytes = 1000 + +[[test]] +testTitle = 'NativeCdcBuggify' +useDB = true +waitForQuiescenceEnd = false +timeout = 600 +# Keep failures bounded by Attrition while BUGGIFY varies internal timing and limits. +connectionFailuresDisableDuration = 1000000 +runFailureWorkloads = false + + [[test.workload]] + testName = 'NativeCdcEndToEnd' + initialStreamCount = 6 + minStreamCount = 3 + maxStreamCount = 8 + keyCount = 16 + writesPerRound = 5 + rounds = 20 + drainProbability = 0.25 + delayBetweenRounds = 0.5 + operationTimeout = 500.0 + + [[test.workload]] + testName = 'Attrition' + machinesToKill = 1 + machinesToLeave = 3 + reboot = true + testDuration = 20.0 + waitForVersion = true + allowFaultInjection = false + killDc = false From e5baeba36db6d7b32208af885c2f22889543abc9 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Mon, 14 Sep 2026 17:05:16 -0700 Subject: [PATCH 059/170] Remove CDC unit test additions and relax simulation topology --- fdbclient/NativeCdc.cpp | 38 --------------------------- fdbserver/cdcproxy/CDCProxy.cpp | 44 -------------------------------- tests/fast/NativeCdcBuggify.toml | 6 ----- 3 files changed, 88 deletions(-) diff --git a/fdbclient/NativeCdc.cpp b/fdbclient/NativeCdc.cpp index c6ece7ce0aa..69e7fc77d1b 100644 --- a/fdbclient/NativeCdc.cpp +++ b/fdbclient/NativeCdc.cpp @@ -1103,44 +1103,6 @@ Future NativeCdcConsumer::acknowledge() { return acknowledgeImpl(Reference::addRef(this)); } -namespace { -struct LegacyCDCConsumeRequest { - constexpr static FileIdentifier file_identifier = CDCConsumeRequest::file_identifier; - CDCCursor cursor; - ReplyPromise reply; - - template - void serialize(Ar& ar) { - serializer(ar, cursor, reply); - } -}; -} // namespace - -TEST_CASE("/NativeCDC/ConsumeIdentityCompatibility") { - const Endpoint endpoint({ NetworkAddress(IPAddress(0x01020304), 4500) }, UID(1, 2)); - LegacyCDCConsumeRequest legacy; - legacy.cursor = CDCCursor(7, 100); - legacy.reply = ReplyPromise(endpoint); - const auto oldBytes = ObjectWriter::toValue(legacy, Unversioned()); - auto upgraded = ObjectReader::fromStringRef(oldBytes, Unversioned()); - ASSERT(!upgraded.consumerId.present()); - ASSERT_EQ(upgraded.cursor.streamId, 7); - ASSERT_EQ(upgraded.cursor.lastConsumedVersion, 100); - ASSERT_EQ(upgraded.reply.getEndpoint().token, endpoint.token); - - CDCConsumeRequest request(legacy.cursor, UID(3, 4)); - request.reply = legacy.reply; - const auto bytes = ObjectWriter::toValue(request, Unversioned()); - auto decoded = ObjectReader::fromStringRef(bytes, Unversioned()); - ASSERT_EQ(decoded.consumerId, request.consumerId); - ASSERT_EQ(decoded.cursor.lastConsumedVersion, 100); - auto downgraded = ObjectReader::fromStringRef(bytes, Unversioned()); - ASSERT_EQ(downgraded.cursor.streamId, 7); - ASSERT_EQ(downgraded.cursor.lastConsumedVersion, 100); - ASSERT_EQ(downgraded.reply.getEndpoint().token, endpoint.token); - return Void(); -} - TEST_CASE("/NativeCDC/LifecycleAllocation") { ASSERT(!validNativeCdcTagCount(-1)); ASSERT(!validNativeCdcTagCount(0)); diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index 8d90464fdf7..07fb51b49e3 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -2053,50 +2053,6 @@ Future cdcProxyServer(CDCProxyInterface proxy, } } -namespace { -Future blockedConsumeReply(int* active) { - ++*active; - ScopeExit release([active]() { --*active; }); - co_await Future(Never()); - co_return CDCConsumeReply(); -} -} // namespace - -TEST_CASE("/NativeCDC/ConsumeRetryLease") { - const UID consumer(1, 2); - auto lease = makeReference(consumer); - ASSERT(lease->belongsTo(consumer)); - ASSERT(!lease->belongsTo(UID(3, 4))); - ASSERT(!lease->belongsTo({})); - ASSERT(!makeReference(Optional())->belongsTo({})); - ASSERT(!makeReference(UID())->belongsTo(UID())); - - int active = 0; - Future first = lease->waitForReply(blockedConsumeReply(&active)); - ASSERT_EQ(active, 1); - ASSERT(!first.isReady()); - lease->supersede(); - ASSERT(first.isReady() && first.isError()); - ASSERT_EQ(first.getError().code(), error_code_request_maybe_delivered); - ASSERT_EQ(active, 0); - - auto replacement = makeReference(consumer); - Promise delivered; - Future second = replacement->waitForReply(delivered.getFuture()); - CDCConsumeReply reply; - reply.lastConsumedVersion = 123; - delivered.send(reply); - ASSERT(second.isReady() && !second.isError()); - ASSERT_EQ(second.get().lastConsumedVersion, 123); - - auto cancelled = makeReference(consumer); - Future pending = cancelled->waitForReply(blockedConsumeReply(&active)); - ASSERT_EQ(active, 1); - pending.cancel(); - ASSERT_EQ(active, 0); - return Void(); -} - TEST_CASE("/NativeCDC/ProxyMutationFiltering") { const KeyRangeRef keys("c"_sr, "m"_sr); diff --git a/tests/fast/NativeCdcBuggify.toml b/tests/fast/NativeCdcBuggify.toml index 733d5d2e996..a165ad00424 100644 --- a/tests/fast/NativeCdcBuggify.toml +++ b/tests/fast/NativeCdcBuggify.toml @@ -1,10 +1,4 @@ [configuration] -config = 'double' -singleRegion = true -datacenters = 1 -machineCount = 16 -processesPerMachine = 1 -statelessProcessClassesPerDC = 3 buggify = true faultInjection = true From 81525eaed7147a4e25dadf545bcb435b983a923f Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Mon, 14 Sep 2026 18:56:26 -0700 Subject: [PATCH 060/170] Prevent commit proxy master replies from being starved by commit intake --- fdbserver/commitproxy/CommitProxyServer.cpp | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/fdbserver/commitproxy/CommitProxyServer.cpp b/fdbserver/commitproxy/CommitProxyServer.cpp index 9a3402d3303..5b2ecc821eb 100644 --- a/fdbserver/commitproxy/CommitProxyServer.cpp +++ b/fdbserver/commitproxy/CommitProxyServer.cpp @@ -1752,13 +1752,14 @@ Future postResolution(CommitBatchContext* self) { // @todo probably there is no need to get the (entire) version vector from the sequencer // in this case, and if so, consider adding a flag to the request to tell the sequencer // to not send the version vector information. + // Receive master replies at socket priority so incoming commits cannot starve an already-arrived reply. auto res = co_await race(pProxyCommitData->committedVersion.whenAtLeast( self->commitVersion - SERVER_KNOBS->MAX_READ_TRANSACTION_LIFE_VERSIONS), pProxyCommitData->cx->onProxiesChanged(), pProxyCommitData->master.getLiveCommittedVersion.getReply( GetRawCommittedVersionRequest(waitVersionSpan.context, debugID, invalidVersion), - TaskPriority::GetLiveCommittedVersionReply)); + TaskPriority::ReadSocket)); if (res.index() == 0) { co_await yield(); break; From 84e7b203896984007add2f88fc9b83c01106e577 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Mon, 14 Sep 2026 20:02:10 -0700 Subject: [PATCH 061/170] Clarify TLog durability metrics and soften retirement diagnostics --- fdbserver/clustercontroller/ClusterController.cpp | 9 ++++++++- fdbserver/core/include/fdbserver/core/TLogInterface.h | 3 ++- fdbserver/tlog/TLogServer.cpp | 2 +- 3 files changed, 11 insertions(+), 3 deletions(-) diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index 1ca9bdeae6d..b0acfdee2aa 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -1479,7 +1479,14 @@ void clusterRegisterMaster(ClusterControllerData* self, RegisterMasterRequest co } if (req.recoveryState == RecoveryState::FULLY_RECOVERED) { - ASSERT(req.logSystemConfig.oldTLogs.empty()); + // Retaining old role advertisements must not interrupt an otherwise completed recovery. + if (!req.logSystemConfig.oldTLogs.empty()) { + TraceEvent(SevError, "FullyRecoveredWithOldTLogs", self->id) + .detail("MasterId", req.id) + .detail("RecoveryCount", req.recoveryCount) + .detail("OldLogGenerations", req.logSystemConfig.oldTLogs.size()); + } + ASSERT_WE_THINK(req.logSystemConfig.oldTLogs.empty()); self->db.unfinishedRecoveries = 0; } diff --git a/fdbserver/core/include/fdbserver/core/TLogInterface.h b/fdbserver/core/include/fdbserver/core/TLogInterface.h index d1ff558636a..10ea6a3a716 100644 --- a/fdbserver/core/include/fdbserver/core/TLogInterface.h +++ b/fdbserver/core/include/fdbserver/core/TLogInterface.h @@ -399,7 +399,8 @@ struct TLogQueuingMetricsReply { int64_t instanceID; // changes if bytesDurable and bytesInput reset int64_t bytesDurable{ 0 }, bytesInput{ 0 }; StorageBytes storageBytes; - Version v; // committed version + // Durable known-committed version. Recovery uses this to decide when old log history is safe to discard. + Version v; template void serialize(Ar& ar) { diff --git a/fdbserver/tlog/TLogServer.cpp b/fdbserver/tlog/TLogServer.cpp index 8c65a7be306..aad19f4a5d0 100644 --- a/fdbserver/tlog/TLogServer.cpp +++ b/fdbserver/tlog/TLogServer.cpp @@ -3212,7 +3212,7 @@ void getQueuingMetrics(TLogData* self, Reference logData, TLogQueuingMe reply.bytesInput = self->bytesInput; reply.bytesDurable = self->bytesDurable; reply.storageBytes = self->persistentData->getStorageBytes(); - // FIXME: Add the knownCommittedVersion to this message and change ratekeeper to use that version. + // FIXME: Add a separate knownCommittedVersion field for ratekeeper; v must remain durable for recovery. reply.v = logData->durableKnownCommittedVersion; req.reply.send(reply); } From 42581b6bc339c25ce8def82cb03bd19bd040a516 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Mon, 14 Sep 2026 20:14:10 -0700 Subject: [PATCH 062/170] Bound recovery initialization waits in buggified CDC test --- tests/fast/NativeCdcBuggify.toml | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tests/fast/NativeCdcBuggify.toml b/tests/fast/NativeCdcBuggify.toml index a165ad00424..6d283e0d288 100644 --- a/tests/fast/NativeCdcBuggify.toml +++ b/tests/fast/NativeCdcBuggify.toml @@ -4,6 +4,8 @@ faultInjection = true [[knobs]] enable_native_cdc = true +# Keep injected lost initialization replies within the workload's operation deadline. +cc_recovery_init_req_max_timeout = 30.0 # More active streams than tags exercise retention shared with a lagging consumer. native_cdc_tag_count = 2 # Up to six TLog replies plus filtered mutations must fit the buggified 10 KB proxy buffer. From d416cf879061e82982cfa701607ca5c9c3d5b702 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Mon, 14 Sep 2026 21:03:58 -0700 Subject: [PATCH 063/170] Size CDC peek replies to fit replicated read buffers --- design/cdc.md | 13 +++++++------ fdbserver/cdcproxy/CDCProxy.cpp | 19 ++++++++++++------- tests/fast/NativeCdcBuggify.toml | 2 +- 3 files changed, 20 insertions(+), 14 deletions(-) diff --git a/design/cdc.md b/design/cdc.md index 9466d21ecd0..911142218ac 100644 --- a/design/cdc.md +++ b/design/cdc.md @@ -511,12 +511,13 @@ All raw peek windows and stream buffers owned by one CDC proxy share a retain one separately capped reply arena from every candidate TLog it consults. CDC history cursors therefore disable cross-generation constructor prefetch, report the maximum number of reply arenas one active generation can retain, -and reserve that count times `MAXIMUM_PEEK_BYTES` before issuing a peek. The -proxy marks these delivery cursors with the same per-reply limit; recovery +and cap each reply at the smaller of `MAXIMUM_PEEK_BYTES` and +`CDC_PROXY_BUFFER_BYTES / (retainedReplyCount + 1)`. The proxy reserves the aggregate raw reply +budget plus one reply-sized materialization window before issuing a peek. +It marks these delivery cursors with the same per-reply limit; recovery cursors remain uncapped so that transaction-system replay is not constrained -by a delivery memory knob. The pass also reserves a bounded materialization -window. It retains the aggregate -raw reservation while filtering and copying, then releases it and transfers +by a delivery memory knob. The pass retains the aggregate raw reservation +while filtering and copying, then releases it and transfers only accepted filtered bytes to the stream buffers. Acknowledgement or stream removal releases those retained permits. The usable retained-batch capacity is the configured CDC budget minus this topology-dependent raw reservation; a @@ -539,7 +540,7 @@ that consume fails with `server_overloaded`; operators must configure the budget to hold both the largest raw peek and the largest supported filtered transaction for one stream. -The TLog applies `MAXIMUM_PEEK_BYTES` at complete commit-version boundaries. +The TLog applies the requested reply limit at complete commit-version boundaries. When several individually valid versions would exceed one raw reply, it returns the prefix and leaves the next version for a later peek. It reports an oversized CDC peek only when one complete version cannot fit by itself; a diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index 07fb51b49e3..1bf9cb0d46d 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -428,7 +428,7 @@ class CDCProxy { void clearBufferedMutations(Reference stream); void addBufferedBatch(Reference stream, CDCBufferedBatch batch); void reconcileStreamMinVersion(Reference stream, Version minVersion); - void markTagStreamsBufferLimitExceeded(Reference tag, Version begin); + void markTagStreamsBufferLimitExceeded(Reference tag, Version begin, int replyByteLimit); void markTagStreamsRawReplyBudgetExceeded(Reference tag, Version begin, int64_t retainedReplyCount); void detachStreamFromTags(Reference stream); void deactivateStream(Reference stream); @@ -702,7 +702,7 @@ void CDCProxy::addBufferedBatch(Reference stream, CDCBuffered totalBufferedMutationBytes += batch.bufferedBytes; } -void CDCProxy::markTagStreamsBufferLimitExceeded(Reference tag, Version begin) { +void CDCProxy::markTagStreamsBufferLimitExceeded(Reference tag, Version begin, int replyByteLimit) { for (const CDCStreamId streamId : tag->streamIds) { auto stream = streams.find(streamId); if (stream == streams.end() || !stream->second->active) { @@ -717,7 +717,7 @@ void CDCProxy::markTagStreamsBufferLimitExceeded(Reference tag, .detail("Tag", tag->tag) .detail("StreamId", streamId) .detail("BeginVersion", begin) - .detail("RawPeekLimit", SERVER_KNOBS->MAXIMUM_PEEK_BYTES); + .detail("RawPeekLimit", replyByteLimit); stream->second->bufferLimitExceeded = true; stream->second->changed.trigger(); } @@ -1146,14 +1146,19 @@ Future CDCProxy::bufferTagPass(Reference // CDC ReplayMultiCursor instances disable constructor prefetch, so constructing this cursor cannot issue a peek // before the proxy has reserved memory for every reply arena that its replicated read may retain. Reference cursor = consumer->peekSingle(id, begin, tag->tag, {}); - cursor->setReplyByteLimit(SERVER_KNOBS->MAXIMUM_PEEK_BYTES); const int64_t retainedReplyCount = cursor->getMaxRetainedReplyCount(); - Optional limits = - calculateBufferPassLimits(bufferLimit, SERVER_KNOBS->MAXIMUM_PEEK_BYTES, retainedReplyCount); + // Leave one reply-sized window for filtered mutations instead of rejecting a topology whose maximum-sized + // replies would consume the entire buffer. Every retained raw arena remains covered by the reservation. + const int replyByteLimit = + std::min(SERVER_KNOBS->MAXIMUM_PEEK_BYTES, bufferLimit / (retainedReplyCount + 1)); + Optional limits = calculateBufferPassLimits(bufferLimit, replyByteLimit, retainedReplyCount); if (!limits.present()) { markTagStreamsRawReplyBudgetExceeded(tag, begin, retainedReplyCount); co_return CDCBufferTagPassResult::RETRY; } + cursor->setReplyByteLimit(replyByteLimit); + CODE_PROBE(replyByteLimit < SERVER_KNOBS->MAXIMUM_PEEK_BYTES, + "CDC proxy sizes raw replies to fit replicated reads within its buffer"); const int64_t rawPeekReservation = limits.get().rawReplyBytes; const int64_t hardBufferedBatchLimit = limits.get().hardBufferedBytes; const int64_t preferredBufferedBatch = limits.get().preferredBufferedBytes; @@ -1197,7 +1202,7 @@ Future CDCProxy::bufferTagPass(Reference if (e.code() != error_code_cdc_tlog_peek_reply_too_large) { throw; } - markTagStreamsBufferLimitExceeded(tag, begin); + markTagStreamsBufferLimitExceeded(tag, begin, replyByteLimit); co_return CDCBufferTagPassResult::RETRY; } } diff --git a/tests/fast/NativeCdcBuggify.toml b/tests/fast/NativeCdcBuggify.toml index 6d283e0d288..cfdac454ca5 100644 --- a/tests/fast/NativeCdcBuggify.toml +++ b/tests/fast/NativeCdcBuggify.toml @@ -8,7 +8,7 @@ enable_native_cdc = true cc_recovery_init_req_max_timeout = 30.0 # More active streams than tags exercise retention shared with a lagging consumer. native_cdc_tag_count = 2 -# Up to six TLog replies plus filtered mutations must fit the buggified 10 KB proxy buffer. +# Small replies exercise shared-tag buffering within the buggified 10 KB proxy buffer. maximum_peek_bytes = 1000 [[test]] From 60ea3994f46948c95fc72d287027497042d6ec2d Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 15 Sep 2026 09:19:25 -0700 Subject: [PATCH 064/170] Remove inconclusive native CDC connection reset check --- fdbserver/workloads/NativeCdcEndToEnd.cpp | 8 -------- 1 file changed, 8 deletions(-) diff --git a/fdbserver/workloads/NativeCdcEndToEnd.cpp b/fdbserver/workloads/NativeCdcEndToEnd.cpp index d4549447926..9008fda2d16 100644 --- a/fdbserver/workloads/NativeCdcEndToEnd.cpp +++ b/fdbserver/workloads/NativeCdcEndToEnd.cpp @@ -1091,14 +1091,6 @@ class NativeCdcEndToEndWorkload : public TestWorkload { ASSERT(overlappingAcknowledgement.isReady() && overlappingAcknowledgement.isError()); ASSERT_EQ(overlappingAcknowledgement.getError().code(), error_code_client_invalid_operation); - // Lose the reply channel while the server still owns the long poll. The public consumer must retry without - // becoming a competing consumer or leaving multiple requests reading the stream. - FlowTransport::transport().resetConnection(proxy->consume.getEndpoint().getPrimaryAddress()); - co_await delay(1.0); - if (idleConsume.isReady()) { - co_await idleConsume; - } - // Assignment publications for unrelated streams used to abandon the client reply without canceling the // corresponding server actor. The active request and read demand must remain bounded at one. for (int i = 0; i < 4; ++i) { From 625e2d03080539274990991b751cfdc0056256e1 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 15 Sep 2026 12:07:45 -0700 Subject: [PATCH 065/170] Prevent GRV intake from starving ratekeeper replies --- fdbserver/grvproxy/GrvProxyServer.cpp | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/fdbserver/grvproxy/GrvProxyServer.cpp b/fdbserver/grvproxy/GrvProxyServer.cpp index 8d5c87c939e..6b54b038a4e 100644 --- a/fdbserver/grvproxy/GrvProxyServer.cpp +++ b/fdbserver/grvproxy/GrvProxyServer.cpp @@ -411,13 +411,15 @@ Future getRate(UID myID, nextRequestTimer = Never(); bool detailed = now() - lastDetailedReply > SERVER_KNOBS->DETAILED_METRIC_UPDATE_RATE; + // Receive ratekeeper replies at socket priority so incoming GRVs cannot starve rate-lease updates. reply = brokenPromiseToNever( db->get().ratekeeper.get().getRateInfo.getReply(GetRateInfoRequest(myID, *inTransactionCount, *inBatchTransactionCount, proxyData->version, *transactionTagCounter, - detailed))); + detailed), + TaskPriority::ReadSocket)); transactionTagCounter->clear(); expectingDetailedReply = detailed; } else if (res.index() == 2) { From 5720f3b45c97757a933b05358b9ea2211933c80e Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 15 Sep 2026 12:22:45 -0700 Subject: [PATCH 066/170] Enable three clang-tidy bug checks and fix assignment violations --- .clang-tidy | 3 ++ bindings/c/foundationdb/CppWorkload.h | 4 ++- bindings/c/test/unit/unit_tests.cpp | 33 +++++++++++++++++++ documentation/sphinx/source/clang-tidy.rst | 9 +++-- fdbclient/include/fdbclient/VersionedMap.h | 6 ++++ fdbrpc/FlowTests.cpp | 14 ++++++++ fdbrpc/include/fdbrpc/fdbrpc.h | 6 ++++ fdbserver/kvstore/RadixTree.h | 8 ++--- flow/CoroTests.cpp | 14 ++++++++ flow/Deque.cpp | 38 ++++++++++++++++++++++ flow/include/flow/Deque.h | 6 ++++ flow/include/flow/FastRef.h | 2 ++ flow/include/flow/FlowThread.h | 5 +++ flow/include/flow/IndexedSet.h | 2 ++ flow/include/flow/ThreadHelper.h | 2 ++ flow/include/flow/flow.h | 9 +++++ flow/include/flow/network.h | 2 ++ 17 files changed, 156 insertions(+), 7 deletions(-) diff --git a/.clang-tidy b/.clang-tidy index 50f4ab64fb7..a9dcfba23ab 100644 --- a/.clang-tidy +++ b/.clang-tidy @@ -6,6 +6,8 @@ Checks: > bugprone-chained-comparison, bugprone-copy-constructor-init, bugprone-dangling-handle, + bugprone-fold-init-type, + bugprone-forwarding-reference-overload, bugprone-implicit-widening-of-multiplication-result, bugprone-inaccurate-erase, bugprone-infinite-loop, @@ -30,6 +32,7 @@ Checks: > bugprone-swapped-arguments, bugprone-too-small-loop-variable, bugprone-undefined-memory-manipulation, + bugprone-unhandled-self-assignment, bugprone-unique-ptr-array-mismatch, bugprone-use-after-move, bugprone-virtual-near-miss, diff --git a/bindings/c/foundationdb/CppWorkload.h b/bindings/c/foundationdb/CppWorkload.h index b0a430bfbba..ad91ea24648 100644 --- a/bindings/c/foundationdb/CppWorkload.h +++ b/bindings/c/foundationdb/CppWorkload.h @@ -25,6 +25,7 @@ #include #include #include +#include #ifndef DLLEXPORT #if defined(_MSC_VER) @@ -81,7 +82,8 @@ class GenericPromise { std::shared_ptr impl; public: - template + template , Ptr&&>::value, int>::type = 0> explicit GenericPromise(Ptr&& impl) : impl(std::forward(impl)) {} void send(T val) { impl->send(&val); } }; diff --git a/bindings/c/test/unit/unit_tests.cpp b/bindings/c/test/unit/unit_tests.cpp index 31e2f25c746..8f5baa2a64f 100644 --- a/bindings/c/test/unit/unit_tests.cpp +++ b/bindings/c/test/unit/unit_tests.cpp @@ -23,6 +23,7 @@ #include "fdb_c_options.g.h" #define FDB_USE_LATEST_API_VERSION #include +#include #include #include @@ -51,6 +52,38 @@ #include "fdb_api.hpp" +TEST_CASE("GenericPromise copy and move") { + class RecordingPromise : public FDBPromise { + public: + explicit RecordingPromise(int& value) : value(value) {} + void send(void* source) override { value = *static_cast(source); } + + private: + int& value; + }; + + int value = 0; + auto state = std::make_shared(value); + GenericPromise original(state); + GenericPromise copy(original); + const GenericPromise& constSource = original; + GenericPromise constCopy(constSource); + GenericPromise moved(std::move(copy)); + constCopy.send(17); + CHECK(value == 17); + moved.send(23); + CHECK(value == 23); + original.send(31); + CHECK(value == 31); + CHECK(state.use_count() == 4); + GenericPromise raw(new RecordingPromise(value)); + raw.send(41); + CHECK(value == 41); + GenericPromise unique(std::make_unique(value)); + unique.send(43); + CHECK(value == 43); +} + void fdb_check(fdb_error_t e) { if (e) { std::cerr << fdb_get_error(e) << std::endl; diff --git a/documentation/sphinx/source/clang-tidy.rst b/documentation/sphinx/source/clang-tidy.rst index f3494befb78..66b5963836c 100644 --- a/documentation/sphinx/source/clang-tidy.rst +++ b/documentation/sphinx/source/clang-tidy.rst @@ -10,11 +10,11 @@ This guide explains how to run ``clang-tidy`` locally so you can fix issues befo What clang-tidy checks ====================== -FoundationDB configures 51 named checks in the ``.clang-tidy`` file at the repository root. The +FoundationDB configures 54 named checks in the ``.clang-tidy`` file at the repository root. The active set depends on the clang-tidy version and can be inspected with ``clang-tidy --list-checks``. The intent is to enable more as we go forward. Here are some example rules: -* **32 Bugprone rules** -- catch potential runtime errors, including mismatched argument comments, obvious infinite loops, chained comparisons, swapped arguments, integer division in floating-point calculations, missed base-class copy construction, repeated macro argument evaluation, near-miss virtual overrides, dangling returned references, and incorrect erase/remove calls +* **35 Bugprone rules** -- catch potential runtime errors, including unsafe self-assignment, forwarding constructors that hide copy or move constructors, narrow accumulation initializers, mismatched argument comments, obvious infinite loops, chained comparisons, swapped arguments, integer division in floating-point calculations, missed base-class copy construction, repeated macro argument evaluation, near-miss virtual overrides, dangling returned references, and incorrect erase/remove calls * **1 C++ Core Guidelines rule** -- catch unsafe captures in coroutine lambdas (``cppcoreguidelines-avoid-capturing-lambda-coroutines``) * **2 Misc rules** -- catch redundant expressions and RAII objects held across coroutine suspension points * **4 Modernize rules** -- encourage modern C++ practices (e.g., ``modernize-use-auto``, ``modernize-use-override``) @@ -26,6 +26,11 @@ The intent is to enable more as we go forward. Here are some example rules: ``std::scoped_lock``. These guards must leave scope before a coroutine suspension point; asynchronous locks designed to span suspension are not included. +``bugprone-unhandled-self-assignment`` retains its default restriction to types +with suspicious fields. Reference-counted assignments that acquire the incoming +reference before releasing the old one use documented, check-specific +``NOLINTNEXTLINE`` annotations where the checker cannot recognize their safety. + Basic examples of ``clang-tidy`` style and performance improvement changes: .. code-block:: cpp diff --git a/fdbclient/include/fdbclient/VersionedMap.h b/fdbclient/include/fdbclient/VersionedMap.h index 46b04389757..30a8041f93f 100644 --- a/fdbclient/include/fdbclient/VersionedMap.h +++ b/fdbclient/include/fdbclient/VersionedMap.h @@ -143,6 +143,9 @@ class PTreeFinger { PTreeFinger(PTreeFinger&& f) { *this = f; } PTreeFinger& operator=(PTreeFinger const& f) { + if (this == &f) { + return *this; + } size_ = f.size_; bound_sz_ = f.bound_sz_; std::copy(f.entries_, f.entries_ + size_, entries_); @@ -150,6 +153,9 @@ class PTreeFinger { } PTreeFinger& operator=(PTreeFinger&& f) { + if (this == &f) { + return *this; + } size_ = std::exchange(f.size_, 0); bound_sz_ = f.bound_sz_; std::copy(f.entries_, f.entries_ + size_, entries_); diff --git a/fdbrpc/FlowTests.cpp b/fdbrpc/FlowTests.cpp index d8d89c6761c..17ef8b6c6c1 100644 --- a/fdbrpc/FlowTests.cpp +++ b/fdbrpc/FlowTests.cpp @@ -1758,6 +1758,20 @@ TEST_CASE("/flow/flow/FlowMutex") { using namespace std::chrono_literals; +TEST_CASE("/flow/thread/ThreadFutureStream/selfAssignment") { + ThreadFutureStream stream; + const auto& source = stream; + stream = source; + ASSERT(!stream.isValid()); + ThreadReturnPromiseStream promise; + stream = promise.getFuture(); + stream = source; + ASSERT(promise.getFutureReferenceCount() == 1); + promise.send(42); + int value = co_await stream; + ASSERT(value == 42); +} + TEST_CASE("/flow/thread/ThreadReturnPromiseStream_Simple") { AsyncTaskExecutor exc(1); noUnseed = true; diff --git a/fdbrpc/include/fdbrpc/fdbrpc.h b/fdbrpc/include/fdbrpc/fdbrpc.h index f5924bb8b44..52ee056faab 100644 --- a/fdbrpc/include/fdbrpc/fdbrpc.h +++ b/fdbrpc/include/fdbrpc/fdbrpc.h @@ -181,6 +181,8 @@ class ReplyPromise final : public ComposedIdentifier { networkSender(Uncancellable(), getFuture(), &sav->getRawEndpoint()); } + // Acquiring each incoming reference before releasing its old reference preserves self-assignment. + // NOLINTNEXTLINE(bugprone-unhandled-self-assignment) void operator=(const ReplyPromise& rhs) { if (rhs.sav) rhs.sav->addPromiseRef(); @@ -618,6 +620,8 @@ class ReplyPromiseStream { // client void setByteLimit(int64_t byteLimit) const { queue->acknowledgements.bytesLimit = byteLimit; } + // Acquiring each incoming reference before releasing its old reference preserves self-assignment. + // NOLINTNEXTLINE(bugprone-unhandled-self-assignment) void operator=(const ReplyPromiseStream& rhs) { rhs.queue->addPromiseRef(); if (queue) @@ -920,6 +924,8 @@ class RequestStream { } RequestStream(const RequestStream& rhs) : queue(rhs.queue) { queue->addPromiseRef(); } RequestStream(RequestStream&& rhs) noexcept : queue(rhs.queue) { rhs.queue = 0; } + // Acquiring each incoming reference before releasing its old reference preserves self-assignment. + // NOLINTNEXTLINE(bugprone-unhandled-self-assignment) void operator=(const RequestStream& rhs) { rhs.queue->addPromiseRef(); if (queue) diff --git a/fdbserver/kvstore/RadixTree.h b/fdbserver/kvstore/RadixTree.h index bb56c5d0edb..c2565675ca2 100644 --- a/fdbserver/kvstore/RadixTree.h +++ b/fdbserver/kvstore/RadixTree.h @@ -91,6 +91,9 @@ class radix_tree { node(const node&) = delete; // delete node& operator=(const node& other) { + if (this == &other) { + return *this; + } m_is_leaf = other.m_is_leaf; m_is_inline = other.m_is_inline; m_inline_length = other.m_inline_length; @@ -237,10 +240,7 @@ class radix_tree { iterator() : m_pointee(nullptr) {} iterator(const iterator& r) : m_pointee(r.m_pointee) {} explicit(false) iterator(node* p) : m_pointee(p) {} - iterator& operator=(const iterator& r) { - m_pointee = r.m_pointee; - return *this; - } + iterator& operator=(const iterator& r) = default; ~iterator() = default; const iterator& operator++(); diff --git a/flow/CoroTests.cpp b/flow/CoroTests.cpp index 2a07f680ad8..8a44c996a1e 100644 --- a/flow/CoroTests.cpp +++ b/flow/CoroTests.cpp @@ -1484,6 +1484,20 @@ TEST_CASE("/flow/coro/PromiseStream/move2") { ASSERT(movedTracker.copied == 0); } +TEST_CASE("/flow/coro/FutureStream/selfAssignment") { + FutureStream stream; + const auto& source = stream; + stream = source; + ASSERT(!stream.isValid()); + PromiseStream promise; + stream = promise.getFuture(); + stream = source; + ASSERT(promise.getFutureReferenceCount() == 1); + promise.send(42); + ASSERT(stream.pop() == 42); + return Void(); +} + TEST_CASE("/flow/coro/FutureStream/rvalueAwait") { { PromiseStream stream; diff --git a/flow/Deque.cpp b/flow/Deque.cpp index 5f1fe4f80f7..8541973b5a9 100644 --- a/flow/Deque.cpp +++ b/flow/Deque.cpp @@ -53,6 +53,44 @@ TEST_CASE("/flow/Deque/queue") { return Void(); } +TEST_CASE("/flow/Deque/self_assignment") { + auto copyAssignToSelf = [](Deque& q) { + const auto& source = q; + q = source; + }; + auto moveAssignToSelf = [](Deque& q) { + auto& source = q; + q = std::move(source); + }; + Deque q; + copyAssignToSelf(q); + moveAssignToSelf(q); + ASSERT(q.empty()); + for (int i = 0; i < 6; ++i) { + q.push_back(i); + } + for (int i = 0; i < 4; ++i) { + q.pop_front(); + } + for (int i = 6; i < 10; ++i) { + q.push_back(i); + } + const int* front = &q.front(); + copyAssignToSelf(q); + ASSERT(&q.front() == front); + ASSERT(q.size() == 6); + for (int i = 0; i < q.size(); ++i) { + ASSERT(q[i] == i + 4); + } + moveAssignToSelf(q); + ASSERT(&q.front() == front); + ASSERT(q.size() == 6); + for (int i = 0; i < q.size(); ++i) { + ASSERT(q[i] == i + 4); + } + return Void(); +} + TEST_CASE("/flow/Deque/max_size") { Deque q; for (int i = 0; i < 10; i++) diff --git a/flow/include/flow/Deque.h b/flow/include/flow/Deque.h index 2b664d2f68a..4f130816547 100644 --- a/flow/include/flow/Deque.h +++ b/flow/include/flow/Deque.h @@ -61,6 +61,9 @@ class Deque { } void operator=(Deque const& r) { + if (this == &r) { + return; + } cleanup(); arr = nullptr; @@ -92,6 +95,9 @@ class Deque { } void operator=(Deque&& r) noexcept { + if (this == &r) { + return; + } cleanup(); begin = r.begin; diff --git a/flow/include/flow/FastRef.h b/flow/include/flow/FastRef.h index 1c9fe802d8f..7276df47a14 100644 --- a/flow/include/flow/FastRef.h +++ b/flow/include/flow/FastRef.h @@ -127,6 +127,8 @@ class Reference { if (ptr) delref(ptr); } + // Comparing pointees also handles self-assignment. + // NOLINTNEXTLINE(bugprone-unhandled-self-assignment) Reference& operator=(const Reference& r) { P* oldPtr = ptr; P* newPtr = r.ptr; diff --git a/flow/include/flow/FlowThread.h b/flow/include/flow/FlowThread.h index 62c91f72777..3d038a71221 100644 --- a/flow/include/flow/FlowThread.h +++ b/flow/include/flow/FlowThread.h @@ -218,6 +218,9 @@ class ThreadFutureStream { } void operator=(const ThreadFutureStream& rhs) { + if (this == &rhs) { + return; + } rhs.queue->addFutureRef(); if (queue) queue->delFutureRef(); @@ -271,6 +274,8 @@ class ThreadReturnPromiseStream { return ThreadFutureStream(queue); } + // The incoming queue reference is acquired before the old one is released. + // NOLINTNEXTLINE(bugprone-unhandled-self-assignment) void operator=(const ThreadReturnPromiseStream& rhs) { rhs.queue->addPromiseRef(); if (queue) diff --git a/flow/include/flow/IndexedSet.h b/flow/include/flow/IndexedSet.h index 2dac816aa5c..5d603340b11 100644 --- a/flow/include/flow/IndexedSet.h +++ b/flow/include/flow/IndexedSet.h @@ -372,6 +372,8 @@ class MapPair { template MapPair(Key_&& key, Value_&& value) : key(std::forward(key)), value(std::forward(value)) {} + // Memberwise assignment has the same self-assignment behavior as a defaulted operator. + // NOLINTNEXTLINE(bugprone-unhandled-self-assignment) void operator=(MapPair const& rhs) { key = rhs.key; value = rhs.value; diff --git a/flow/include/flow/ThreadHelper.h b/flow/include/flow/ThreadHelper.h index 762977c0598..1e0e91a12fd 100644 --- a/flow/include/flow/ThreadHelper.h +++ b/flow/include/flow/ThreadHelper.h @@ -615,6 +615,8 @@ class ThreadFuture { if (sav) sav->delref(); } + // Acquiring the incoming reference first keeps the shared state alive during self-assignment. + // NOLINTNEXTLINE(bugprone-unhandled-self-assignment) void operator=(const ThreadFuture& rhs) { if (rhs.sav) rhs.sav->addref(); diff --git a/flow/include/flow/flow.h b/flow/include/flow/flow.h index 86926a3c56d..0db95246dd0 100644 --- a/flow/include/flow/flow.h +++ b/flow/include/flow/flow.h @@ -1012,6 +1012,8 @@ SWIFT_CONFORMS_TO_PROTOCOL(flow_swift.FlowFutureOps) if (sav) sav->delFutureRef(); } + // Acquiring the incoming reference first keeps the shared state alive during self-assignment. + // NOLINTNEXTLINE(bugprone-unhandled-self-assignment) void operator=(const Future& rhs) { if (rhs.sav) rhs.sav->addFutureRef(); @@ -1117,6 +1119,8 @@ class SWIFT_SENDABLE Promise final { sav->delPromiseRef(); } + // Acquiring the incoming reference first keeps the shared state alive during self-assignment. + // NOLINTNEXTLINE(bugprone-unhandled-self-assignment) void operator=(const Promise& rhs) { if (rhs.sav) rhs.sav->addPromiseRef(); @@ -1303,6 +1307,9 @@ class SWIFT_SENDABLE FutureStream { queue->delFutureRef(); } void operator=(const FutureStream& rhs) { + if (this == &rhs) { + return; + } rhs.queue->addFutureRef(); if (queue) queue->delFutureRef(); @@ -1410,6 +1417,8 @@ class SWIFT_SENDABLE PromiseStream { PromiseStream() : queue(new NotifiedQueue(0, 1)) {} PromiseStream(const PromiseStream& rhs) : queue(rhs.queue) { queue->addPromiseRef(); } PromiseStream(PromiseStream&& rhs) noexcept : queue(rhs.queue) { rhs.queue = 0; } + // Acquiring the incoming reference first keeps the shared state alive during self-assignment. + // NOLINTNEXTLINE(bugprone-unhandled-self-assignment) void operator=(const PromiseStream& rhs) { rhs.queue->addPromiseRef(); if (queue) diff --git a/flow/include/flow/network.h b/flow/include/flow/network.h index 4b712a13315..72c68421acc 100644 --- a/flow/include/flow/network.h +++ b/flow/include/flow/network.h @@ -87,6 +87,8 @@ struct NetworkMetrics { } // Since networkBusyness is atomic we need to redefine copy assignment operator + // All fields support self-assignment; the array is copied element by element. + // NOLINTNEXTLINE(bugprone-unhandled-self-assignment) NetworkMetrics& operator=(const NetworkMetrics& rhs) { for (int i = 0; i < SLOW_EVENT_BINS; i++) { countSlowEvents[i] = rhs.countSlowEvents[i]; From 05c54444235ea042e173a540bbc01ecf54e078ac Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 15 Sep 2026 12:28:25 -0700 Subject: [PATCH 067/170] Remove added clang-tidy regression unit tests --- bindings/c/test/unit/unit_tests.cpp | 33 ------------------------- fdbrpc/FlowTests.cpp | 14 ----------- flow/CoroTests.cpp | 14 ----------- flow/Deque.cpp | 38 ----------------------------- 4 files changed, 99 deletions(-) diff --git a/bindings/c/test/unit/unit_tests.cpp b/bindings/c/test/unit/unit_tests.cpp index 8f5baa2a64f..31e2f25c746 100644 --- a/bindings/c/test/unit/unit_tests.cpp +++ b/bindings/c/test/unit/unit_tests.cpp @@ -23,7 +23,6 @@ #include "fdb_c_options.g.h" #define FDB_USE_LATEST_API_VERSION #include -#include #include #include @@ -52,38 +51,6 @@ #include "fdb_api.hpp" -TEST_CASE("GenericPromise copy and move") { - class RecordingPromise : public FDBPromise { - public: - explicit RecordingPromise(int& value) : value(value) {} - void send(void* source) override { value = *static_cast(source); } - - private: - int& value; - }; - - int value = 0; - auto state = std::make_shared(value); - GenericPromise original(state); - GenericPromise copy(original); - const GenericPromise& constSource = original; - GenericPromise constCopy(constSource); - GenericPromise moved(std::move(copy)); - constCopy.send(17); - CHECK(value == 17); - moved.send(23); - CHECK(value == 23); - original.send(31); - CHECK(value == 31); - CHECK(state.use_count() == 4); - GenericPromise raw(new RecordingPromise(value)); - raw.send(41); - CHECK(value == 41); - GenericPromise unique(std::make_unique(value)); - unique.send(43); - CHECK(value == 43); -} - void fdb_check(fdb_error_t e) { if (e) { std::cerr << fdb_get_error(e) << std::endl; diff --git a/fdbrpc/FlowTests.cpp b/fdbrpc/FlowTests.cpp index 17ef8b6c6c1..d8d89c6761c 100644 --- a/fdbrpc/FlowTests.cpp +++ b/fdbrpc/FlowTests.cpp @@ -1758,20 +1758,6 @@ TEST_CASE("/flow/flow/FlowMutex") { using namespace std::chrono_literals; -TEST_CASE("/flow/thread/ThreadFutureStream/selfAssignment") { - ThreadFutureStream stream; - const auto& source = stream; - stream = source; - ASSERT(!stream.isValid()); - ThreadReturnPromiseStream promise; - stream = promise.getFuture(); - stream = source; - ASSERT(promise.getFutureReferenceCount() == 1); - promise.send(42); - int value = co_await stream; - ASSERT(value == 42); -} - TEST_CASE("/flow/thread/ThreadReturnPromiseStream_Simple") { AsyncTaskExecutor exc(1); noUnseed = true; diff --git a/flow/CoroTests.cpp b/flow/CoroTests.cpp index 8a44c996a1e..2a07f680ad8 100644 --- a/flow/CoroTests.cpp +++ b/flow/CoroTests.cpp @@ -1484,20 +1484,6 @@ TEST_CASE("/flow/coro/PromiseStream/move2") { ASSERT(movedTracker.copied == 0); } -TEST_CASE("/flow/coro/FutureStream/selfAssignment") { - FutureStream stream; - const auto& source = stream; - stream = source; - ASSERT(!stream.isValid()); - PromiseStream promise; - stream = promise.getFuture(); - stream = source; - ASSERT(promise.getFutureReferenceCount() == 1); - promise.send(42); - ASSERT(stream.pop() == 42); - return Void(); -} - TEST_CASE("/flow/coro/FutureStream/rvalueAwait") { { PromiseStream stream; diff --git a/flow/Deque.cpp b/flow/Deque.cpp index 8541973b5a9..5f1fe4f80f7 100644 --- a/flow/Deque.cpp +++ b/flow/Deque.cpp @@ -53,44 +53,6 @@ TEST_CASE("/flow/Deque/queue") { return Void(); } -TEST_CASE("/flow/Deque/self_assignment") { - auto copyAssignToSelf = [](Deque& q) { - const auto& source = q; - q = source; - }; - auto moveAssignToSelf = [](Deque& q) { - auto& source = q; - q = std::move(source); - }; - Deque q; - copyAssignToSelf(q); - moveAssignToSelf(q); - ASSERT(q.empty()); - for (int i = 0; i < 6; ++i) { - q.push_back(i); - } - for (int i = 0; i < 4; ++i) { - q.pop_front(); - } - for (int i = 6; i < 10; ++i) { - q.push_back(i); - } - const int* front = &q.front(); - copyAssignToSelf(q); - ASSERT(&q.front() == front); - ASSERT(q.size() == 6); - for (int i = 0; i < q.size(); ++i) { - ASSERT(q[i] == i + 4); - } - moveAssignToSelf(q); - ASSERT(&q.front() == front); - ASSERT(q.size() == 6); - for (int i = 0; i < q.size(); ++i) { - ASSERT(q[i] == i + 4); - } - return Void(); -} - TEST_CASE("/flow/Deque/max_size") { Deque q; for (int i = 0; i < 10; i++) From df56278dd2cc9ff4b42d971f30d0683e6907f627 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 15 Sep 2026 12:51:35 -0700 Subject: [PATCH 068/170] Deduplicate ExtStringRef materialization while preserving ownership --- fdbclient/RYWIterator.cpp | 42 +++++++++++++++++++++++++++++++++++++++ fdbclient/SnapshotCache.h | 24 ++++++++++------------ 2 files changed, 52 insertions(+), 14 deletions(-) diff --git a/fdbclient/RYWIterator.cpp b/fdbclient/RYWIterator.cpp index c420a479c89..426940621df 100644 --- a/fdbclient/RYWIterator.cpp +++ b/fdbclient/RYWIterator.cpp @@ -424,6 +424,48 @@ static int getWriteMapCount(WriteMap* p) { return count; } +TEST_CASE("/fdbclient/ExtStringRef/materialization") { + struct MaterializationCase { + StringRef base; + int padding; + StringRef expected; + }; + const MaterializationCase cases[] = { + { ""_sr, 0, ""_sr }, + { ""_sr, 3, "\x00\x00\x00"_sr }, + { "a\x00z"_sr, 0, "a\x00z"_sr }, + { "a\x00z"_sr, 1, "a\x00z\x00"_sr }, + { "a\x00z"_sr, 3, "a\x00z\x00\x00\x00"_sr }, + }; + for (const auto& test : cases) { + Standalone base(test.base); + ExtStringRef extended(base, test.padding); + Arena ownedArena; + Arena borrowArena; + StringRef owned = extended.toArena(ownedArena); + StringRef borrowedOrOwned = extended.toArenaOrRef(borrowArena); + Standalone standalone = extended.toStandaloneStringRef(); + ASSERT(owned == test.expected); + ASSERT(borrowedOrOwned == test.expected); + ASSERT(standalone == test.expected); + if (test.padding == 0) { + ASSERT(borrowedOrOwned.begin() == base.begin()); + ASSERT(borrowArena.getSize() == 0); + } + if (!base.empty()) { + mutateString(base)[0] = 'b'; + ASSERT(owned == test.expected); + ASSERT(standalone == test.expected); + if (test.padding == 0) { + ASSERT(borrowedOrOwned == base); + } else { + ASSERT(borrowedOrOwned == test.expected); + } + } + } + return Void(); +} + TEST_CASE("/fdbclient/WriteMap/emptiness") { Arena arena = Arena(); WriteMap writes = WriteMap(&arena); diff --git a/fdbclient/SnapshotCache.h b/fdbclient/SnapshotCache.h index cdd85e14332..e5ad588b6f0 100644 --- a/fdbclient/SnapshotCache.h +++ b/fdbclient/SnapshotCache.h @@ -34,21 +34,13 @@ struct ExtStringRef { Standalone toStandaloneStringRef() const { auto s = makeString(size()); - if (!base.empty()) { - memcpy(mutateString(s), base.begin(), base.size()); - } - memset(mutateString(s) + base.size(), 0, extra_zero_bytes); + copyTo(mutateString(s)); return s; }; StringRef toArenaOrRef(Arena& a) const { if (extra_zero_bytes) { - StringRef dest = StringRef(new (a) uint8_t[size()], size()); - if (!base.empty()) { - memcpy(mutateString(dest), base.begin(), base.size()); - } - memset(mutateString(dest) + base.size(), 0, extra_zero_bytes); - return dest; + return toArena(a); } else { return base; } @@ -62,10 +54,7 @@ struct ExtStringRef { StringRef toArena(Arena& a) const { if (extra_zero_bytes) { StringRef dest = StringRef(new (a) uint8_t[size()], size()); - if (!base.empty()) { - memcpy(mutateString(dest), base.begin(), base.size()); - } - memset(mutateString(dest) + base.size(), 0, extra_zero_bytes); + copyTo(mutateString(dest)); return dest; } else { return StringRef(a, base); @@ -115,6 +104,13 @@ struct ExtStringRef { ExtStringRef keyAfter() const { return ExtStringRef(base, extra_zero_bytes + 1); } private: + void copyTo(uint8_t* dest) const { + if (!base.empty()) { + memcpy(dest, base.begin(), base.size()); + } + memset(dest + base.size(), 0, extra_zero_bytes); + } + friend struct Traceable; StringRef base; int extra_zero_bytes; From cb993c451a1c08bb90fa3a9002262d9c8a221207 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 15 Sep 2026 12:53:48 -0700 Subject: [PATCH 069/170] Honor assertion suppression and improve runtime diagnostics --- fdbrpc/AsyncFileEncrypted.cpp | 4 ++-- fdbrpc/FailureMonitor.cpp | 2 +- fdbrpc/FlowTransport.cpp | 2 +- fdbrpc/HealthMonitor.cpp | 2 +- flow/Error.cpp | 38 +++++++++++++++++++++++++++++++++++ flow/Net2Packet.cpp | 4 ++-- flow/include/flow/Error.h | 2 +- 7 files changed, 46 insertions(+), 8 deletions(-) diff --git a/fdbrpc/AsyncFileEncrypted.cpp b/fdbrpc/AsyncFileEncrypted.cpp index d70abbf9b1d..2bd9c193e4e 100644 --- a/fdbrpc/AsyncFileEncrypted.cpp +++ b/fdbrpc/AsyncFileEncrypted.cpp @@ -152,7 +152,7 @@ class AsyncFileEncryptedImpl { AsyncFileEncrypted::AsyncFileEncrypted(Reference file, Mode mode, int encryptionBlockSize) : file(file), mode(mode), currentBlock(0), encryptionBlockSize(encryptionBlockSize) { - ASSERT(encryptionBlockSize > 0); + ASSERT_GT(encryptionBlockSize, 0); firstBlockIV = AsyncFileEncryptedImpl::getFirstBlockIV(file->getFilename()); if (mode == Mode::APPEND_ONLY) { writeBuffer = std::vector(encryptionBlockSize, 0); @@ -173,7 +173,7 @@ int64_t AsyncFileEncrypted::rawToLogicalSize(int64_t rawSize, int blockSize) { const int64_t trailing = rawSize % rawBlockSize; int64_t logical = fullBlocks * blockSize; if (trailing > 0) { - ASSERT(trailing > GCM_TAG_LEN); + ASSERT_GT(trailing, GCM_TAG_LEN); logical += trailing - GCM_TAG_LEN; } return logical; diff --git a/fdbrpc/FailureMonitor.cpp b/fdbrpc/FailureMonitor.cpp index 75579ec0ed9..d8114082bac 100644 --- a/fdbrpc/FailureMonitor.cpp +++ b/fdbrpc/FailureMonitor.cpp @@ -65,7 +65,7 @@ Future IFailureMonitor::onStateEqual(Endpoint const& endpoint, FailureStat } Future IFailureMonitor::onFailedFor(Endpoint const& endpoint, double sustainedFailureDuration, double slope) { - ASSERT(slope < 1.0); + ASSERT_LT(slope, 1.0); return waitForContinuousFailure(this, endpoint, sustainedFailureDuration, slope); } diff --git a/fdbrpc/FlowTransport.cpp b/fdbrpc/FlowTransport.cpp index ec217c7c35e..ce5b50d0652 100644 --- a/fdbrpc/FlowTransport.cpp +++ b/fdbrpc/FlowTransport.cpp @@ -150,7 +150,7 @@ void EndpointMap::realloc() { void EndpointMap::insertWellKnown(NetworkMessageReceiver* r, const Endpoint::Token& token, TaskPriority priority) { const auto index = token.second(); - ASSERT(index < uint64_t(wellKnownEndpointCount)); + ASSERT_LT(index, uint64_t(wellKnownEndpointCount)); ASSERT(data[index].receiver == nullptr); data[index].receiver = r; data[index].token() = diff --git a/fdbrpc/HealthMonitor.cpp b/fdbrpc/HealthMonitor.cpp index 95527e3b643..3ca63736878 100644 --- a/fdbrpc/HealthMonitor.cpp +++ b/fdbrpc/HealthMonitor.cpp @@ -32,7 +32,7 @@ void HealthMonitor::purgeOutdatedHistory() { if (p.first < now() - FLOW_KNOBS->HEALTH_MONITOR_CLIENT_REQUEST_INTERVAL_SECS) { auto& count = peerClosedNum[p.second]; --count; - ASSERT(count >= 0); + ASSERT_GE(count, 0); if (count == 0) { peerClosedNum.erase(p.second); } diff --git a/flow/Error.cpp b/flow/Error.cpp index ffb4d34b19e..3e3bb4a527b 100644 --- a/flow/Error.cpp +++ b/flow/Error.cpp @@ -18,11 +18,16 @@ * limitations under the License. */ +#include "flow/BooleanParam.h" #include "flow/Error.h" #include "flow/Knobs.h" +#include "flow/ScopeExit.h" #include "flow/Trace.h" #include "flow/UnitTest.h" +FDB_BOOLEAN_PARAM(Randomize); +FDB_BOOLEAN_PARAM(IsSimulated); + bool g_crashOnError = false; #define DEBUG_ERROR 0 @@ -234,3 +239,36 @@ TEST_CASE("/flow/AssertTest") { return Void(); } + +TEST_CASE("/flow/AssertTest/DisableAsserts") { + FlowKnobs knobs(Randomize::False, IsSimulated::False); + const FlowKnobs* previousKnobs = FLOW_KNOBS; + auto restoreKnobs = ScopeExit([previousKnobs]() { FLOW_KNOBS = previousKnobs; }); + FLOW_KNOBS = &knobs; + + for (bool disableAll : { false, true }) { + auto disableAssertion = [&](int line) { + knobs.DISABLE_ASSERTS = disableAll ? -1 : line; + UNSTOPPABLE_ASSERT(isAssertDisabled(line)); + UNSTOPPABLE_ASSERT(isAssertDisabled(line + 1) == disableAll); + }; + disableAssertion(__LINE__ + 1); + ASSERT_EQ(1, 2); + disableAssertion(__LINE__ + 1); + ASSERT_NE(1, 1); + disableAssertion(__LINE__ + 1); + ASSERT_LT(2, 1); + disableAssertion(__LINE__ + 1); + ASSERT_LE(2, 1); + disableAssertion(__LINE__ + 1); + ASSERT_GT(1, 2); + disableAssertion(__LINE__ + 1); + ASSERT_GE(1, 2); + } + + knobs.DISABLE_ASSERTS = -1; + int evaluations = 0; + ASSERT_EQ(++evaluations, 0); + UNSTOPPABLE_ASSERT(evaluations == 1); + return Void(); +} diff --git a/flow/Net2Packet.cpp b/flow/Net2Packet.cpp index efa77e115d5..f73c11893cf 100644 --- a/flow/Net2Packet.cpp +++ b/flow/Net2Packet.cpp @@ -147,14 +147,14 @@ void UnsentPacketQueue::sent(int bytes) { if (b->bytes_sent + bytes <= b->bytes_written && (b->bytes_sent + bytes != b->bytes_written || (!b->next && b->bytes_unwritten()))) { b->bytes_sent += bytes; - ASSERT(b->bytes_sent <= b->size()); + ASSERT_LE(b->bytes_sent, b->size()); break; } // We've sent an entire buffer bytes -= b->bytes_written - b->bytes_sent; b->bytes_sent = b->bytes_written; - ASSERT(b->bytes_written <= b->size()); + ASSERT_LE(b->bytes_written, b->size()); double queue_time = now() - b->enqueue_time; sendQueueLatencyHistogram->sampleSeconds(queue_time); unsent_first = b->nextPacketBuffer(); diff --git a/flow/include/flow/Error.h b/flow/include/flow/Error.h index 028a7826222..b8df273e006 100644 --- a/flow/include/flow/Error.h +++ b/flow/include/flow/Error.h @@ -159,7 +159,7 @@ void assert_impl(char const* a_nm, bool (*compare)(T const&, U const&), char const* file, int line) { - if (!compare(a, b)) { + if (!(compare(a, b) || isAssertDisabled(line))) { throw internal_error_impl(a_nm, Traceable::toString(a), opName, b_nm, Traceable::toString(b), file, line); } } From 015c9d9a6e4e57baa9d782445036177aa8d0ce4b Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 15 Sep 2026 12:51:03 -0700 Subject: [PATCH 070/170] Preserve tag throttle state across moves and expiration --- fdbserver/ratekeeper/CMakeLists.txt | 3 + .../ratekeeper/RkTagThrottleCollection.cpp | 74 +++++++++++++++---- .../ratekeeper/RkTagThrottleCollection.h | 4 +- 3 files changed, 65 insertions(+), 16 deletions(-) diff --git a/fdbserver/ratekeeper/CMakeLists.txt b/fdbserver/ratekeeper/CMakeLists.txt index 317cc8276c8..ed6e8261136 100644 --- a/fdbserver/ratekeeper/CMakeLists.txt +++ b/fdbserver/ratekeeper/CMakeLists.txt @@ -4,6 +4,9 @@ add_flow_target(STATIC_LIBRARY NAME fdbserver_ratekeeper SRCS ${FDBSERVER_RATEKE add_fdbserver_link_test(fdbserver_ratekeeperlinktest fdbserver_ratekeeper fdbserver_core) +add_fdbserver_unit_test(fdbserver_ratekeeper_test ratekeeper + fdbserver_ratekeeper + fdbserver_core) configure_fdbserver_common_includes(fdbserver_ratekeeper) target_include_directories(fdbserver_ratekeeper diff --git a/fdbserver/ratekeeper/RkTagThrottleCollection.cpp b/fdbserver/ratekeeper/RkTagThrottleCollection.cpp index 27022ef68fe..1b3491adf1a 100644 --- a/fdbserver/ratekeeper/RkTagThrottleCollection.cpp +++ b/fdbserver/ratekeeper/RkTagThrottleCollection.cpp @@ -21,6 +21,7 @@ #include "fdbserver/core/Knobs.h" #include "RkTagThrottleCollection.h" +#include "flow/UnitTest.h" double RkTagThrottleCollection::RkTagThrottleData::getTargetRate(Optional requestRate) const { if (limits.tpsRate == 0.0 || !requestRate.present() || requestRate.get() == 0.0 || !rateSet) { @@ -54,19 +55,6 @@ Optional RkTagThrottleCollection::RkTagThrottleData::updateAndGetClientR } } -RkTagThrottleCollection::RkTagThrottleCollection(RkTagThrottleCollection&& other) { - autoThrottledTags = std::move(other.autoThrottledTags); - manualThrottledTags = std::move(other.manualThrottledTags); - tagData = std::move(other.tagData); -} - -RkTagThrottleCollection& RkTagThrottleCollection::RkTagThrottleCollection::operator=(RkTagThrottleCollection&& other) { - autoThrottledTags = std::move(other.autoThrottledTags); - manualThrottledTags = std::move(other.manualThrottledTags); - tagData = std::move(other.tagData); - return *this; -} - double RkTagThrottleCollection::computeTargetTpsRate(double currentBusyness, double targetBusyness, double requestRate) { @@ -250,7 +238,6 @@ PrioritizedTransactionTagMap RkTagThrottleCollection::g if (manualItr->second.empty()) { CODE_PROBE(true, "All manual throttles expired"); manualThrottledTags.erase(manualItr); - break; } } @@ -359,3 +346,62 @@ void RkTagThrottleCollection::incrementBusyTagCount(TagThrottledReason reason) { TraceEvent(SevWarn, "UnsetTagThrottledReason"); } } + +TEST_CASE("/fdbserver/ratekeeper/TagThrottleCollection/MoveConstruction") { + RkTagThrottleCollection source; + source.incrementBusyTagCount(TagThrottledReason::BUSY_READ); + source.incrementBusyTagCount(TagThrottledReason::BUSY_WRITE); + source.incrementBusyTagCount(TagThrottledReason::BUSY_WRITE); + + RkTagThrottleCollection destination(std::move(source)); + ASSERT_EQ(destination.getBusyReadTagCount(), 1); + ASSERT_EQ(destination.getBusyWriteTagCount(), 2); + co_return; +} + +TEST_CASE("/fdbserver/ratekeeper/TagThrottleCollection/MoveAssignment") { + RkTagThrottleCollection source; + source.incrementBusyTagCount(TagThrottledReason::BUSY_READ); + source.incrementBusyTagCount(TagThrottledReason::BUSY_WRITE); + source.incrementBusyTagCount(TagThrottledReason::BUSY_WRITE); + RkTagThrottleCollection destination; + destination.incrementBusyTagCount(TagThrottledReason::BUSY_READ); + destination.incrementBusyTagCount(TagThrottledReason::BUSY_READ); + + destination = std::move(source); + ASSERT_EQ(destination.getBusyReadTagCount(), 1); + ASSERT_EQ(destination.getBusyWriteTagCount(), 2); + + destination = RkTagThrottleCollection(); + ASSERT_EQ(destination.getBusyReadTagCount(), 0); + ASSERT_EQ(destination.getBusyWriteTagCount(), 0); + co_return; +} + +TEST_CASE("/fdbserver/ratekeeper/TagThrottleCollection/ExpiredManualPreservesActiveRates") { + RkTagThrottleCollection throttles; + const double expiration = now() + 1.0; + const double activeExpiration = now() + 3600.0; + const TransactionTag firstTag = "first"_sr; + const TransactionTag secondTag = "second"_sr; + const TransactionTag manualTag = "manual"_sr; + + // An expired manual entry on each tag makes an early exit skip the active + // auto throttle regardless of hash iteration order. + for (const auto& tag : { firstTag, secondTag }) { + throttles.manualThrottleTag(UID(), tag, TransactionPriority::DEFAULT, 10.0, expiration, {}); + } + ASSERT(throttles.autoThrottleTag(UID(), firstTag, 0, 0.0, activeExpiration).present()); + throttles.manualThrottleTag(UID(), manualTag, TransactionPriority::DEFAULT, 20.0, activeExpiration, {}); + + co_await delay(2.0); + const auto rates = throttles.getClientRates(true); + ASSERT_EQ(throttles.manualThrottleCount(), 1); + ASSERT_EQ(throttles.autoThrottleCount(), 1); + ASSERT_EQ(rates.at(TransactionPriority::DEFAULT).size(), 2); + ASSERT_EQ(rates.at(TransactionPriority::BATCH).size(), 2); + ASSERT_EQ(rates.at(TransactionPriority::DEFAULT).at(firstTag).tpsRate, 0.0); + ASSERT_EQ(rates.at(TransactionPriority::BATCH).at(firstTag).tpsRate, 0.0); + ASSERT_EQ(rates.at(TransactionPriority::DEFAULT).at(manualTag).tpsRate, 20.0); + co_return; +} diff --git a/fdbserver/ratekeeper/RkTagThrottleCollection.h b/fdbserver/ratekeeper/RkTagThrottleCollection.h index 803fd249cd7..d20267c1ca3 100644 --- a/fdbserver/ratekeeper/RkTagThrottleCollection.h +++ b/fdbserver/ratekeeper/RkTagThrottleCollection.h @@ -55,8 +55,8 @@ class RkTagThrottleCollection : NonCopyable { public: RkTagThrottleCollection() = default; - RkTagThrottleCollection(RkTagThrottleCollection&& other); - RkTagThrottleCollection& operator=(RkTagThrottleCollection&& other); + RkTagThrottleCollection(RkTagThrottleCollection&& other) = default; + RkTagThrottleCollection& operator=(RkTagThrottleCollection&& other) = default; Optional autoThrottleTag(UID id, TransactionTag const& tag, From 43c5ad422c28a736b71c5cc3393ba8e95461e1be Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 15 Sep 2026 13:02:04 -0700 Subject: [PATCH 071/170] Remove comparison assertion suppression unit test --- flow/Error.cpp | 38 -------------------------------------- 1 file changed, 38 deletions(-) diff --git a/flow/Error.cpp b/flow/Error.cpp index 3e3bb4a527b..ffb4d34b19e 100644 --- a/flow/Error.cpp +++ b/flow/Error.cpp @@ -18,16 +18,11 @@ * limitations under the License. */ -#include "flow/BooleanParam.h" #include "flow/Error.h" #include "flow/Knobs.h" -#include "flow/ScopeExit.h" #include "flow/Trace.h" #include "flow/UnitTest.h" -FDB_BOOLEAN_PARAM(Randomize); -FDB_BOOLEAN_PARAM(IsSimulated); - bool g_crashOnError = false; #define DEBUG_ERROR 0 @@ -239,36 +234,3 @@ TEST_CASE("/flow/AssertTest") { return Void(); } - -TEST_CASE("/flow/AssertTest/DisableAsserts") { - FlowKnobs knobs(Randomize::False, IsSimulated::False); - const FlowKnobs* previousKnobs = FLOW_KNOBS; - auto restoreKnobs = ScopeExit([previousKnobs]() { FLOW_KNOBS = previousKnobs; }); - FLOW_KNOBS = &knobs; - - for (bool disableAll : { false, true }) { - auto disableAssertion = [&](int line) { - knobs.DISABLE_ASSERTS = disableAll ? -1 : line; - UNSTOPPABLE_ASSERT(isAssertDisabled(line)); - UNSTOPPABLE_ASSERT(isAssertDisabled(line + 1) == disableAll); - }; - disableAssertion(__LINE__ + 1); - ASSERT_EQ(1, 2); - disableAssertion(__LINE__ + 1); - ASSERT_NE(1, 1); - disableAssertion(__LINE__ + 1); - ASSERT_LT(2, 1); - disableAssertion(__LINE__ + 1); - ASSERT_LE(2, 1); - disableAssertion(__LINE__ + 1); - ASSERT_GT(1, 2); - disableAssertion(__LINE__ + 1); - ASSERT_GE(1, 2); - } - - knobs.DISABLE_ASSERTS = -1; - int evaluations = 0; - ASSERT_EQ(++evaluations, 0); - UNSTOPPABLE_ASSERT(evaluations == 1); - return Void(); -} From 121add0d80b0958f9749fb45c86ec1f13875d6a3 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 15 Sep 2026 13:13:36 -0700 Subject: [PATCH 072/170] Improve more runtime assertions and mark failures unlikely --- fdbrpc/AsyncFileCached.cpp | 2 +- fdbrpc/AsyncFileKAIO.h | 2 +- fdbrpc/FlowTransport.cpp | 2 +- fdbrpc/HTTP.cpp | 2 +- fdbrpc/include/fdbrpc/AsyncFileCached.h | 8 ++++---- fdbrpc/include/fdbrpc/AsyncFileReadAhead.h | 4 ++-- fdbrpc/include/fdbrpc/fdbrpc.h | 2 +- flow/Histogram.cpp | 2 +- flow/Net2.cpp | 8 ++++---- flow/TLSConfig.cpp | 2 +- flow/flow.cpp | 6 +++--- flow/include/flow/Error.h | 2 +- 12 files changed, 21 insertions(+), 21 deletions(-) diff --git a/fdbrpc/AsyncFileCached.cpp b/fdbrpc/AsyncFileCached.cpp index b8b9b4ce967..20d36127525 100644 --- a/fdbrpc/AsyncFileCached.cpp +++ b/fdbrpc/AsyncFileCached.cpp @@ -265,7 +265,7 @@ Future AsyncFileCached::flush() { if (!f.isReady()) i++; } - ASSERT(flushable.size() <= debug_count); + ASSERT_LE(flushable.size(), debug_count); return waitForAll(unflushed); } diff --git a/fdbrpc/AsyncFileKAIO.h b/fdbrpc/AsyncFileKAIO.h index 67004e5c952..acdf24e6e97 100644 --- a/fdbrpc/AsyncFileKAIO.h +++ b/fdbrpc/AsyncFileKAIO.h @@ -433,7 +433,7 @@ class AsyncFileKAIO final : public IAsyncFile, public ReferenceCountedowner->lastFileSize != io->owner->nextFileSize) { ++ctx.countPreSubmitTruncate; int64_t truncateSize = io->owner->nextFileSize - io->owner->lastFileSize; - ASSERT(truncateSize > 0); + ASSERT_GT(truncateSize, 0); ctx.preSubmitTruncateBytes += truncateSize; largestTruncate = std::max(largestTruncate, truncateSize); io->owner->truncate(io->owner->nextFileSize); diff --git a/fdbrpc/FlowTransport.cpp b/fdbrpc/FlowTransport.cpp index ce5b50d0652..202e7a3ccd9 100644 --- a/fdbrpc/FlowTransport.cpp +++ b/fdbrpc/FlowTransport.cpp @@ -1290,7 +1290,7 @@ static void deliverNow(TransportData* self, g_currentDeliveryPeerDisconnect = nullptr; }); StringRef data = reader.arenaReadAll(); - ASSERT(data.size() > 8); + ASSERT_GT(data.size(), 8); ArenaObjectReader objReader(std::move(reader.arena()), data, AssumeVersion(reader.protocolVersion())); receiver->receive(objReader); } catch (Error& e) { diff --git a/fdbrpc/HTTP.cpp b/fdbrpc/HTTP.cpp index fc817b7885f..7de0e52c23d 100644 --- a/fdbrpc/HTTP.cpp +++ b/fdbrpc/HTTP.cpp @@ -267,7 +267,7 @@ Future read_delimited_into_string(Reference conn, size_t pos) { size_t sPos = pos; int lookBack = strlen(delim) - 1; - ASSERT(lookBack >= 0); + ASSERT_GE(lookBack, 0); while (true) { size_t endPos = buf->find(delim, sPos); diff --git a/fdbrpc/include/fdbrpc/AsyncFileCached.h b/fdbrpc/include/fdbrpc/AsyncFileCached.h index 59e6d67309e..fdbb6710576 100644 --- a/fdbrpc/include/fdbrpc/AsyncFileCached.h +++ b/fdbrpc/include/fdbrpc/AsyncFileCached.h @@ -162,7 +162,7 @@ class AsyncFileCached final : public IAsyncFile, public ReferenceCounted this->length) { length = int(this->length - offset); - ASSERT(length >= 0); + ASSERT_GE(length, 0); } auto f = read_write_impl(static_cast(data), length, offset); if (f.isReady() && !f.isError()) @@ -446,7 +446,7 @@ struct AFCPage : public EvictablePage, public FastAllocated { } void releaseZeroCopy() { --zeroCopyRefCount; - ASSERT(zeroCopyRefCount >= 0); + ASSERT_GE(zeroCopyRefCount, 0); } Future read(void* data, int length, int offset) { @@ -522,13 +522,13 @@ struct AFCPage : public EvictablePage, public FastAllocated { if (FLOW_KNOBS->FLOW_CACHEDFILE_WRITE_IO_SIZE > 0) { allowance = (pageCache->pageSize + FLOW_KNOBS->FLOW_CACHEDFILE_WRITE_IO_SIZE - 1) / FLOW_KNOBS->FLOW_CACHEDFILE_WRITE_IO_SIZE; // round up - ASSERT(allowance > 0); + ASSERT_GT(allowance, 0); } co_await owner->getRateControl()->getAllowance(allowance); } if (pageOffset + pageCache->pageSize > owner->length) { - ASSERT(pageOffset < owner->length); + ASSERT_LT(pageOffset, owner->length); memset(static_cast(data) + owner->length - pageOffset, 0, pageCache->pageSize - (owner->length - pageOffset)); diff --git a/fdbrpc/include/fdbrpc/AsyncFileReadAhead.h b/fdbrpc/include/fdbrpc/AsyncFileReadAhead.h index 3002286b486..e6323140ed0 100644 --- a/fdbrpc/include/fdbrpc/AsyncFileReadAhead.h +++ b/fdbrpc/include/fdbrpc/AsyncFileReadAhead.h @@ -83,7 +83,7 @@ class AsyncFileReadAheadCache final : public IAsyncFile, public ReferenceCounted // Start blocks up to the read ahead size beyond the last needed block but don't go past the end of the file int lastBlockNumInFile = ((fileSize + f->m_block_size - 1) / f->m_block_size) - 1; - ASSERT(lastBlockNum <= lastBlockNumInFile); + ASSERT_LE(lastBlockNum, lastBlockNumInFile); int lastBlockToStart = std::min(lastBlockNum + f->m_read_ahead_blocks, lastBlockNumInFile); int blockNum{ 0 }; @@ -138,7 +138,7 @@ class AsyncFileReadAheadCache final : public IAsyncFile, public ReferenceCounted } } - ASSERT(wpos == length); + ASSERT_EQ(wpos, length); ASSERT(localCache.empty()); // If the cache is too large then go through the cache in block number order and remove any entries whose future diff --git a/fdbrpc/include/fdbrpc/fdbrpc.h b/fdbrpc/include/fdbrpc/fdbrpc.h index f5924bb8b44..0aae6d92166 100644 --- a/fdbrpc/include/fdbrpc/fdbrpc.h +++ b/fdbrpc/include/fdbrpc/fdbrpc.h @@ -597,7 +597,7 @@ class ReplyPromiseStream { // Must be called on the server before sending results on the stream to ratelimit the amount of data outstanding to // the client Future onReady() const { - ASSERT(queue->acknowledgements.bytesLimit > 0); + ASSERT_GT(queue->acknowledgements.bytesLimit, 0); if (queue->acknowledgements.failures.isError()) { return queue->acknowledgements.failures.getError(); } diff --git a/flow/Histogram.cpp b/flow/Histogram.cpp index f9304796019..4e03d0ecc07 100644 --- a/flow/Histogram.cpp +++ b/flow/Histogram.cpp @@ -56,7 +56,7 @@ void HistogramRegistry::unregisterHistogram(Histogram* h) { TraceEvent(SevError, "HistogramNotRegistered").detail("group", h->group).detail("op", h->op); } int count = histograms.erase(name); - ASSERT(count == 1); + ASSERT_EQ(count, 1); } Histogram* HistogramRegistry::lookupHistogram(std::string const& name) { diff --git a/flow/Net2.cpp b/flow/Net2.cpp index 50d96a5d98d..4a6cccf8668 100644 --- a/flow/Net2.cpp +++ b/flow/Net2.cpp @@ -218,7 +218,7 @@ class Net2 final : public INetwork, public INetworkConnections { flowGlobalType global(int id) const override { return (globals.size() > id) ? globals[id] : nullptr; } void setGlobal(size_t id, flowGlobalType v) override { - ASSERT(id < globals.size()); + ASSERT_LT(id, globals.size()); globals[id] = v; } @@ -525,7 +525,7 @@ class Connection final : public IConnection, ReferenceCounted { if (err) { // Since there was an error, sent's value can't be used to infer that the buffer has data and the limit is // positive so check explicitly. - ASSERT(limit > 0); + ASSERT_GT(limit, 0); bool notEmpty = false; for (auto p = data; p; p = p->next) { if (p->bytes_written - p->bytes_sent > 0) { @@ -1230,7 +1230,7 @@ class SSLConnection final : public IConnection, ReferenceCounted if (err) { // Since there was an error, sent's value can't be used to infer that the buffer has data and the limit is // positive so check explicitly. - ASSERT(limit > 0); + ASSERT_GT(limit, 0); bool notEmpty = false; for (auto p = data; p; p = p->next) { if (p->bytes_written - p->bytes_sent > 0) { @@ -2252,7 +2252,7 @@ void ASIOReactor::wake() { } // namespace N2 SendBufferIterator::SendBufferIterator(SendBuffer const* p, int limit) : p(p), limit(limit) { - ASSERT(limit > 0); + ASSERT_GT(limit, 0); } void SendBufferIterator::operator++() { diff --git a/flow/TLSConfig.cpp b/flow/TLSConfig.cpp index 0ab43b54341..af75e0dcbd2 100644 --- a/flow/TLSConfig.cpp +++ b/flow/TLSConfig.cpp @@ -737,7 +737,7 @@ std::string getX509Name(const X509_NAME* name) { X509_NAME_print_ex(out.get(), name, /* indent= */ 0, /* flags */ XN_FLAG_ONELINE); unsigned char* rawName = nullptr; long length = BIO_get_mem_data(out.get(), &rawName); - ASSERT(length > 0); + ASSERT_GT(length, 0); std::string result((const char*)rawName, length); return result; } diff --git a/flow/flow.cpp b/flow/flow.cpp index 8debd77524a..4dc1e790608 100644 --- a/flow/flow.cpp +++ b/flow/flow.cpp @@ -137,10 +137,10 @@ std::string UID::toString() const { } UID UID::fromString(std::string const& s) { - ASSERT(s.size() == 32); + ASSERT_EQ(s.size(), 32); uint64_t a = 0, b = 0; int r = sscanf(s.c_str(), "%16" SCNx64 "%16" SCNx64, &a, &b); - ASSERT(r == 2); + ASSERT_EQ(r, 2); return UID(a, b); } @@ -437,7 +437,7 @@ int nChooseK(int n, int k) { ret *= n - i + 1; ret /= i; } - ASSERT(ret <= INT_MAX); + ASSERT_LE(ret, INT_MAX); return ret; } diff --git a/flow/include/flow/Error.h b/flow/include/flow/Error.h index b8df273e006..e13e222cca3 100644 --- a/flow/include/flow/Error.h +++ b/flow/include/flow/Error.h @@ -159,7 +159,7 @@ void assert_impl(char const* a_nm, bool (*compare)(T const&, U const&), char const* file, int line) { - if (!(compare(a, b) || isAssertDisabled(line))) { + if (!(compare(a, b) || isAssertDisabled(line))) [[unlikely]] { throw internal_error_impl(a_nm, Traceable::toString(a), opName, b_nm, Traceable::toString(b), file, line); } } From d616d655575f33b85fc40ef8de358acd2a31cfdd Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 15 Sep 2026 13:18:13 -0700 Subject: [PATCH 073/170] Extract network diagnostics into a standalone fdbrpc target --- contrib/mtlsbenchmark/client.sh | 5 +- contrib/mtlsbenchmark/readme.md | 16 +- contrib/mtlsbenchmark/server.sh | 5 +- fdbrpc/tests/CMakeLists.txt | 13 + .../fdbserver => fdbrpc/tests}/NetworkTest.h | 18 +- {fdbserver => fdbrpc/tests}/networktest.cpp | 26 +- fdbrpc/tests/networktest.md | 65 ++++ fdbrpc/tests/networktest_main.cpp | 359 ++++++++++++++++++ fdbrpc/tests/networktest_smoke.py | 226 +++++++++++ fdbserver/fdbserver.cpp | 34 +- 10 files changed, 695 insertions(+), 72 deletions(-) rename {fdbserver/include/fdbserver => fdbrpc/tests}/NetworkTest.h (77%) rename {fdbserver => fdbrpc/tests}/networktest.cpp (95%) create mode 100644 fdbrpc/tests/networktest.md create mode 100644 fdbrpc/tests/networktest_main.cpp create mode 100644 fdbrpc/tests/networktest_smoke.py diff --git a/contrib/mtlsbenchmark/client.sh b/contrib/mtlsbenchmark/client.sh index dafa2369a44..d6935f2f08a 100644 --- a/contrib/mtlsbenchmark/client.sh +++ b/contrib/mtlsbenchmark/client.sh @@ -9,9 +9,8 @@ # knob_disable_mainthread_tls_handshake is enabled to use background threads for TLS handshakes only # knob_tls_handshake_flowlock_priority is set to 8900 for enabling TLS flowlock priority as high as the handshake priority. Default is 7000. -taskset -c 0-0 /root/build_output/bin/fdbserver \ - -r unittests \ - -f :/network/p2ptest \ +taskset -c 0-0 "${FDBRPC_NETWORK_TEST:-/root/build_output/bin/fdbrpc_network_test}" \ + --mode p2p \ --test_remoteAddresses=127.0.0.1:4500:tls \ --test_targetDuration=10 \ --test_connectionsOut=10 \ diff --git a/contrib/mtlsbenchmark/readme.md b/contrib/mtlsbenchmark/readme.md index 34c2b58e1cd..418e082931c 100644 --- a/contrib/mtlsbenchmark/readme.md +++ b/contrib/mtlsbenchmark/readme.md @@ -5,7 +5,8 @@ A testing framework for benchmarking TLS performance in peer-to-peer network sce ## Prerequisites - OpenSSL or compatible tool for certificate generation -- Environment for FoundationDB unit test execution +- Build `fdbrpc_network_test` (`cmake --build build --target fdbrpc_network_test`). +- Set `FDBRPC_NETWORK_TEST` to the executable path if it differs from `/root/build_output/bin/fdbrpc_network_test`. ## Quick Start @@ -22,14 +23,17 @@ Generate the required certificate files: ### Step 2: Configure Test in Scripts -The test scripts support two unit tests that can be configured: +The standalone network diagnostic supports two P2P modes: -| Test Mode | Purpose | Configuration (set by -f) | +| Test Mode | Purpose | Configuration (set by --mode) | |-----------|---------|---------------| -| **Long Running** | Testing with connections and messages | `:/network/p2ptest` | -| **One Shot** | One-time connection only and no message | `:/network/p2poneshottest` | +| **Long Running** | Testing with connections and messages | `p2p` | +| **One Shot** | One-time connection only and no message | `p2p-oneshot` | -Set the desired test mode in your script before running. +Set the desired test mode in your script before running. The `--test_*`, +`--knob_*`, and `--tls_*` options retain their meanings. These modes were +previously run through `fdbserver -r unittests`; they now use the standalone +[`fdbrpc_network_test`](../../fdbrpc/tests/networktest.md) executable. ## Folder Structure diff --git a/contrib/mtlsbenchmark/server.sh b/contrib/mtlsbenchmark/server.sh index 037a9395252..731f1b023cf 100644 --- a/contrib/mtlsbenchmark/server.sh +++ b/contrib/mtlsbenchmark/server.sh @@ -11,9 +11,8 @@ # knob_tls_handshake_flowlock_priority is set to 8900 for enabling TLS flowlock priority as high as the handshake priority. Default is 7000. # knob_tls_handshake_timeout_seconds is set to 3.0 seconds to timeout a handshake if not completed in 3 seconds. The default is 2.0 seconds. -taskset -c 1-1 /root/build_output/bin/fdbserver \ - -r unittests \ - -f :/network/p2ptest \ +taskset -c 1-1 "${FDBRPC_NETWORK_TEST:-/root/build_output/bin/fdbrpc_network_test}" \ + --mode p2p \ --test_listenerAddresses=0.0.0.0:4500:tls \ --test_targetDuration=0 \ --knob_tls_handshake_limit=1000 \ diff --git a/fdbrpc/tests/CMakeLists.txt b/fdbrpc/tests/CMakeLists.txt index a0ba2a2dbba..5a46faf07e6 100644 --- a/fdbrpc/tests/CMakeLists.txt +++ b/fdbrpc/tests/CMakeLists.txt @@ -13,3 +13,16 @@ if(WITH_GRPC) endif() add_flow_target(EXECUTABLE NAME fdbrpc_transport_bench SRCS fdbrpc_bench.cpp) target_link_libraries(fdbrpc_transport_bench PUBLIC flow fdbrpc boost_target_program_options) + +add_flow_target(EXECUTABLE NAME fdbrpc_network_test SRCS networktest_main.cpp networktest.cpp NetworkTest.h) +target_link_libraries(fdbrpc_network_test PRIVATE flow fdbrpc fmt::fmt) + +if(FDB_REGISTER_UNIT_TESTS AND Python3_EXECUTABLE) + add_dependencies(unit_tests fdbrpc_network_test) + add_test(NAME unit/fdbrpc_network_test/native + COMMAND ${Python3_EXECUTABLE} ${CMAKE_CURRENT_SOURCE_DIR}/networktest_smoke.py $) + set_tests_properties(unit/fdbrpc_network_test/native PROPERTIES + LABELS "unit;native;fdbrpc_network_test" + TIMEOUT 90 + ENVIRONMENT "${SANITIZER_OPTIONS}") +endif() diff --git a/fdbserver/include/fdbserver/NetworkTest.h b/fdbrpc/tests/NetworkTest.h similarity index 77% rename from fdbserver/include/fdbserver/NetworkTest.h rename to fdbrpc/tests/NetworkTest.h index ae51501c22b..8bea06f75d7 100644 --- a/fdbserver/include/fdbserver/NetworkTest.h +++ b/fdbrpc/tests/NetworkTest.h @@ -18,13 +18,15 @@ * limitations under the License. */ -#ifndef FDBSERVER_NETWORKTEST_H -#define FDBSERVER_NETWORKTEST_H +#ifndef FDBRPC_TESTS_NETWORKTEST_H +#define FDBRPC_TESTS_NETWORKTEST_H #pragma once -#include "fdbclient/FDBTypes.h" #include "fdbrpc/fdbrpc.h" #include "flow/FileIdentifier.h" +#include "flow/UnitTest.h" + +constexpr int WLTOKEN_NETWORKTEST = WLTOKEN_FIRST_AVAILABLE; struct NetworkTestInterface { RequestStream test; @@ -35,9 +37,9 @@ struct NetworkTestInterface { struct NetworkTestReply { constexpr static FileIdentifier file_identifier = 14465374; - Value value; + Standalone value; NetworkTestReply() = default; - explicit NetworkTestReply(Value value) : value(value) {} + explicit NetworkTestReply(Standalone value) : value(value) {} template void serialize(Ar& ar) { serializer(ar, value); @@ -46,11 +48,11 @@ struct NetworkTestReply { struct NetworkTestRequest { constexpr static FileIdentifier file_identifier = 4146513; - Key key; + Standalone key; uint32_t replySize; ReplyPromise reply; NetworkTestRequest() = default; - NetworkTestRequest(Key key, uint32_t replySize) : key(key), replySize(replySize) {} + NetworkTestRequest(Standalone key, uint32_t replySize) : key(key), replySize(replySize) {} template void serialize(Ar& ar) { serializer(ar, key, replySize, reply); @@ -61,4 +63,6 @@ Future networkTestServer(); Future networkTestClient(std::string const& testServers); +Future networkTestP2P(const UnitTestParameters& params, bool oneshot); + #endif diff --git a/fdbserver/networktest.cpp b/fdbrpc/tests/networktest.cpp similarity index 95% rename from fdbserver/networktest.cpp rename to fdbrpc/tests/networktest.cpp index d8f6473cbc6..7138dfd4e95 100644 --- a/fdbserver/networktest.cpp +++ b/fdbrpc/tests/networktest.cpp @@ -19,7 +19,7 @@ */ #include "fmt/format.h" -#include "fdbserver/NetworkTest.h" +#include "NetworkTest.h" #include "flow/ActorCollection.h" #include "flow/CoroUtils.h" #include "flow/Knobs.h" @@ -28,8 +28,6 @@ #include "flow/IConnection.h" -constexpr int WLTOKEN_NETWORKTEST = WLTOKEN_FIRST_AVAILABLE; - struct LatencyStats { using sample = double; double x = 0; @@ -75,7 +73,7 @@ class NetworkTestServer { while (true) { NetworkTestRequest req = co_await interf.test.getFuture(); LatencyStats::sample sample = latency.tick(); - req.reply.send(NetworkTestReply(Value(std::string(req.replySize, '.')))); + req.reply.send(NetworkTestReply(Standalone(std::string(req.replySize, '.')))); latency.tock(sample); sent++; } @@ -631,23 +629,7 @@ struct P2PNetworkTest { // - wait for a random replyBytes sized response. // The client will close the connection after a random idleMilliseconds. // Reads and writes can optionally preceded by random delays, waitReadMilliseconds and waitWriteMilliseconds. -TEST_CASE(":/network/p2ptest") { - P2PNetworkTest p2p(params.get("listenerAddresses").orDefault(""), - params.get("remoteAddresses").orDefault(""), - params.getInt("connectionsOut").orDefault(1), - params.get("requestBytes").orDefault("50:100"), - params.get("replyBytes").orDefault("500:1000"), - params.get("requests").orDefault("10:10000"), - params.get("idleMilliseconds").orDefault("0"), - params.get("waitReadMilliseconds").orDefault("0"), - params.get("waitWriteMilliseconds").orDefault("0"), - params.getDouble("targetDuration").orDefault(0.0), - false); - - co_await p2p.run(); -} - -TEST_CASE(":/network/p2poneshottest") { +Future networkTestP2P(const UnitTestParameters& params, bool oneshot) { P2PNetworkTest p2p(params.get("listenerAddresses").orDefault(""), params.get("remoteAddresses").orDefault(""), params.getInt("connectionsOut").orDefault(1), @@ -658,7 +640,7 @@ TEST_CASE(":/network/p2poneshottest") { params.get("waitReadMilliseconds").orDefault("0"), params.get("waitWriteMilliseconds").orDefault("0"), params.getDouble("targetDuration").orDefault(0.0), - true); + oneshot); co_await p2p.run(); } diff --git a/fdbrpc/tests/networktest.md b/fdbrpc/tests/networktest.md new file mode 100644 index 00000000000..03128567ed8 --- /dev/null +++ b/fdbrpc/tests/networktest.md @@ -0,0 +1,65 @@ +# Network diagnostics + +`fdbrpc_network_test` runs RPC and socket diagnostics using Flow and fdbrpc, +without linking the database client or server. Build it with: + +```sh +cmake --build build --target fdbrpc_network_test +``` + +The executable is `build/bin/fdbrpc_network_test`. It is a development tool; +copy it to each participating host when testing across machines. + +## RPC request/reply + +Start a server, then run a client in another terminal: + +```sh +build/bin/fdbrpc_network_test --mode server --public-address 127.0.0.1:4500 +build/bin/fdbrpc_network_test --mode client --testservers 127.0.0.1:4500 \ + --knob_network_test_script_mode=true +``` + +The client supports comma-separated server addresses. Flow knobs retain their +existing names, including `network_test_request_size`, `network_test_reply_size`, +`network_test_client_count`, and `network_test_request_count`. Script mode reports +one measurement after a warmup interval and then finishes. Without a request +limit or script mode, the client runs until stopped. The server runs until stopped. + +`--listen-address` overrides the bind address while `--public-address` specifies +the advertised endpoint. Use `--help` for TLS and tracing options. + +## Raw connections and TLS + +P2P mode can listen, connect, or do both. For example, a bounded loopback run: + +```sh +build/bin/fdbrpc_network_test --mode p2p \ + --test_listenerAddresses=127.0.0.1:4501 \ + --test_remoteAddresses=127.0.0.1:4501 \ + --test_connectionsOut=2 --test_targetDuration=5 +``` + +The existing `--test_*` parameters control payload size ranges, requests per +connection, and delays before reads, writes, and connection close. A duration +of zero means run until stopped. `--mode p2p-oneshot` tests one connection and +handshake without sending the traffic workload; run its listener and connector +as separate processes. + +Append `:tls` to addresses to enable TLS and supply `--tls_certificate_file`, +`--tls_key_file`, `--tls_ca_file`, and `--tls_verify_peers` as appropriate. See +[`contrib/mtlsbenchmark`](../../contrib/mtlsbenchmark/readme.md) for a two-process +TLS example with handshake knobs. + +## Migration from fdbserver + +| Previous invocation | Standalone invocation | +| --- | --- | +| `fdbserver -r networktestserver` | `fdbrpc_network_test --mode server` | +| `fdbserver -r networktestclient` | `fdbrpc_network_test --mode client` | +| `fdbserver -r unittests -f :/network/p2ptest` | `fdbrpc_network_test --mode p2p` | +| `fdbserver -r unittests -f :/network/p2poneshottest` | `fdbrpc_network_test --mode p2p-oneshot` | + +The networking options above retain their names. The old server roles and P2P +unit-test registrations are removed. The standalone target's CTest entry runs a +bounded loopback smoke test; it does not start an indefinite benchmark. diff --git a/fdbrpc/tests/networktest_main.cpp b/fdbrpc/tests/networktest_main.cpp new file mode 100644 index 00000000000..60670b51045 --- /dev/null +++ b/fdbrpc/tests/networktest_main.cpp @@ -0,0 +1,359 @@ +/* + * networktest_main.cpp + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "NetworkTest.h" +#include "SimpleOpt/SimpleOpt.h" +#include "fdbrpc/FlowTransport.h" +#include "fdbrpc/Net2FileSystem.h" +#include "flow/ArgParseUtil.h" +#include "flow/BooleanParam.h" +#include "flow/Knobs.h" +#include "flow/Platform.h" +#include "flow/TLSConfig.h" +#include "flow/Trace.h" +#include "flow/UnitTest.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +FDB_BOOLEAN_PARAM(Randomize); +FDB_BOOLEAN_PARAM(IsSimulated); + +namespace { + +enum Option { + OPT_HELP, + OPT_MODE, + OPT_TESTSERVERS, + OPT_PUBLIC_ADDRESS, + OPT_LISTEN_ADDRESS, + OPT_TEST_PARAM, + OPT_KNOB, + OPT_TRACE_DIR, +}; + +CSimpleOpt::SOption options[] = { { OPT_HELP, "-h", SO_NONE }, + { OPT_HELP, "--help", SO_NONE }, + { OPT_MODE, "-m", SO_REQ_SEP }, + { OPT_MODE, "--mode", SO_REQ_SEP }, + { OPT_TESTSERVERS, "--testservers", SO_REQ_SEP }, + { OPT_PUBLIC_ADDRESS, "-p", SO_REQ_SEP }, + { OPT_PUBLIC_ADDRESS, "--public-address", SO_REQ_SEP }, + { OPT_LISTEN_ADDRESS, "-l", SO_REQ_SEP }, + { OPT_LISTEN_ADDRESS, "--listen-address", SO_REQ_SEP }, + { OPT_TEST_PARAM, "--test-", SO_REQ_SEP }, + { OPT_KNOB, "--knob-", SO_REQ_SEP }, + { OPT_TRACE_DIR, "--trace-dir", SO_REQ_SEP }, + { OPT_TRACE_DIR, "--logdir", SO_REQ_SEP }, + TLS_OPTION_FLAGS, + SO_END_OF_OPTIONS }; + +struct Options { + std::string mode; + std::string testServers; + std::vector publicAddresses; + std::vector listenAddresses; + UnitTestParameters testParams; + TLSConfig tlsConfig{ TLSEndpointType::SERVER }; + std::string traceDir = "."; + bool showHelp = false; +}; + +void printUsage(const char* program) { + printf("Usage: %s --mode MODE [OPTIONS]\n" + "\n" + "Modes:\n" + " server Serve RPC network-test requests\n" + " client Send RPC requests to --testservers ADDRESS[,ADDRESS...]\n" + " p2p Exercise raw connections and traffic\n" + " p2p-oneshot Exercise connection handshakes without traffic\n" + "\n" + "Options:\n" + " -p, --public-address ADDR RPC server public IP:PORT[:tls]\n" + " -l, --listen-address ADDR RPC server bind address (default: public)\n" + " --testservers ADDRS RPC server addresses, or nanosleep\n" + " --test_NAME VALUE P2P parameter (e.g. listenerAddresses,\n" + " remoteAddresses, targetDuration, connectionsOut)\n" + " --knob_NAME VALUE Override a Flow knob\n" + " --trace-dir DIR Trace directory (default: .)\n" + " -h, --help Show this help\n" + "\n" + "Public/listen addresses accept comma-separated lists or repeated options,\n" + "with at most two addresses. A listen address may be 'public'.\n" + "Option names accept either hyphens or underscores.\n" + "\n%s", + program, + TLS_HELP); +} + +void appendAddresses(std::vector& addresses, const char* text) { + std::string remaining(text); + for (;;) { + const auto comma = remaining.find(','); + addresses.push_back(remaining.substr(0, comma)); + if (comma == std::string::npos) { + return; + } + remaining.erase(0, comma + 1); + } +} + +bool validNonnegativeInt(std::string_view text, int maximum = std::numeric_limits::max()) { + int value; + const auto [end, error] = std::from_chars(text.data(), text.data() + text.size(), value); + return error == std::errc() && end == text.data() + text.size() && value >= 0 && value <= maximum; +} + +bool validateP2PParams(const UnitTestParameters& params) { + for (const auto& [name, value] : params.params) { + bool valid = false; + if (name == "listenerAddresses" || name == "remoteAddresses") { + valid = value.empty(); + if (!value.empty()) { + const auto addresses = NetworkAddress::parseList(value); + valid = !addresses.empty() && std::all_of(addresses.begin(), addresses.end(), [](const auto& address) { + return address.isValid(); + }); + } + } else if (name == "connectionsOut") { + valid = validNonnegativeInt(value); + } else if (name == "targetDuration") { + try { + size_t end; + const auto duration = std::stod(value, &end); + valid = end == value.size() && std::isfinite(duration) && duration >= 0; + } catch (const std::exception&) { + } + } else if (name == "requestBytes" || name == "replyBytes" || name == "requests" || name == "idleMilliseconds" || + name == "waitReadMilliseconds" || name == "waitWriteMilliseconds") { + const auto colon = value.find(':'); + // RandomIntRange samples an inclusive upper bound by adding one. + constexpr int maximum = std::numeric_limits::max() - 1; + valid = validNonnegativeInt(value.substr(0, colon), maximum) && + (colon == std::string::npos || validNonnegativeInt(value.substr(colon + 1), maximum)); + } + if (!valid) { + fprintf(stderr, "ERROR: Invalid P2P parameter --test_%s=%s\n", name.c_str(), value.c_str()); + return false; + } + } + if (params.get("listenerAddresses").orDefault("").empty() && + (params.get("remoteAddresses").orDefault("").empty() || params.getInt("connectionsOut").orDefault(1) == 0)) { + fprintf(stderr, "ERROR: P2P mode requires a listener or a remote with positive connectionsOut\n"); + return false; + } + return true; +} + +bool parseArgs(int argc, char** argv, Options& result, FlowKnobs& knobs) { + CSimpleOpt args(argc, argv, options, SO_O_EXACT | SO_O_HYPHEN_TO_UNDERSCORE); + while (args.Next()) { + if (args.LastError() != SO_SUCCESS) { + fprintf(stderr, "ERROR: Invalid or incomplete option '%s'\n", args.OptionText()); + return false; + } + switch (args.OptionId()) { + case OPT_HELP: + result.showHelp = true; + return true; + case OPT_MODE: + result.mode = args.OptionArg(); + break; + case OPT_TESTSERVERS: + result.testServers = args.OptionArg(); + break; + case OPT_PUBLIC_ADDRESS: + appendAddresses(result.publicAddresses, args.OptionArg()); + break; + case OPT_LISTEN_ADDRESS: + appendAddresses(result.listenAddresses, args.OptionArg()); + break; + case OPT_TEST_PARAM: { + auto name = extractPrefixedArgument("--test", args.OptionSyntax()); + if (!name.present() || name.get().empty()) { + return false; + } + result.testParams.set(name.get(), args.OptionArg()); + break; + } + case OPT_KNOB: { + auto name = extractPrefixedArgument("--knob", args.OptionSyntax()); + if (!name.present() || name.get().empty()) { + return false; + } + const auto value = knobs.parseKnobValue(name.get(), args.OptionArg()); + const bool set = std::visit( + [&](const auto& parsed) { + if constexpr (std::is_same_v, NoKnobFound>) { + return false; + } else { + return knobs.setKnob(name.get(), parsed); + } + }, + value); + if (!set) { + fprintf(stderr, "ERROR: Unknown Flow knob '%s'\n", name.get().c_str()); + return false; + } + break; + } + case OPT_TRACE_DIR: + result.traceDir = args.OptionArg(); + break; + case TLSConfig::OPT_TLS_PLUGIN: + break; + case TLSConfig::OPT_TLS_CERTIFICATES: + result.tlsConfig.setCertificatePath(args.OptionArg()); + break; + case TLSConfig::OPT_TLS_KEY: + result.tlsConfig.setKeyPath(args.OptionArg()); + break; + case TLSConfig::OPT_TLS_CA_FILE: + result.tlsConfig.setCAPath(args.OptionArg()); + break; + case TLSConfig::OPT_TLS_PASSWORD: + result.tlsConfig.setPassword(args.OptionArg()); + break; + case TLSConfig::OPT_TLS_VERIFY_PEERS: + result.tlsConfig.addVerifyPeers(args.OptionArg()); + break; + case TLSConfig::OPT_TLS_DISABLE_PLAINTEXT_CONNECTION: + result.tlsConfig.setDisablePlainTextConnection(true); + break; + } + } + if (args.FileCount() != 0 || + (result.mode != "client" && result.mode != "server" && result.mode != "p2p" && result.mode != "p2p-oneshot")) { + fprintf(stderr, "ERROR: Expected --mode client, server, p2p, or p2p-oneshot\n"); + return false; + } + if (result.mode == "server") { + if (result.publicAddresses.empty() || result.publicAddresses.size() > 2 || + (!result.listenAddresses.empty() && result.listenAddresses.size() != result.publicAddresses.size())) { + fprintf(stderr, "ERROR: Server requires one or two public addresses and matching listen addresses\n"); + return false; + } + result.listenAddresses.resize(result.publicAddresses.size(), "public"); + } else if (!result.publicAddresses.empty() || !result.listenAddresses.empty()) { + fprintf(stderr, "ERROR: --public-address and --listen-address require --mode server\n"); + return false; + } + if ((result.mode == "client") != !result.testServers.empty()) { + fprintf(stderr, "ERROR: --testservers is required for client mode and is only valid in client mode\n"); + return false; + } + if (result.mode == "p2p" || result.mode == "p2p-oneshot") { + return validateP2PParams(result.testParams); + } + if (!result.testParams.params.empty()) { + fprintf(stderr, "ERROR: --test_NAME parameters require a P2P mode\n"); + return false; + } + return true; +} + +Future stopNetworkAfter(Future work) { + try { + co_await work; + } catch (Error&) { + g_network->stop(); + throw; + } + g_network->stop(); +} + +} // namespace + +int main(int argc, char** argv) { + try { + platformInit(); + Error::init(); + setvbuf(stdout, nullptr, _IOLBF, BUFSIZ); + setvbuf(stderr, nullptr, _IOLBF, BUFSIZ); + setThreadLocalDeterministicRandomSeed(platform::getRandomSeed()); + // Network and trace globals retain these knobs through process shutdown. + auto* knobs = new FlowKnobs(Randomize::False, IsSimulated::False); + FLOW_KNOBS = knobs; + Options opts; + if (!parseArgs(argc, argv, opts, *knobs)) { + printUsage(argv[0]); + return 1; + } + if (opts.showHelp) { + printUsage(argv[0]); + return 0; + } + + TraceEvent::setNetworkThread(); + g_network = newNet2(opts.tlsConfig, false, true); + g_network->addStopCallback(Net2FileSystem::stop); + Net2FileSystem::newFileSystem(); + FlowTransport::createInstance(false, 1, WLTOKEN_NETWORKTEST + 1); + openTraceFile({}, 10 << 20, 10 << 20, opts.traceDir, "networktest"); + g_network->initTLS(); + g_network->initMetrics(); + FlowTransport::transport().initMetrics(); + + std::vector> work; + if (opts.mode == "server") { + for (size_t i = 0; i < opts.publicAddresses.size(); ++i) { + const auto publicAddress = NetworkAddress::parse(opts.publicAddresses[i]); + const auto listenAddress = opts.listenAddresses[i] == "public" + ? publicAddress + : NetworkAddress::parse(opts.listenAddresses[i]); + if (!publicAddress.isValid() || !listenAddress.isValid() || + publicAddress.isTLS() != listenAddress.isTLS()) { + fprintf(stderr, "ERROR: Public/listen addresses must be valid and use matching TLS settings\n"); + return 1; + } + auto listenError = FlowTransport::transport().bind(publicAddress, listenAddress); + if (listenError.isReady()) { + listenError.get(); + } + work.push_back(listenError); + printf("Listener: %s\n", listenAddress.toString().c_str()); + } + work.push_back(networkTestServer()); + } else if (opts.mode == "client") { + work.push_back(networkTestClient(opts.testServers)); + } else { + work.push_back(networkTestP2P(opts.testParams, opts.mode == "p2p-oneshot")); + } + Future done = stopNetworkAfter(waitForAny(work)); + g_network->run(); + flushTraceFileVoid(); + done.get(); + return 0; + } catch (Error& e) { + fprintf(stderr, "ERROR: Network test failed: %s (%d)\n", e.what(), e.code()); + } catch (std::exception& e) { + fprintf(stderr, "ERROR: Network test failed: %s\n", e.what()); + } + flushTraceFileVoid(); + return 1; +} diff --git a/fdbrpc/tests/networktest_smoke.py b/fdbrpc/tests/networktest_smoke.py new file mode 100644 index 00000000000..b102f60229f --- /dev/null +++ b/fdbrpc/tests/networktest_smoke.py @@ -0,0 +1,226 @@ +#!/usr/bin/env python3 +"""Exercise the standalone network diagnostic over bounded loopback connections.""" + +import argparse +import contextlib +import re +import socket +import subprocess +import tempfile +import time +from pathlib import Path + + +def unused_port(): + with socket.socket() as sock: + sock.bind(("127.0.0.1", 0)) + return sock.getsockname()[1] + + +class SmokeTest: + def __init__(self, executable, root): + self.executable = executable + self.root = root + self.deadline = time.monotonic() + 75 + self.tls_options = [] + self.tls_suffix = "" + + def timeout(self, seconds): + remaining = self.deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError("network test smoke exceeded its deadline") + return min(seconds, remaining) + + @contextlib.contextmanager + def process(self, name, arguments): + directory = self.root / name + directory.mkdir() + output = directory / "output.log" + with output.open("w") as log: + process = subprocess.Popen( + [self.executable, *arguments, *self.tls_options], + cwd=directory, + stdout=log, + stderr=subprocess.STDOUT, + ) + try: + yield process, output + except BaseException: + print("{} output:\n{}".format(name, output.read_text())) + raise + finally: + if process.poll() is None: + process.terminate() + try: + process.wait(timeout=2) + except subprocess.TimeoutExpired: + process.kill() + process.wait(timeout=2) + + def finish(self, process, output, seconds=15, success=True): + result = process.wait(timeout=self.timeout(seconds)) + text = output.read_text() + if (result == 0) != success: + raise AssertionError("unexpected exit status {}:\n{}".format(result, text)) + return text + + def ready(self, process, output): + deadline = time.monotonic() + self.timeout(10) + while time.monotonic() < deadline: + if process.poll() is not None: + raise AssertionError("listener exited before becoming ready") + if "Listener: " in output.read_text(): + return + time.sleep(0.05) + raise TimeoutError("listener did not report readiness") + + def address(self, port): + return "127.0.0.1:{}{}".format(port, self.tls_suffix) + + def enable_tls(self): + cert = self.root / "cert.pem" + key = self.root / "key.pem" + config = self.root / "openssl.cnf" + config.write_text( + "[req]\n" + "distinguished_name=subject\n" + "x509_extensions=extensions\n" + "prompt=no\n" + "[subject]\n" + "CN=networktest-smoke\n" + "[extensions]\n" + "basicConstraints=critical,CA:TRUE\n" + "keyUsage=critical,digitalSignature,keyEncipherment,keyCertSign\n" + "extendedKeyUsage=serverAuth,clientAuth\n" + "subjectAltName=IP:127.0.0.1\n" + ) + subprocess.run( + [ + "openssl", + "req", + "-x509", + "-newkey", + "rsa:2048", + "-nodes", + "-days", + "1", + "-config", + str(config), + "-keyout", + str(key), + "-out", + str(cert), + ], + check=True, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + timeout=self.timeout(15), + ) + self.tls_suffix = ":tls" + self.tls_options = [ + "--tls_certificate_file", + str(cert), + "--tls_key_file", + str(key), + "--tls_ca_file", + str(cert), + "--tls_verify_peers", + "Root.CN=networktest-smoke", + ] + + def rpc(self): + port = unused_port() + address = self.address(port) + with self.process("rpc-server", ["--mode", "server", "-p", address]) as ( + server, + server_output, + ): + self.ready(server, server_output) + with self.process( + "rpc-client", + [ + "--mode", + "client", + "--testservers", + address, + "--knob_network_test_script_mode=true", + ], + ) as (client, output): + text = self.finish(client, output) + rates = re.findall(r"(?m)^([0-9]+(?:\.[0-9]+)?)\t", text) + assert any(float(rate) > 0 for rate in rates), text + assert server.poll() is None, "RPC server exited during traffic" + + def p2p(self): + address = self.address(unused_port()) + with self.process( + "p2p", + [ + "--mode", + "p2p", + "--test_listenerAddresses=" + address, + "--test_remoteAddresses=" + address, + "--test_connectionsOut=1", + "--test_requests=1", + "--test_targetDuration=2", + ], + ) as (process, output): + text = self.finish(process, output) + for direction in ("in", "out"): + rates = re.findall(r"([0-9.]+)/s completed sessions " + direction, text) + assert any(float(rate) > 0 for rate in rates), text + errors = re.findall(r"Total Errors (\d+)", text) + assert errors and all(int(count) == 0 for count in errors), text + + def handshake(self): + port = unused_port() + address = self.address(port) + with self.process( + "handshake-server", + ["--mode", "p2p-oneshot", "--test_listenerAddresses=" + address], + ) as (server, server_output): + self.ready(server, server_output) + with self.process( + "handshake-client", + ["--mode", "p2p-oneshot", "--test_remoteAddresses=" + address], + ) as (client, output): + text = self.finish(client, output) + assert re.search(r"Client: connected to .*handshake done", text), text + text = self.finish(server, server_output, seconds=20) + assert re.search(r"Server: connected from .*handshake done", text), text + assert "handshake error" not in text, text + + def invalid_arguments(self): + server = ["--mode", "server", "-p", self.address(unused_port())] + cases = [ + [], + server + ["--knob_network_test_script_mode=invalid"], + server + ["--knob_not_a_network_test_knob=1"], + server + ["--not-a-network-test-option"], + ] + for index, arguments in enumerate(cases): + with self.process("invalid-{}".format(index), arguments) as ( + process, + output, + ): + self.finish(process, output, seconds=5, success=False) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("executable", type=lambda value: str(Path(value).resolve())) + parser.add_argument("--tls", action="store_true", help="use temporary TLS fixtures") + args = parser.parse_args() + with tempfile.TemporaryDirectory(prefix="fdbrpc-network-smoke-") as directory: + smoke = SmokeTest(args.executable, Path(directory)) + smoke.invalid_arguments() + if args.tls: + smoke.enable_tls() + smoke.rpc() + smoke.p2p() + smoke.handshake() + print("network test smoke passed ({})".format("TLS" if args.tls else "plain TCP")) + + +if __name__ == "__main__": + main() diff --git a/fdbserver/fdbserver.cpp b/fdbserver/fdbserver.cpp index 8e4f8f37de8..40c2f3995a1 100644 --- a/fdbserver/fdbserver.cpp +++ b/fdbserver/fdbserver.cpp @@ -59,7 +59,6 @@ #include "fdbserver/CoroFlow.h" #include "fdbserver/core/MoveKeys.h" #include "fdbserver/core/Knobs.h" -#include "fdbserver/NetworkTest.h" #include "fdbserver/kvstore/KVFileUtils.h" #include "fdbserver/core/ServerDBInfo.h" #include "fdbserver/core/FDBSimulationPolicy.h" @@ -130,7 +129,7 @@ enum { OPT_CONNFILE, OPT_SEEDCONNFILE, OPT_SEEDCONNSTRING, OPT_ROLE, OPT_LISTEN, OPT_PUBLICADDR, OPT_DATAFOLDER, OPT_TLOG_SPILL_DATAFOLDER, OPT_LOGFOLDER, OPT_PARENTPID, OPT_TRACER, OPT_NEWCONSOLE, OPT_NOBOX, OPT_TESTFILE, OPT_RESTARTING, OPT_RESTORING, OPT_RANDOMSEED, OPT_RESEED_TIME, OPT_KEY, OPT_MEMLIMIT, OPT_VMEMLIMIT, OPT_STORAGEMEMLIMIT, OPT_CACHEMEMLIMIT, OPT_MACHINEID, OPT_DCID, OPT_MACHINE_CLASS, OPT_BUGGIFY, OPT_VERSION, OPT_BUILD_FLAGS, OPT_CRASHONERROR, OPT_HELP, OPT_NETWORKIMPL, OPT_NOBUFSTDOUT, OPT_BUFSTDOUTERR, - OPT_TRACECLOCK, OPT_NUMTESTERS, OPT_DEVHELP, OPT_PRINT_CODE_PROBES, OPT_ROLLSIZE, OPT_MAXLOGS, OPT_MAXLOGSSIZE, OPT_KNOB, OPT_UNITTESTPARAM, OPT_TESTSERVERS, OPT_TEST_ON_SERVERS, OPT_METRICSCONNFILE, + OPT_TRACECLOCK, OPT_NUMTESTERS, OPT_DEVHELP, OPT_PRINT_CODE_PROBES, OPT_ROLLSIZE, OPT_MAXLOGS, OPT_MAXLOGSSIZE, OPT_KNOB, OPT_UNITTESTPARAM, OPT_TEST_ON_SERVERS, OPT_METRICSCONNFILE, OPT_METRICSPREFIX, OPT_LOGGROUP, OPT_LOCALITY, OPT_IO_TRUST_SECONDS, OPT_IO_TRUST_WARN_ONLY, OPT_FILESYSTEM, OPT_TLOG_SPILL_FILESYSTEM, OPT_PROFILER_RSS_SIZE, OPT_KVFILE, OPT_TRACE_FORMAT, OPT_WHITELIST_BINPATH, OPT_BLOB_CREDENTIALS, OPT_PROXY, OPT_DEPRECATED_CONFIG_PATH, OPT_DEPRECATED_USE_TEST_CONFIG_DB, OPT_DEPRECATED_NO_CONFIG_DB, OPT_FAULT_INJECTION, OPT_PROFILER, OPT_PRINT_SIMTIME, OPT_IP_TRUSTED_MASK, @@ -214,7 +213,6 @@ CSimpleOpt::SOption g_rgOptions[] = { { OPT_KNOB, "--knob-", SO_REQ_SEP }, { OPT_UNITTESTPARAM, "--test-", SO_REQ_SEP }, { OPT_LOCALITY, "--locality-", SO_REQ_SEP }, - { OPT_TESTSERVERS, "--testservers", SO_REQ_SEP }, { OPT_TEST_ON_SERVERS, "--testonservers", SO_NONE }, { OPT_METRICSCONNFILE, "--metrics-cluster", SO_REQ_SEP }, { OPT_METRICSPREFIX, "--metrics-prefix", SO_REQ_SEP }, @@ -608,7 +606,7 @@ static void printUsage(const char* name, bool devhelp) { printf(" --build-flags Print build information and exit.\n"); printOptionUsage("-r ROLE, --role ROLE", " Server role (valid options are fdbd, test, multitest," - " simulation, networktestclient, networktestserver, restore" + " simulation, restore" " consistencycheck, consistencycheckurgent, kvfileintegritycheck, kvfilegeneratesums, " "kvfiledump, mocks3server, unittests)." " The default is `fdbd'."); @@ -648,9 +646,6 @@ static void printUsage(const char* name, bool devhelp) { " the given threshold. fdbserver needs to be compiled with" " USE_GPERFTOOLS flag in order to use this feature."); #endif - printOptionUsage("--testservers ADDRESSES", - " The addresses of networktestservers" - " specified as ADDRESS:PORT,ADDRESS:PORT..."); printOptionUsage("--testonservers", " Testers are recruited on servers."); printOptionUsage("--metrics-cluster CONNFILE", " The cluster file designating where this process will" @@ -944,8 +939,6 @@ enum class ServerRole { KVFileDump, MockS3Server, MultiTester, - NetworkTestClient, - NetworkTestServer, Restore, SearchMutations, Simulation, @@ -972,7 +965,6 @@ struct CLIOptions { const char* testFile = "tests/default.txt"; std::string kvFile; - std::string testServersStr; std::string whitelistBinPaths; std::vector publicAddressStrs, listenAddressStrs, grpcAddressStrs; @@ -1227,10 +1219,6 @@ struct CLIOptions { role = ServerRole::VersionedMapTest; else if (!strcmp(sRole, "createtemplatedb")) role = ServerRole::CreateTemplateDatabase; - else if (!strcmp(sRole, "networktestclient")) - role = ServerRole::NetworkTestClient; - else if (!strcmp(sRole, "networktestserver")) - role = ServerRole::NetworkTestServer; else if (!strcmp(sRole, "kvfileintegritycheck")) role = ServerRole::KVFileIntegrityCheck; else if (!strcmp(sRole, "kvfilegeneratesums")) @@ -1547,9 +1535,6 @@ struct CLIOptions { case OPT_CRASHONERROR: g_crashOnError = true; break; - case OPT_TESTSERVERS: - testServersStr = args.OptionArg(); - break; case OPT_TEST_ON_SERVERS: testOnServers = true; break; @@ -1778,12 +1763,6 @@ struct CLIOptions { flushAndExit(FDB_EXIT_ERROR); } - if (role == ServerRole::NetworkTestClient && testServersStr.empty()) { - fprintf(stderr, "ERROR: please specify --testservers\n"); - printHelpTeaser(argv[0]); - flushAndExit(FDB_EXIT_ERROR); - } - if (role == ServerRole::ChangeClusterKey) { bool error = false; if (newClusterKey.empty()) { @@ -1985,8 +1964,7 @@ int main(int argc, char* argv[]) { FlowTransport::createInstance(false, 1, WLTOKEN_RESERVED_COUNT, &opts.allowList); opts.buildNetwork(argv[0]); - const bool expectsPublicAddress = - (role == ServerRole::FDBD || role == ServerRole::NetworkTestServer || role == ServerRole::MockS3Server); + const bool expectsPublicAddress = (role == ServerRole::FDBD || role == ServerRole::MockS3Server); if (opts.publicAddressStrs.empty()) { if (expectsPublicAddress) { fprintf(stderr, "ERROR: The -p or --public-address option is required\n"); @@ -2382,12 +2360,6 @@ int main(int argc, char* argv[]) { g_network->run(); } else if (role == ServerRole::CreateTemplateDatabase) { createTemplateDatabase(); - } else if (role == ServerRole::NetworkTestClient) { - f = stopAfter(networkTestClient(opts.testServersStr)); - g_network->run(); - } else if (role == ServerRole::NetworkTestServer) { - f = stopAfter(networkTestServer()); - g_network->run(); } else if (role == ServerRole::KVFileIntegrityCheck) { f = stopAfter(KVFileCheck(opts.kvFile, true)); g_network->run(); From 2a237f842972432bda3e0e8a983cbe4ba444b26f Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 15 Sep 2026 13:36:40 -0700 Subject: [PATCH 074/170] Remove unnecessary tag throttle unit tests --- fdbserver/ratekeeper/CMakeLists.txt | 3 - .../ratekeeper/RkTagThrottleCollection.cpp | 60 ------------------- 2 files changed, 63 deletions(-) diff --git a/fdbserver/ratekeeper/CMakeLists.txt b/fdbserver/ratekeeper/CMakeLists.txt index ed6e8261136..317cc8276c8 100644 --- a/fdbserver/ratekeeper/CMakeLists.txt +++ b/fdbserver/ratekeeper/CMakeLists.txt @@ -4,9 +4,6 @@ add_flow_target(STATIC_LIBRARY NAME fdbserver_ratekeeper SRCS ${FDBSERVER_RATEKE add_fdbserver_link_test(fdbserver_ratekeeperlinktest fdbserver_ratekeeper fdbserver_core) -add_fdbserver_unit_test(fdbserver_ratekeeper_test ratekeeper - fdbserver_ratekeeper - fdbserver_core) configure_fdbserver_common_includes(fdbserver_ratekeeper) target_include_directories(fdbserver_ratekeeper diff --git a/fdbserver/ratekeeper/RkTagThrottleCollection.cpp b/fdbserver/ratekeeper/RkTagThrottleCollection.cpp index 1b3491adf1a..4ed89fcf217 100644 --- a/fdbserver/ratekeeper/RkTagThrottleCollection.cpp +++ b/fdbserver/ratekeeper/RkTagThrottleCollection.cpp @@ -21,7 +21,6 @@ #include "fdbserver/core/Knobs.h" #include "RkTagThrottleCollection.h" -#include "flow/UnitTest.h" double RkTagThrottleCollection::RkTagThrottleData::getTargetRate(Optional requestRate) const { if (limits.tpsRate == 0.0 || !requestRate.present() || requestRate.get() == 0.0 || !rateSet) { @@ -346,62 +345,3 @@ void RkTagThrottleCollection::incrementBusyTagCount(TagThrottledReason reason) { TraceEvent(SevWarn, "UnsetTagThrottledReason"); } } - -TEST_CASE("/fdbserver/ratekeeper/TagThrottleCollection/MoveConstruction") { - RkTagThrottleCollection source; - source.incrementBusyTagCount(TagThrottledReason::BUSY_READ); - source.incrementBusyTagCount(TagThrottledReason::BUSY_WRITE); - source.incrementBusyTagCount(TagThrottledReason::BUSY_WRITE); - - RkTagThrottleCollection destination(std::move(source)); - ASSERT_EQ(destination.getBusyReadTagCount(), 1); - ASSERT_EQ(destination.getBusyWriteTagCount(), 2); - co_return; -} - -TEST_CASE("/fdbserver/ratekeeper/TagThrottleCollection/MoveAssignment") { - RkTagThrottleCollection source; - source.incrementBusyTagCount(TagThrottledReason::BUSY_READ); - source.incrementBusyTagCount(TagThrottledReason::BUSY_WRITE); - source.incrementBusyTagCount(TagThrottledReason::BUSY_WRITE); - RkTagThrottleCollection destination; - destination.incrementBusyTagCount(TagThrottledReason::BUSY_READ); - destination.incrementBusyTagCount(TagThrottledReason::BUSY_READ); - - destination = std::move(source); - ASSERT_EQ(destination.getBusyReadTagCount(), 1); - ASSERT_EQ(destination.getBusyWriteTagCount(), 2); - - destination = RkTagThrottleCollection(); - ASSERT_EQ(destination.getBusyReadTagCount(), 0); - ASSERT_EQ(destination.getBusyWriteTagCount(), 0); - co_return; -} - -TEST_CASE("/fdbserver/ratekeeper/TagThrottleCollection/ExpiredManualPreservesActiveRates") { - RkTagThrottleCollection throttles; - const double expiration = now() + 1.0; - const double activeExpiration = now() + 3600.0; - const TransactionTag firstTag = "first"_sr; - const TransactionTag secondTag = "second"_sr; - const TransactionTag manualTag = "manual"_sr; - - // An expired manual entry on each tag makes an early exit skip the active - // auto throttle regardless of hash iteration order. - for (const auto& tag : { firstTag, secondTag }) { - throttles.manualThrottleTag(UID(), tag, TransactionPriority::DEFAULT, 10.0, expiration, {}); - } - ASSERT(throttles.autoThrottleTag(UID(), firstTag, 0, 0.0, activeExpiration).present()); - throttles.manualThrottleTag(UID(), manualTag, TransactionPriority::DEFAULT, 20.0, activeExpiration, {}); - - co_await delay(2.0); - const auto rates = throttles.getClientRates(true); - ASSERT_EQ(throttles.manualThrottleCount(), 1); - ASSERT_EQ(throttles.autoThrottleCount(), 1); - ASSERT_EQ(rates.at(TransactionPriority::DEFAULT).size(), 2); - ASSERT_EQ(rates.at(TransactionPriority::BATCH).size(), 2); - ASSERT_EQ(rates.at(TransactionPriority::DEFAULT).at(firstTag).tpsRate, 0.0); - ASSERT_EQ(rates.at(TransactionPriority::BATCH).at(firstTag).tpsRate, 0.0); - ASSERT_EQ(rates.at(TransactionPriority::DEFAULT).at(manualTag).tpsRate, 20.0); - co_return; -} From 60faf29837f0280e0e541027023de9a0cb3b1460 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 15 Sep 2026 14:37:33 -0700 Subject: [PATCH 075/170] Reuse the allocator accounting helper --- flow/SystemMonitor.cpp | 14 +------------- 1 file changed, 1 insertion(+), 13 deletions(-) diff --git a/flow/SystemMonitor.cpp b/flow/SystemMonitor.cpp index a8761b829d0..945abd096b8 100644 --- a/flow/SystemMonitor.cpp +++ b/flow/SystemMonitor.cpp @@ -262,19 +262,7 @@ SystemStatistics customSystemMonitor(std::string const& eventName, StatisticsSta total_memory += FastAllocator<8192>::getTotalMemory(); total_memory += FastAllocator<16384>::getTotalMemory(); - uint64_t unused_memory = 0; - unused_memory += FastAllocator<16>::getApproximateMemoryUnused(); - unused_memory += FastAllocator<32>::getApproximateMemoryUnused(); - unused_memory += FastAllocator<64>::getApproximateMemoryUnused(); - unused_memory += FastAllocator<96>::getApproximateMemoryUnused(); - unused_memory += FastAllocator<128>::getApproximateMemoryUnused(); - unused_memory += FastAllocator<256>::getApproximateMemoryUnused(); - unused_memory += FastAllocator<512>::getApproximateMemoryUnused(); - unused_memory += FastAllocator<1024>::getApproximateMemoryUnused(); - unused_memory += FastAllocator<2048>::getApproximateMemoryUnused(); - unused_memory += FastAllocator<4096>::getApproximateMemoryUnused(); - unused_memory += FastAllocator<8192>::getApproximateMemoryUnused(); - unused_memory += FastAllocator<16384>::getApproximateMemoryUnused(); + const uint64_t unused_memory = getTotalUnusedAllocatedMemory(); if (total_memory > 0) { TraceEvent("FastAllocMemoryUsage") From d729ebe03bc00b092f5aa43ef58a0b0651a5ffc8 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 15 Sep 2026 14:41:34 -0700 Subject: [PATCH 076/170] Use typed options for standalone network diagnostics --- fdbrpc/tests/NetworkTest.h | 30 +++++++- fdbrpc/tests/networktest.cpp | 97 +++++++---------------- fdbrpc/tests/networktest.md | 32 ++++---- fdbrpc/tests/networktest_main.cpp | 124 ++++++++++++++++++++---------- fdbrpc/tests/networktest_smoke.py | 30 +++++++- 5 files changed, 183 insertions(+), 130 deletions(-) diff --git a/fdbrpc/tests/NetworkTest.h b/fdbrpc/tests/NetworkTest.h index 8bea06f75d7..f0cd7a92812 100644 --- a/fdbrpc/tests/NetworkTest.h +++ b/fdbrpc/tests/NetworkTest.h @@ -24,7 +24,6 @@ #include "fdbrpc/fdbrpc.h" #include "flow/FileIdentifier.h" -#include "flow/UnitTest.h" constexpr int WLTOKEN_NETWORKTEST = WLTOKEN_FIRST_AVAILABLE; @@ -63,6 +62,33 @@ Future networkTestServer(); Future networkTestClient(std::string const& testServers); -Future networkTestP2P(const UnitTestParameters& params, bool oneshot); +class NetworkTestIntRange { +public: + NetworkTestIntRange() = default; + NetworkTestIntRange(int low, int high); + + int get() const; + int maximum() const { return max; } + std::string toString() const; + +private: + int min = 0; + int max = 0; +}; + +struct P2PNetworkTestOptions { + std::vector listenerAddresses; + std::vector remoteAddresses; + int connectionsOut = 1; + NetworkTestIntRange requestBytes{ 50, 100 }; + NetworkTestIntRange replyBytes{ 500, 1000 }; + NetworkTestIntRange requests{ 10, 10000 }; + NetworkTestIntRange idleMilliseconds; + NetworkTestIntRange waitReadMilliseconds; + NetworkTestIntRange waitWriteMilliseconds; + double targetDuration = 0.0; +}; + +Future networkTestP2P(P2PNetworkTestOptions options, bool oneshot); #endif diff --git a/fdbrpc/tests/networktest.cpp b/fdbrpc/tests/networktest.cpp index 7138dfd4e95..c6739106a14 100644 --- a/fdbrpc/tests/networktest.cpp +++ b/fdbrpc/tests/networktest.cpp @@ -23,8 +23,9 @@ #include "flow/ActorCollection.h" #include "flow/CoroUtils.h" #include "flow/Knobs.h" -#include "flow/UnitTest.h" +#include #include +#include #include "flow/IConnection.h" @@ -244,31 +245,20 @@ Future networkTestClient(std::string const& testServers) { co_await waitForAll(clients); } -struct RandomIntRange { - int min; - int max; - - explicit(false) RandomIntRange(int low = 0, int high = 0) : min(low), max(high) {} - - // Accepts strings of the form "min:max" or "N" - // where N will be used for both min and max - explicit(false) RandomIntRange(std::string str) { - StringRef high = str; - StringRef low = high.eat(":"); - if (high.empty()) { - high = low; - } - min = low.empty() ? 0 : atol(low.toString().c_str()); - max = high.empty() ? 0 : atol(high.toString().c_str()); - if (min > max) { - std::swap(min, max); - } +NetworkTestIntRange::NetworkTestIntRange(int low, int high) : min(std::min(low, high)), max(std::max(low, high)) { + // Sampling uses an exclusive upper bound one greater than max. + if (min < 0 || max == std::numeric_limits::max()) { + throw invalid_option_value(); } +} - int get() const { return (max == 0) ? 0 : nondeterministicRandom()->randomInt(min, max + 1); } +int NetworkTestIntRange::get() const { + return (max == 0) ? 0 : nondeterministicRandom()->randomInt(min, max + 1); +} - std::string toString() const { return format("%d:%d", min, max); } -}; +std::string NetworkTestIntRange::toString() const { + return format("%d:%d", min, max); +} struct P2PNetworkTest { // Addresses to listen on @@ -278,17 +268,17 @@ struct P2PNetworkTest { // Number of outgoing connections to maintain int connectionsOut; // Message size range to send on outgoing established connections - RandomIntRange requestBytes; + NetworkTestIntRange requestBytes; // Message size to reply with on incoming established connections - RandomIntRange replyBytes; + NetworkTestIntRange replyBytes; // Number of requests/replies per session - RandomIntRange requests; + NetworkTestIntRange requests; // Delay after message send and receive are complete before closing connection - RandomIntRange idleMilliseconds; + NetworkTestIntRange idleMilliseconds; // Random delay before socket reads - RandomIntRange waitReadMilliseconds; + NetworkTestIntRange waitReadMilliseconds; // Random delay before socket writes - RandomIntRange waitWriteMilliseconds; + NetworkTestIntRange waitWriteMilliseconds; double targetDuration; double startTime; double globalStartTime; @@ -328,20 +318,11 @@ struct P2PNetworkTest { P2PNetworkTest() = default; - P2PNetworkTest(std::string listenerAddresses, - std::string remoteAddresses, - int connectionsOut, - RandomIntRange sendMsgBytes, - RandomIntRange recvMsgBytes, - RandomIntRange requests, - RandomIntRange idleMilliseconds, - RandomIntRange waitReadMilliseconds, - RandomIntRange waitWriteMilliseconds, - double targetDuration, - bool oneshot) - : connectionsOut(connectionsOut), requestBytes(sendMsgBytes), replyBytes(recvMsgBytes), requests(requests), - idleMilliseconds(idleMilliseconds), waitReadMilliseconds(waitReadMilliseconds), - waitWriteMilliseconds(waitWriteMilliseconds), targetDuration(targetDuration), oneshot(oneshot) { + P2PNetworkTest(const P2PNetworkTestOptions& options, bool oneshot) + : remotes(options.remoteAddresses), connectionsOut(options.connectionsOut), requestBytes(options.requestBytes), + replyBytes(options.replyBytes), requests(options.requests), idleMilliseconds(options.idleMilliseconds), + waitReadMilliseconds(options.waitReadMilliseconds), waitWriteMilliseconds(options.waitWriteMilliseconds), + targetDuration(options.targetDuration), oneshot(oneshot) { bytesSent = 0; bytesReceived = 0; sessionsIn = 0; @@ -349,16 +330,10 @@ struct P2PNetworkTest { connectErrors = 0; acceptErrors = 0; sessionErrors = 0; - msgBuffer = makeString(std::max(sendMsgBytes.max, recvMsgBytes.max)); - - if (!remoteAddresses.empty()) { - remotes = NetworkAddress::parseList(remoteAddresses); - } + msgBuffer = makeString(std::max(requestBytes.maximum(), replyBytes.maximum())); - if (!listenerAddresses.empty()) { - for (auto a : NetworkAddress::parseList(listenerAddresses)) { - listeners.push_back(INetworkConnections::net()->listen(a)); - } + for (auto address : options.listenerAddresses) { + listeners.push_back(INetworkConnections::net()->listen(address)); } } @@ -620,27 +595,13 @@ struct P2PNetworkTest { // Each instance // - listens on 0 or more listenerAddresses // - maintains 0 or more connectionsOut at a time, each to a random choice from remoteAddresses -// Address lists are a string of comma-separated IP:port[:tls] strings. -// -// The other arguments can be specified as "fixedValue" or "minValue:maxValue". // Each outgoing connection will live for a random requests count. // Each request will // - send a random requestBytes sized message // - wait for a random replyBytes sized response. // The client will close the connection after a random idleMilliseconds. // Reads and writes can optionally preceded by random delays, waitReadMilliseconds and waitWriteMilliseconds. -Future networkTestP2P(const UnitTestParameters& params, bool oneshot) { - P2PNetworkTest p2p(params.get("listenerAddresses").orDefault(""), - params.get("remoteAddresses").orDefault(""), - params.getInt("connectionsOut").orDefault(1), - params.get("requestBytes").orDefault("50:100"), - params.get("replyBytes").orDefault("500:1000"), - params.get("requests").orDefault("10:10000"), - params.get("idleMilliseconds").orDefault("0"), - params.get("waitReadMilliseconds").orDefault("0"), - params.get("waitWriteMilliseconds").orDefault("0"), - params.getDouble("targetDuration").orDefault(0.0), - oneshot); - +Future networkTestP2P(P2PNetworkTestOptions options, bool oneshot) { + P2PNetworkTest p2p(options, oneshot); co_await p2p.run(); } diff --git a/fdbrpc/tests/networktest.md b/fdbrpc/tests/networktest.md index 03128567ed8..901638e7ad9 100644 --- a/fdbrpc/tests/networktest.md +++ b/fdbrpc/tests/networktest.md @@ -1,7 +1,7 @@ # Network diagnostics -`fdbrpc_network_test` runs RPC and socket diagnostics using Flow and fdbrpc, -without linking the database client or server. Build it with: +`fdbrpc_network_test` runs RPC and socket diagnostics using Flow and fdbrpc. +Build it with: ```sh cmake --build build --target fdbrpc_network_test @@ -20,8 +20,8 @@ build/bin/fdbrpc_network_test --mode client --testservers 127.0.0.1:4500 \ --knob_network_test_script_mode=true ``` -The client supports comma-separated server addresses. Flow knobs retain their -existing names, including `network_test_request_size`, `network_test_reply_size`, +The client supports comma-separated server addresses. Flow knobs include +`network_test_request_size`, `network_test_reply_size`, `network_test_client_count`, and `network_test_request_count`. Script mode reports one measurement after a warmup interval and then finishes. Without a request limit or script mode, the client runs until stopped. The server runs until stopped. @@ -40,7 +40,7 @@ build/bin/fdbrpc_network_test --mode p2p \ --test_connectionsOut=2 --test_targetDuration=5 ``` -The existing `--test_*` parameters control payload size ranges, requests per +The `--test_*` options control payload size ranges, requests per connection, and delays before reads, writes, and connection close. A duration of zero means run until stopped. `--mode p2p-oneshot` tests one connection and handshake without sending the traffic workload; run its listener and connector @@ -51,15 +51,17 @@ Append `:tls` to addresses to enable TLS and supply `--tls_certificate_file`, [`contrib/mtlsbenchmark`](../../contrib/mtlsbenchmark/readme.md) for a two-process TLS example with handshake knobs. -## Migration from fdbserver +## Smoke test -| Previous invocation | Standalone invocation | -| --- | --- | -| `fdbserver -r networktestserver` | `fdbrpc_network_test --mode server` | -| `fdbserver -r networktestclient` | `fdbrpc_network_test --mode client` | -| `fdbserver -r unittests -f :/network/p2ptest` | `fdbrpc_network_test --mode p2p` | -| `fdbserver -r unittests -f :/network/p2poneshottest` | `fdbrpc_network_test --mode p2p-oneshot` | +The target's CTest entry runs bounded loopback checks for RPC traffic, P2P +sessions, handshake-only mode, and invalid options: -The networking options above retain their names. The old server roles and P2P -unit-test registrations are removed. The standalone target's CTest entry runs a -bounded loopback smoke test; it does not start an indefinite benchmark. +```sh +ctest --test-dir build -R '^unit/fdbrpc_network_test/native$' --output-on-failure +``` + +To run the same checks with temporary TLS certificates (requires OpenSSL): + +```sh +python3 fdbrpc/tests/networktest_smoke.py build/bin/fdbrpc_network_test --tls +``` diff --git a/fdbrpc/tests/networktest_main.cpp b/fdbrpc/tests/networktest_main.cpp index 60670b51045..92ccbcc801d 100644 --- a/fdbrpc/tests/networktest_main.cpp +++ b/fdbrpc/tests/networktest_main.cpp @@ -28,7 +28,6 @@ #include "flow/Platform.h" #include "flow/TLSConfig.h" #include "flow/Trace.h" -#include "flow/UnitTest.h" #include #include @@ -39,6 +38,7 @@ #include #include #include +#include #include FDB_BOOLEAN_PARAM(Randomize); @@ -52,7 +52,7 @@ enum Option { OPT_TESTSERVERS, OPT_PUBLIC_ADDRESS, OPT_LISTEN_ADDRESS, - OPT_TEST_PARAM, + OPT_P2P_OPTION, OPT_KNOB, OPT_TRACE_DIR, }; @@ -66,7 +66,7 @@ CSimpleOpt::SOption options[] = { { OPT_HELP, "-h", SO_NONE }, { OPT_PUBLIC_ADDRESS, "--public-address", SO_REQ_SEP }, { OPT_LISTEN_ADDRESS, "-l", SO_REQ_SEP }, { OPT_LISTEN_ADDRESS, "--listen-address", SO_REQ_SEP }, - { OPT_TEST_PARAM, "--test-", SO_REQ_SEP }, + { OPT_P2P_OPTION, "--test-", SO_REQ_SEP }, { OPT_KNOB, "--knob-", SO_REQ_SEP }, { OPT_TRACE_DIR, "--trace-dir", SO_REQ_SEP }, { OPT_TRACE_DIR, "--logdir", SO_REQ_SEP }, @@ -78,7 +78,8 @@ struct Options { std::string testServers; std::vector publicAddresses; std::vector listenAddresses; - UnitTestParameters testParams; + P2PNetworkTestOptions p2pOptions; + bool hasP2POptions = false; TLSConfig tlsConfig{ TLSEndpointType::SERVER }; std::string traceDir = "."; bool showHelp = false; @@ -123,50 +124,80 @@ void appendAddresses(std::vector& addresses, const char* text) { } } -bool validNonnegativeInt(std::string_view text, int maximum = std::numeric_limits::max()) { +Optional parseNonnegativeInt(std::string_view text, int maximum = std::numeric_limits::max()) { int value; const auto [end, error] = std::from_chars(text.data(), text.data() + text.size(), value); - return error == std::errc() && end == text.data() + text.size() && value >= 0 && value <= maximum; + if (error != std::errc() || end != text.data() + text.size() || value < 0 || value > maximum) { + return {}; + } + return value; } -bool validateP2PParams(const UnitTestParameters& params) { - for (const auto& [name, value] : params.params) { - bool valid = false; - if (name == "listenerAddresses" || name == "remoteAddresses") { - valid = value.empty(); - if (!value.empty()) { - const auto addresses = NetworkAddress::parseList(value); - valid = !addresses.empty() && std::all_of(addresses.begin(), addresses.end(), [](const auto& address) { - return address.isValid(); - }); - } - } else if (name == "connectionsOut") { - valid = validNonnegativeInt(value); - } else if (name == "targetDuration") { - try { - size_t end; - const auto duration = std::stod(value, &end); - valid = end == value.size() && std::isfinite(duration) && duration >= 0; - } catch (const std::exception&) { +Optional parseRange(std::string_view text) { + const auto colon = text.find(':'); + const auto low = parseNonnegativeInt(text.substr(0, colon), std::numeric_limits::max() - 1); + const auto high = colon == std::string_view::npos + ? low + : parseNonnegativeInt(text.substr(colon + 1), std::numeric_limits::max() - 1); + if (!low.present() || !high.present()) { + return {}; + } + return NetworkTestIntRange(low.get(), high.get()); +} + +bool parseP2POption(P2PNetworkTestOptions& options, const std::string& name, const std::string& value) { + if (name == "listenerAddresses" || name == "remoteAddresses") { + std::vector addresses; + if (!value.empty()) { + addresses = NetworkAddress::parseList(value); + if (addresses.empty() || !std::all_of(addresses.begin(), addresses.end(), [](const auto& address) { + return address.isValid(); + })) { + return false; } - } else if (name == "requestBytes" || name == "replyBytes" || name == "requests" || name == "idleMilliseconds" || - name == "waitReadMilliseconds" || name == "waitWriteMilliseconds") { - const auto colon = value.find(':'); - // RandomIntRange samples an inclusive upper bound by adding one. - constexpr int maximum = std::numeric_limits::max() - 1; - valid = validNonnegativeInt(value.substr(0, colon), maximum) && - (colon == std::string::npos || validNonnegativeInt(value.substr(colon + 1), maximum)); } - if (!valid) { - fprintf(stderr, "ERROR: Invalid P2P parameter --test_%s=%s\n", name.c_str(), value.c_str()); + (name == "listenerAddresses" ? options.listenerAddresses : options.remoteAddresses) = std::move(addresses); + return true; + } + if (name == "connectionsOut") { + const auto count = parseNonnegativeInt(value); + if (!count.present()) { return false; } + options.connectionsOut = count.get(); + return true; + } + if (name == "targetDuration") { + try { + size_t end; + const auto duration = std::stod(value, &end); + if (end == value.size() && std::isfinite(duration) && duration >= 0) { + options.targetDuration = duration; + return true; + } + } catch (const std::exception&) { + } + return false; + } + NetworkTestIntRange* range = nullptr; + if (name == "requestBytes") { + range = &options.requestBytes; + } else if (name == "replyBytes") { + range = &options.replyBytes; + } else if (name == "requests") { + range = &options.requests; + } else if (name == "idleMilliseconds") { + range = &options.idleMilliseconds; + } else if (name == "waitReadMilliseconds") { + range = &options.waitReadMilliseconds; + } else if (name == "waitWriteMilliseconds") { + range = &options.waitWriteMilliseconds; } - if (params.get("listenerAddresses").orDefault("").empty() && - (params.get("remoteAddresses").orDefault("").empty() || params.getInt("connectionsOut").orDefault(1) == 0)) { - fprintf(stderr, "ERROR: P2P mode requires a listener or a remote with positive connectionsOut\n"); + const auto parsed = parseRange(value); + if (!range || !parsed.present()) { return false; } + *range = parsed.get(); return true; } @@ -193,12 +224,16 @@ bool parseArgs(int argc, char** argv, Options& result, FlowKnobs& knobs) { case OPT_LISTEN_ADDRESS: appendAddresses(result.listenAddresses, args.OptionArg()); break; - case OPT_TEST_PARAM: { + case OPT_P2P_OPTION: { auto name = extractPrefixedArgument("--test", args.OptionSyntax()); if (!name.present() || name.get().empty()) { return false; } - result.testParams.set(name.get(), args.OptionArg()); + if (!parseP2POption(result.p2pOptions, name.get(), args.OptionArg())) { + fprintf(stderr, "ERROR: Invalid P2P option --test_%s=%s\n", name.get().c_str(), args.OptionArg()); + return false; + } + result.hasP2POptions = true; break; } case OPT_KNOB: { @@ -268,9 +303,14 @@ bool parseArgs(int argc, char** argv, Options& result, FlowKnobs& knobs) { return false; } if (result.mode == "p2p" || result.mode == "p2p-oneshot") { - return validateP2PParams(result.testParams); + if (result.p2pOptions.listenerAddresses.empty() && + (result.p2pOptions.remoteAddresses.empty() || result.p2pOptions.connectionsOut == 0)) { + fprintf(stderr, "ERROR: P2P mode requires a listener or a remote with positive connectionsOut\n"); + return false; + } + return true; } - if (!result.testParams.params.empty()) { + if (result.hasP2POptions) { fprintf(stderr, "ERROR: --test_NAME parameters require a P2P mode\n"); return false; } @@ -342,7 +382,7 @@ int main(int argc, char** argv) { } else if (opts.mode == "client") { work.push_back(networkTestClient(opts.testServers)); } else { - work.push_back(networkTestP2P(opts.testParams, opts.mode == "p2p-oneshot")); + work.push_back(networkTestP2P(opts.p2pOptions, opts.mode == "p2p-oneshot")); } Future done = stopNetworkAfter(waitForAny(work)); g_network->run(); diff --git a/fdbrpc/tests/networktest_smoke.py b/fdbrpc/tests/networktest_smoke.py index b102f60229f..956286d620b 100644 --- a/fdbrpc/tests/networktest_smoke.py +++ b/fdbrpc/tests/networktest_smoke.py @@ -160,12 +160,27 @@ def p2p(self): "p2p", "--test_listenerAddresses=" + address, "--test_remoteAddresses=" + address, - "--test_connectionsOut=1", - "--test_requests=1", + "--test_connectionsOut=2", + "--test_requestBytes=32:48", + "--test_replyBytes=96:64", + "--test_requests=2:3", + "--test_idleMilliseconds=1:0", + "--test_waitReadMilliseconds=1:2", + "--test_waitWriteMilliseconds=0:1", "--test_targetDuration=2", ], ) as (process, output): text = self.finish(process, output) + for expected in ( + "2 outgoing connections", + "Request size: 32:48", + "Response size: 64:96", + "Requests per outgoing session: 2:3", + "Delay before socket read: 1:2", + "Delay before socket write: 0:1", + "Delay before session close: 0:1", + ): + assert expected in text, text for direction in ("in", "out"): rates = re.findall(r"([0-9.]+)/s completed sessions " + direction, text) assert any(float(rate) > 0 for rate in rates), text @@ -192,8 +207,16 @@ def handshake(self): def invalid_arguments(self): server = ["--mode", "server", "-p", self.address(unused_port())] + p2p = ["--mode", "p2p", "--test_listenerAddresses=" + self.address(unused_port())] cases = [ [], + ["--mode", "p2p"], + p2p + ["--test_unknown=1"], + p2p + ["--test_connectionsOut=-1"], + p2p + ["--test_connectionsOut=invalid"], + p2p + ["--test_requestBytes=1::2"], + p2p + ["--test_replyBytes=2147483647"], + p2p + ["--test_targetDuration=nan"], server + ["--knob_network_test_script_mode=invalid"], server + ["--knob_not_a_network_test_knob=1"], server + ["--not-a-network-test-option"], @@ -203,7 +226,8 @@ def invalid_arguments(self): process, output, ): - self.finish(process, output, seconds=5, success=False) + text = self.finish(process, output, seconds=5, success=False) + assert "ERROR:" in text, text def main(): From a2fe9f54be4711bb0af55f35090833af3b35f69b Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 15 Sep 2026 14:49:37 -0700 Subject: [PATCH 077/170] Reuse file-transfer metadata without truncating file sizes --- fdbrpc/FileTransfer.cpp | 23 ++++++----------------- 1 file changed, 6 insertions(+), 17 deletions(-) diff --git a/fdbrpc/FileTransfer.cpp b/fdbrpc/FileTransfer.cpp index 479507d7439..ba2712baf88 100644 --- a/fdbrpc/FileTransfer.cpp +++ b/fdbrpc/FileTransfer.cpp @@ -19,6 +19,7 @@ */ #ifdef FLOW_GRPC_ENABLED #include +#include #include "FileTransfer.h" #include "flow/IRandom.h" @@ -144,21 +145,9 @@ std::optional FileTransferClient::DownloadFile(const std::string& filena const std::string& output_filename, bool verify) { - uint32_t expected_crc = 0; - uint32_t expected_size = 0; - { - fdbrpc::GetFileInfoRequest request; - grpc::ClientContext context; - request.set_file_name(filename); - request.set_get_crc_checksum(verify); - request.set_get_size(true); - fdbrpc::GetFileInfoReply response; - auto res = stub_->GetFileInfo(&context, request, &response); - if (!res.ok()) { - return std::nullopt; - } - expected_crc = response.crc_checksum(); - expected_size = response.file_size(); + const auto fileInfo = GetFileInfo(filename, verify); + if (!fileInfo.has_value()) { + return std::nullopt; } fdbrpc::DownloadRequest request; @@ -188,13 +177,13 @@ std::optional FileTransferClient::DownloadFile(const std::string& filena // Close file after writing output_file.close(); - failed = failed || (bytes_read != expected_size); + failed = failed || std::cmp_not_equal(bytes_read, fileInfo->file_size()); // Verify checksum if (!failed && verify) { std::ifstream output_file_reader(output_filename); uint32_t actual_crc = crc32_checksum_ifstream(&output_file_reader); - failed = (actual_crc != expected_crc); + failed = (actual_crc != fileInfo->crc_checksum()); } // Check final gRPC status From e80a78e0149b5fff71faa72422de9f961dc5a16c Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 15 Sep 2026 14:53:36 -0700 Subject: [PATCH 078/170] Include configuration in backup container cache identity --- fdbclient/BackupContainer.cpp | 22 +++++++++++----------- 1 file changed, 11 insertions(+), 11 deletions(-) diff --git a/fdbclient/BackupContainer.cpp b/fdbclient/BackupContainer.cpp index 5f58605e418..43866175239 100644 --- a/fdbclient/BackupContainer.cpp +++ b/fdbclient/BackupContainer.cpp @@ -33,6 +33,7 @@ #include "fdbclient/RunRYWTransaction.h" #include #include +#include namespace IBackupFile_impl { @@ -266,7 +267,14 @@ Reference IBackupContainer::openContainer(const std::string& u const Optional& proxy, const Optional& encryptionKeyFileName, int encryptionBlockSize) { - static std::map> m_cache; + using CacheKey = std::tuple, Optional, int>; + static std::map> m_cache; + + Optional blobstoreProxy; + if (isBlobstoreUrl(url)) { + // The backup-agent fallback is part of the effective connection configuration. + blobstoreProxy = proxy.present() ? proxy : fileBackupAgentProxy; + } // In simulation, disable caching for blobstore:// URLs to prevent cross-process connection issues. // @@ -286,7 +294,8 @@ Reference IBackupContainer::openContainer(const std::string& u // Use a reference to the cache entry (for automatic cache population) unless we're skipping cache Reference r_local; - Reference& r = skipCache ? r_local : m_cache[url]; + Reference& r = + skipCache ? r_local : m_cache[{ url, blobstoreProxy, encryptionKeyFileName, encryptionBlockSize }]; if (r) { return r; } @@ -297,15 +306,6 @@ Reference IBackupContainer::openContainer(const std::string& u r = makeReference(url, encryptionKeyFileName, encryptionBlockSize); } else if (u.startsWith("blobstore://"_sr)) { std::string resource; - Optional blobstoreProxy; - - // If no proxy is passed down to the openContainer method, try to fallback to the - // fileBackupAgentProxy which is a global variable and will be set for the backup_agent. - if (proxy.present()) { - blobstoreProxy = proxy.get(); - } else if (fileBackupAgentProxy.present()) { - blobstoreProxy = fileBackupAgentProxy.get(); - } // The URL parameters contain blobstore endpoint tunables as well as possible backup-specific options. IBlobStoreEndpoint::ParametersT backupParams; From a4c20d518493bc1adac7610dd0db92c4f4db9da2 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 15 Sep 2026 14:56:56 -0700 Subject: [PATCH 079/170] Wait for native CDC test fetch startup --- fdbserver/cdcproxy/CDCProxy.cpp | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index 4214fa7e454..03e252ce1b2 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -2431,7 +2431,10 @@ class CDCProxyPrefetchTest { second->tagIntervals.back().bufferedThrough = next - 1; Promise ready; auto cursor = makeReference(ready.getFuture()); + auto fetchStarted = cursor->onFetchStarted(); auto work = test.proxy.bufferTagCursor(test.tag, 100, cursor, Never(), Prefetch::True); + auto start = co_await race(fetchStarted, work); + ASSERT_EQ(start.index(), 0); co_await delay(0); ASSERT_EQ(cursor->fetchCount(), 1); auto waiter = test.proxy.waitForBufferedVersion(second, next); @@ -2481,7 +2484,10 @@ class CDCProxyPrefetchTest { auto waiter = test.proxy.waitForBufferedVersion(stream, 100); ASSERT(awakened.isReady()); // No interest -> real demand still wakes a dormant tag. auto cursor = makeReference(Never()); + auto fetchStarted = cursor->onFetchStarted(); auto work = test.proxy.bufferTagCursor(test.tag, 100, cursor, Never(), Prefetch::False); + auto start = co_await race(fetchStarted, work); + ASSERT_EQ(start.index(), 0); co_await delay(0); ASSERT_EQ(cursor->fetchCount(), 1); ASSERT(!work.isReady()); @@ -2681,7 +2687,10 @@ class CDCProxyPrefetchTest { Promise ready; Promise generationChanged; auto cursor = makeReference(ready.getFuture()); + auto fetchStarted = cursor->onFetchStarted(); auto work = test.proxy.bufferTagCursor(test.tag, 100, cursor, generationChanged.getFuture(), Prefetch::True); + auto start = co_await race(fetchStarted, work); + ASSERT_EQ(start.index(), 0); co_await delay(0); ASSERT_EQ(cursor->fetchCount(), 1); ASSERT(stream->readAhead.claimedBy(test.tag.getPtr())); From fbee775a69ca22065f517d67bc10a1d7c8b9207b Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 15 Sep 2026 15:38:36 -0700 Subject: [PATCH 080/170] Propagate broken promises from CDC proxy rebalancing --- fdbserver/clustercontroller/ClusterController.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index ce9e0960034..6283a246e4a 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -2519,7 +2519,7 @@ Future rebalanceCDCProxyAssignments(ClusterControllerData* self) { }; co_await rebalanceNativeCdcProxyAssignments(self->db.db, std::move(available), std::move(stillEligible)); } catch (Error& e) { - if (e.code() == error_code_actor_cancelled) { + if (e.code() == error_code_actor_cancelled || e.code() == error_code_broken_promise) { throw; } // An ambiguous commit is reconciled by the assignment monitor; the next scheduled pass may try again. From 7ee5dedee01f0146979ad6601885c7756f0dc052 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 15 Sep 2026 16:08:25 -0700 Subject: [PATCH 081/170] Correct CDC C API overview for multiple ranges --- documentation/sphinx/source/api-c.rst | 4 ++-- fdbclient/NativeCdc.cpp | 4 ++-- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/documentation/sphinx/source/api-c.rst b/documentation/sphinx/source/api-c.rst index 7af87e31e04..db35a55756b 100644 --- a/documentation/sphinx/source/api-c.rst +++ b/documentation/sphinx/source/api-c.rst @@ -564,8 +564,8 @@ An |database-blurb1| Modifications to a database are performed via transactions. CDC --- -CDC exposes durable, named streams of committed mutations for one half-open -user-key range. New stream registration requires CDC admission +CDC exposes durable, named streams of committed mutations for an immutable union of +half-open user-key ranges. New stream registration requires CDC admission to be enabled on the cluster. Listing, removal, consumer creation, resume, consume, and acknowledgement remain available for already durable streams while new admission is disabled so that callers can drain or remove them. diff --git a/fdbclient/NativeCdc.cpp b/fdbclient/NativeCdc.cpp index c37c6c0c947..e14d7da6ca4 100644 --- a/fdbclient/NativeCdc.cpp +++ b/fdbclient/NativeCdc.cpp @@ -912,8 +912,8 @@ Future getNativeCdcStatus(Database cx) { stream.info.streamId = streamId; std::sort(stream.tags.begin(), stream.tags.end()); stream.tags.erase(std::unique(stream.tags.begin(), stream.tags.end()), stream.tags.end()); - if (stream.info.name.empty() || stream.info.ranges.empty() || stream.info.minVersion == invalidVersion || - stream.tags.empty()) { + if (stream.info.name.empty() || stream.info.ranges.empty() || + stream.info.minVersion == invalidVersion || stream.tags.empty()) { result.metadataComplete = false; } for (const Tag& tag : stream.tags) { From c8b31b698aacf39cd53e99ef6edd1948039251f6 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 15 Sep 2026 17:26:42 -0700 Subject: [PATCH 082/170] Explain native CDC committed-frontier wait guards --- fdbserver/tlog/TLogServer.cpp | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/fdbserver/tlog/TLogServer.cpp b/fdbserver/tlog/TLogServer.cpp index e880cf3d6cf..9b9f476f46f 100644 --- a/fdbserver/tlog/TLogServer.cpp +++ b/fdbserver/tlog/TLogServer.cpp @@ -2319,13 +2319,18 @@ Future tLogPeekMessages(PromiseType replyPromise, co_await delay(0, TaskPriority::TLogSpilledPeekReply); } + // Capped CDC peeks feed a proxy that can only consume the committed prefix; uncapped peeks and other + // tag consumers retain their existing behavior. Nonblocking requests must return without waiting, and + // spill-only reads must drain persisted data without waiting for live commits. A stopped TLog cannot + // advance its frontier, and finite recovery/tag-history ranges must drain without requiring new commits. + // Only an unbounded live-tail peek can use commit progress to wake this wait. const bool waitForCommittedFrontier = replyByteLimit > 0 && reqTag.locality == tagLocalityCDC && !reqReturnIfBlocked && !reqOnlySpilled && !logData->stopped() && (!reqEnd.present() || reqEnd.get() == std::numeric_limits::max()); if (waitForCommittedFrontier && poppedVersion(logData, reqTag) <= reqBegin) { // A speculative message (or an empty tail) cannot advance native CDC until this frontier reaches begin. // Waiting here lets commit progress wake the same capped peek instead of returning a stale frontier to - // the proxy. Keep finite/recovery, nonblocking, and uncapped peeks on their existing paths. + // the proxy. co_await waitForCommittedVersion(logData, reqBegin, now() + SERVER_KNOBS->BLOCKING_PEEK_TIMEOUT); } From ceaa930d283592fac62b39b13ac9afa33272521e Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 15 Sep 2026 18:38:14 -0700 Subject: [PATCH 083/170] Handle completed CDC consumes in retry validation --- fdbserver/cdcproxy/CDCProxy.cpp | 31 +++++++++++++++++++++++ fdbserver/workloads/NativeCdcEndToEnd.cpp | 13 ++++++---- 2 files changed, 39 insertions(+), 5 deletions(-) diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index 1bf9cb0d46d..8d7c14c0443 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -2058,6 +2058,37 @@ Future cdcProxyServer(CDCProxyInterface proxy, } } +TEST_CASE("/NativeCDC/ConsumeLeaseSupersession") { + const UID consumerId(1, 2); + auto lease = makeReference(consumerId); + ASSERT(lease->belongsTo(consumerId)); + ASSERT(!lease->belongsTo(UID(3, 4))); + ASSERT(!lease->belongsTo(Optional())); + ASSERT(!makeReference(UID())->belongsTo(UID())); + + Promise pendingReply; + Future original = lease->waitForReply(pendingReply.getFuture()); + ASSERT(!original.isReady()); + lease->supersede(); + ASSERT(original.isReady() && original.isError()); + ASSERT_EQ(original.getError().code(), error_code_request_maybe_delivered); + return Void(); +} + +TEST_CASE("/NativeCDC/ConsumeLeaseCompletedReply") { + auto lease = makeReference(UID(1, 2)); + Promise pendingReply; + Future original = lease->waitForReply(pendingReply.getFuture()); + CDCConsumeReply reply; + reply.lastConsumedVersion = 10; + pendingReply.send(reply); + ASSERT(original.isReady() && !original.isError()); + lease->supersede(); + ASSERT(!original.isError()); + ASSERT_EQ(original.get().lastConsumedVersion, 10); + return Void(); +} + TEST_CASE("/NativeCDC/ProxyMutationFiltering") { const KeyRangeRef keys("c"_sr, "m"_sr); diff --git a/fdbserver/workloads/NativeCdcEndToEnd.cpp b/fdbserver/workloads/NativeCdcEndToEnd.cpp index 57bd86be300..ce7e6374b5a 100644 --- a/fdbserver/workloads/NativeCdcEndToEnd.cpp +++ b/fdbserver/workloads/NativeCdcEndToEnd.cpp @@ -1128,16 +1128,19 @@ class NativeCdcEndToEndWorkload : public TestWorkload { first.cancel(); co_await waitForNoActiveConsumes(cx, streamId, proxy); - // Reproduce the server-side overlap without relying on the timing of a socket reset: both requests belong - // to one consumer, so the retry must supersede the pending metadata read instead of failing exclusivity. + // The first request may finish before the retry reaches the proxy. A pending request is superseded, while + // an already-completed request retains its reply; either ordering must allow the same consumer to retry. const UID consumerId = deterministicRandom()->randomUniqueID(); Future> original = proxy->consume.tryGetReply(CDCConsumeRequest(currentCursor, consumerId)); Future> retry = proxy->consume.tryGetReply(CDCConsumeRequest(currentCursor, consumerId)); - const ErrorOr superseded = co_await timeoutError(original, operationTimeout); - ASSERT(superseded.isError()); - ASSERT_EQ(superseded.getError().code(), error_code_request_maybe_delivered); + const ErrorOr firstReply = co_await timeoutError(original, operationTimeout); + if (firstReply.isError()) { + ASSERT_EQ(firstReply.getError().code(), error_code_request_maybe_delivered); + } else { + ASSERT_GE(firstReply.get().lastConsumedVersion, currentCursor.lastConsumedVersion); + } co_await timeoutError(throwErrorOr(retry), operationTimeout); co_await waitForNoActiveConsumes(cx, streamId, proxy); } From 730901523a1ce3e58eb3e8ac2f1f804742f06a56 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 15 Sep 2026 18:45:28 -0700 Subject: [PATCH 084/170] Return zero for zero-length encrypted-file reads --- fdbrpc/AsyncFileEncrypted.cpp | 34 ++++++++++++++++++++++++++++++++++ 1 file changed, 34 insertions(+) diff --git a/fdbrpc/AsyncFileEncrypted.cpp b/fdbrpc/AsyncFileEncrypted.cpp index d70abbf9b1d..171a513426d 100644 --- a/fdbrpc/AsyncFileEncrypted.cpp +++ b/fdbrpc/AsyncFileEncrypted.cpp @@ -65,6 +65,9 @@ class AsyncFileEncryptedImpl { } static Future read(Reference self, void* data, int length, int64_t offset) { + if (length == 0) { + co_return 0; + } if (self->fileSize == -1) { int64_t rawSize = co_await self->file->size(); self->fileSize = AsyncFileEncrypted::rawToLogicalSize(rawSize, self->encryptionBlockSize); @@ -299,3 +302,34 @@ TEST_CASE("fdbrpc/AsyncFileEncrypted") { } ASSERT(writeBuffer == readBuffer); } + +TEST_CASE("fdbrpc/AsyncFileEncrypted/ZeroLengthRead") { + const int encryptionBlockSize = 4096; + const int bytes = 2 * encryptionBlockSize + 1; + StreamCipherKey::initializeGlobalRandomTestKey(); + int flags = IAsyncFile::OPEN_READWRITE | IAsyncFile::OPEN_CREATE | IAsyncFile::OPEN_ATOMIC_WRITE_AND_CREATE | + IAsyncFile::OPEN_UNBUFFERED | IAsyncFile::OPEN_UNCACHED | IAsyncFile::OPEN_NO_AIO; + Reference rawFile = co_await IAsyncFileSystem::filesystem()->open( + joinPath(params.getDataDir(), "test-encrypted-file-zero-length"), flags, 0600); + std::vector readBuffer(encryptionBlockSize, 0xa5); + const auto untouchedBuffer = readBuffer; + Reference file = + makeReference(rawFile, AsyncFileEncrypted::Mode::READ_ONLY, encryptionBlockSize); + int bytesRead = co_await file->read(readBuffer.data(), 0, 0); + ASSERT_EQ(bytesRead, 0); + ASSERT(readBuffer == untouchedBuffer); + + file = makeReference(rawFile, AsyncFileEncrypted::Mode::APPEND_ONLY, encryptionBlockSize); + std::vector writeBuffer(bytes, 0x5a); + co_await file->write(writeBuffer.data(), bytes, 0); + co_await file->sync(); + file = makeReference(rawFile, AsyncFileEncrypted::Mode::READ_ONLY, encryptionBlockSize); + for (int offset : { 0, 1, encryptionBlockSize, encryptionBlockSize + 1, bytes, bytes + 1 }) { + bytesRead = co_await file->read(readBuffer.data(), 0, offset); + ASSERT_EQ(bytesRead, 0); + ASSERT(readBuffer == untouchedBuffer); + } + bytesRead = co_await file->read(readBuffer.data(), 1, 0); + ASSERT_EQ(bytesRead, 1); + ASSERT_EQ(readBuffer[0], writeBuffer[0]); +} From ab3986f6e67500018e36e9bc74c4acb9dc599065 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 15 Sep 2026 23:25:08 -0700 Subject: [PATCH 085/170] Remove redundant completed CDC reply test --- fdbserver/cdcproxy/CDCProxy.cpp | 14 -------------- 1 file changed, 14 deletions(-) diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index 8d7c14c0443..99e67b89ba4 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -2075,20 +2075,6 @@ TEST_CASE("/NativeCDC/ConsumeLeaseSupersession") { return Void(); } -TEST_CASE("/NativeCDC/ConsumeLeaseCompletedReply") { - auto lease = makeReference(UID(1, 2)); - Promise pendingReply; - Future original = lease->waitForReply(pendingReply.getFuture()); - CDCConsumeReply reply; - reply.lastConsumedVersion = 10; - pendingReply.send(reply); - ASSERT(original.isReady() && !original.isError()); - lease->supersede(); - ASSERT(!original.isError()); - ASSERT_EQ(original.get().lastConsumedVersion, 10); - return Void(); -} - TEST_CASE("/NativeCDC/ProxyMutationFiltering") { const KeyRangeRef keys("c"_sr, "m"_sr); From 69c43c0a59efc7c9dd9c9850776f1750a76646cc Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Wed, 16 Sep 2026 00:29:08 -0700 Subject: [PATCH 086/170] Move actor pattern timing cases into flow microbenchmarks --- fdbrpc/FlowTests.cpp | 373 +++++++----------------------- flow/bench/BenchActorPatterns.cpp | 373 ++++++++++++++++++++++++++++++ flow/bench/README.md | 18 ++ 3 files changed, 478 insertions(+), 286 deletions(-) create mode 100644 flow/bench/BenchActorPatterns.cpp diff --git a/fdbrpc/FlowTests.cpp b/fdbrpc/FlowTests.cpp index d8d89c6761c..3e3bf77441f 100644 --- a/fdbrpc/FlowTests.cpp +++ b/fdbrpc/FlowTests.cpp @@ -118,18 +118,6 @@ void onReady(FutureStream&& f, Func&& func, ErrFunc&& errFunc) { } } -static Future emptyVoidActor(Uncancellable = Uncancellable()) { - co_return; -} - -static Future emptyActor() { - return Void(); -} - -static Future oneWaitVoidActor(Future f, Uncancellable = Uncancellable()) { - co_await f; -} - static Future oneWaitActor(Future f) { co_await f; } @@ -1027,301 +1015,114 @@ TEST_CASE("/flow/flow/chooseTwoActor") { return Void(); } -TEST_CASE("#flow/flow/perf/actor patterns") { - double start; - int N = 1000000; - - start = timer(); - for (int i = 0; i < N; i++) - emptyVoidActor(); - printf("emptyVoidActor(): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - - start = timer(); - for (int i = 0; i < N; i++) { - emptyActor(); - } - printf("emptyActor(): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - - Promise neverSet; - Future never = neverSet.getFuture(); - Future already = Void(); - - start = timer(); - for (int i = 0; i < N; i++) - oneWaitVoidActor(already); - printf("oneWaitVoidActor(already): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - - /*start = timer(); - for (int i = 0; i < N; i++) - oneWaitVoidActor(never); - printf("oneWaitVoidActor(never): %0.1f M/sec\n", N / 1e6 / (timer() - start));*/ - - { - start = timer(); - for (int i = 0; i < N; i++) { - Future f = oneWaitActor(already); - ASSERT(f.isReady()); - } - printf("oneWaitActor(already): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - start = timer(); - for (int i = 0; i < N; i++) { - Future f = oneWaitActor(never); - ASSERT(!f.isReady()); - } - printf("(cancelled) oneWaitActor(never): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - start = timer(); - for (int i = 0; i < N; i++) { - Promise p; - Future f = oneWaitActor(p.getFuture()); - p.send(Void()); - ASSERT(f.isReady()); - } - printf("oneWaitActor(after): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - std::vector> pipe(N); - std::vector> out(N); - start = timer(); - for (int i = 0; i < N; i++) { - out[i] = oneWaitActor(pipe[i].getFuture()); - } - for (int i = 0; i < N; i++) { - pipe[i].send(Void()); - ASSERT(out[i].isReady()); - } - printf("oneWaitActor(fifo): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - std::vector> pipe(N); - std::vector> out(N); - start = timer(); - for (int i = 0; i < N; i++) { - out[i] = oneWaitActor(pipe[i].getFuture()); - } - for (int i = N - 1; i >= 0; i--) { - pipe[i].send(Void()); - ASSERT(out[i].isReady()); - } - printf("oneWaitActor(lifo): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - start = timer(); - for (int i = 0; i < N; i++) { - Future f = chooseTwoActor(already, already); - ASSERT(f.isReady()); - } - printf("chooseTwoActor(already, already): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - start = timer(); - for (int i = 0; i < N; i++) { - Future f = chooseTwoActor(already, never); - ASSERT(f.isReady()); - } - printf("chooseTwoActor(already, never): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - start = timer(); - for (int i = 0; i < N; i++) { - Future f = chooseTwoActor(never, already); - ASSERT(f.isReady()); - } - printf("chooseTwoActor(never, already): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - start = timer(); - for (int i = 0; i < N; i++) { - Future f = chooseTwoActor(never, never); - ASSERT(!f.isReady()); - } - printf("(cancelled) chooseTwoActor(never, never): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - start = timer(); - for (int i = 0; i < N; i++) { - Promise p; - Future f = chooseTwoActor(p.getFuture(), never); - p.send(Void()); - ASSERT(f.isReady()); - } - printf("chooseTwoActor(after, never): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - std::vector> pipe(N); - std::vector> out(N); - start = timer(); - for (int i = 0; i < N; i++) { - out[i] = chooseTwoActor(pipe[i].getFuture(), never); - } - for (int i = 0; i < N; i++) { - pipe[i].send(Void()); - ASSERT(out[i].isReady()); - } - printf("chooseTwoActor(fifo, never): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - std::vector> pipe(N); - std::vector> out(N); - start = timer(); - for (int i = 0; i < N; i++) { - out[i] = chooseTwoActor(pipe[i].getFuture(), pipe[i].getFuture()); - } - for (int i = 0; i < N; i++) { - pipe[i].send(Void()); - ASSERT(out[i].isReady()); - } - printf("chooseTwoActor(fifo, fifo): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - std::vector> pipe(N); - std::vector> out(N); - start = timer(); - for (int i = 0; i < N; i++) { - out[i] = chooseTwoActor(chooseTwoActor(pipe[i].getFuture(), never), never); - } - for (int i = 0; i < N; i++) { - pipe[i].send(Void()); - ASSERT(out[i].isReady()); - } - printf("chooseTwoActor^2((fifo, never), never): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - } +TEST_CASE("/flow/flow/actor patterns/wait") { + Future ready = oneWaitActor(Void()); + ASSERT(ready.isReady() && !ready.isError()); + Promise input; { - start = timer(); - for (int i = 0; i < N; i++) { - Promise p; - Future f = oneWaitActor(chooseTwoActor(p.getFuture(), never)); - p.send(Void()); - ASSERT(f.isReady()); - } - printf("oneWaitActor(chooseTwoActor(after, never)): %0.1f M/sec\n", N / 1e6 / (timer() - start)); + Future cancelled = oneWaitActor(input.getFuture()); + ASSERT(!cancelled.isReady() && input.getFutureReferenceCount() > 0); } + ASSERT(input.getFutureReferenceCount() == 0); { - std::vector> pipe(N); - std::vector> out(N); - start = timer(); - for (int i = 0; i < N; i++) { - out[i] = oneWaitActor(chooseTwoActor(pipe[i].getFuture(), never)); - } - for (int i = 0; i < N; i++) { - pipe[i].send(Void()); - ASSERT(out[i].isReady()); - } - printf("oneWaitActor(chooseTwoActor(fifo, never)): %0.1f M/sec\n", N / 1e6 / (timer() - start)); + Future completed = oneWaitActor(input.getFuture()); + ASSERT(!completed.isReady()); + input.send(Void()); + ASSERT(completed.isReady() && !completed.isError()); } + ASSERT(input.getFutureReferenceCount() == 0); + return Void(); +} +TEST_CASE("/flow/flow/actor patterns/race") { + Promise pending; { - start = timer(); - for (int i = 0; i < N; i++) { - Promise p; - Future f = chooseTwoActor(p.getFuture(), never); - Future a = oneWaitActor(f); - Future b = oneWaitActor(f); - p.send(Void()); - ASSERT(f.isReady()); - } - printf("2xoneWaitActor(chooseTwoActor(after, never)): %0.2f M/sec\n", N / 1e6 / (timer() - start)); + Future ready = Void(); + Future bothReady = chooseTwoActor(ready, ready); + Future firstReady = chooseTwoActor(ready, pending.getFuture()); + Future secondReady = chooseTwoActor(pending.getFuture(), ready); + ASSERT(bothReady.isReady() && !bothReady.isError()); + ASSERT(firstReady.isReady() && !firstReady.isError()); + ASSERT(secondReady.isReady() && !secondReady.isError()); } + ASSERT(pending.getFutureReferenceCount() == 0); { - std::vector> pipe(N); - std::vector> out1(N); - std::vector> out2(N); - start = timer(); - for (int i = 0; i < N; i++) { - Future f = chooseTwoActor(pipe[i].getFuture(), never); - out1[i] = oneWaitActor(f); - out2[i] = oneWaitActor(f); - } - for (int i = 0; i < N; i++) { - pipe[i].send(Void()); - ASSERT(out2[i].isReady()); - } - printf("2xoneWaitActor(chooseTwoActor(fifo, never)): %0.2f M/sec\n", N / 1e6 / (timer() - start)); + Future cancelled = chooseTwoActor(pending.getFuture(), pending.getFuture()); + ASSERT(!cancelled.isReady() && pending.getFutureReferenceCount() > 0); } + ASSERT(pending.getFutureReferenceCount() == 0); { - std::vector> pipe(N); - std::vector> out1(N); - std::vector> out2(N); - start = timer(); - for (int i = 0; i < N; i++) { - Future f = chooseTwoActor(oneWaitActor(pipe[i].getFuture()), never); - out1[i] = oneWaitActor(f); - out2[i] = oneWaitActor(f); - } - for (int i = 0; i < N; i++) { - pipe[i].send(Void()); - ASSERT(out2[i].isReady()); - } - printf("2xoneWaitActor(chooseTwoActor(oneWaitActor(fifo), never)): %0.2f M/sec\n", N / 1e6 / (timer() - start)); + Future sharedInput = chooseTwoActor(pending.getFuture(), pending.getFuture()); + ASSERT(!sharedInput.isReady()); + pending.send(Void()); + ASSERT(sharedInput.isReady() && !sharedInput.isError()); } + ASSERT(pending.getFutureReferenceCount() == 0); + return Void(); +} - { - std::vector> pipe(N); - std::vector> out1(N); - std::vector> out2(N); - start = timer(); - for (int i = 0; i < N; i++) { - g_cheese = pipe[i].getFuture(); - Future f = chooseTwoActor(cheeseWaitActor(), never); - g_cheese = f; - out1[i] = cheeseWaitActor(); - out2[i] = cheeseWaitActor(); +TEST_CASE("/flow/flow/actor patterns/composition") { + const int batchSize = 4; + for (bool lifo : { false, true }) { + Promise never; + std::vector> inputs(batchSize); + { + std::vector> waits(batchSize); + std::vector> sharedRaces(batchSize); + std::vector> nestedRaces(batchSize); + std::vector> firstOutputs(batchSize); + std::vector> secondOutputs(batchSize); + for (int i = 0; i < batchSize; ++i) { + waits[i] = oneWaitActor(inputs[i].getFuture()); + sharedRaces[i] = chooseTwoActor(inputs[i].getFuture(), inputs[i].getFuture()); + nestedRaces[i] = + chooseTwoActor(chooseTwoActor(inputs[i].getFuture(), never.getFuture()), never.getFuture()); + Future fanout = chooseTwoActor(oneWaitActor(inputs[i].getFuture()), never.getFuture()); + firstOutputs[i] = oneWaitActor(fanout); + secondOutputs[i] = oneWaitActor(fanout); + ASSERT(!waits[i].isReady() && !sharedRaces[i].isReady() && !nestedRaces[i].isReady() && + !firstOutputs[i].isReady() && !secondOutputs[i].isReady()); + } + for (int i = 0; i < batchSize; ++i) { + const int index = lifo ? batchSize - 1 - i : i; + inputs[index].send(Void()); + for (int j = 0; j < batchSize; ++j) { + const bool completed = lifo ? j >= index : j <= index; + ASSERT(waits[j].isReady() == completed && sharedRaces[j].isReady() == completed && + nestedRaces[j].isReady() == completed && firstOutputs[j].isReady() == completed && + secondOutputs[j].isReady() == completed); + } + ASSERT(!waits[index].isError() && !sharedRaces[index].isError() && !nestedRaces[index].isError() && + !firstOutputs[index].isError() && !secondOutputs[index].isError()); + } } - for (int i = 0; i < N; i++) { - pipe[i].send(Void()); - ASSERT(out2[i].isReady()); + ASSERT(never.getFutureReferenceCount() == 0); + for (const auto& input : inputs) { + ASSERT(input.getFutureReferenceCount() == 0); } - printf("2xcheeseActor(chooseTwoActor(cheeseActor(fifo), never)): %0.2f M/sec\n", N / 1e6 / (timer() - start)); - } - - { - PromiseStream data; - start = timer(); - Future sum = sumActor(data.getFuture()); - for (int i = 0; i < N; i++) - data.send(1); - data.sendError(end_of_stream()); - ASSERT(sum.get() == N); - printf("sumActor: %0.2f M/sec\n", N / 1e6 / (timer() - start)); } + return Void(); +} +TEST_CASE("/flow/flow/actor patterns/global input") { + Promise original, replacement; { - start = timer(); - std::vector> ps(3); - std::vector> fs(3); - - for (int i = 0; i < N; i++) { - ps.clear(); - ps.resize(3); - for (int j = 0; j < ps.size(); j++) - fs[j] = ps[j].getFuture(); - - Future q = quorum(fs, 2); - for (auto& p : ps) - p.send(Void()); - } - printf("quorum(2/3): %0.2f M/sec\n", N / 1e6 / (timer() - start)); - } - + g_cheese = original.getFuture(); + Future first = cheeseWaitActor(); + g_cheese = replacement.getFuture(); + Future second = cheeseWaitActor(); + g_cheese = Future(); + ASSERT(!first.isReady() && !second.isReady()); + original.send(Void()); + ASSERT(first.isReady() && !first.isError() && !second.isReady()); + replacement.send(Void()); + ASSERT(second.isReady() && !second.isError()); + } + ASSERT(original.getFutureReferenceCount() == 0 && replacement.getFutureReferenceCount() == 0); return Void(); } diff --git a/flow/bench/BenchActorPatterns.cpp b/flow/bench/BenchActorPatterns.cpp new file mode 100644 index 00000000000..78ee601af0f --- /dev/null +++ b/flow/bench/BenchActorPatterns.cpp @@ -0,0 +1,373 @@ +/* + * BenchActorPatterns.cpp + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "benchmark/benchmark.h" + +#include "flow/ThreadHelper.h" +#include "flow/genericactors.h" + +#include + +namespace { + +Future emptyUncancellable(Uncancellable = {}) { + co_return; +} + +Future readyFuture() { + return Void(); +} + +Future waitUncancellable(Future input, Uncancellable = {}) { + co_await input; +} + +Future waitOne(Future input) { + co_await input; +} + +Future waitEither(Future first, Future second) { + co_await race(first, second); +} + +Future g_input; + +Future waitSnapshot() { + Future input = g_input; + co_await input; +} + +Future sumStream(FutureStream input) { + int sum = 0; + try { + while (true) { + sum += co_await input; + } + } catch (Error& e) { + if (e.code() != error_code_end_of_stream) { + throw; + } + } + co_return sum; +} + +enum class ScalarPattern { + EmptyUncancellable, + ReadyFuture, + WaitUncancellableReady, + WaitReady, + WaitCancel, + WaitAfter, + RaceBothReady, + RaceFirstReady, + RaceSecondReady, + RaceCancel, + RaceAfter, + WaitRaceAfter, + FanoutRaceAfter, + Quorum +}; + +template +Future runScalar(benchmark::State* state) { + Promise pending; + Future never = pending.getFuture(); + Future ready = Void(); + std::vector> promises; + std::vector> inputs; + if constexpr (pattern == ScalarPattern::Quorum) { + promises.resize(3); + inputs.resize(3); + } + + // Destruction, including cancellation of unresolved actors, is part of each operation. + for (auto _ : *state) { + if constexpr (pattern == ScalarPattern::EmptyUncancellable) { + Future f = emptyUncancellable(); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::ReadyFuture) { + Future f = readyFuture(); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::WaitUncancellableReady) { + Future f = waitUncancellable(ready); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::WaitReady) { + Future f = waitOne(ready); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::WaitCancel) { + Future f = waitOne(never); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::RaceBothReady) { + Future f = waitEither(ready, ready); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::RaceFirstReady) { + Future f = waitEither(ready, never); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::RaceSecondReady) { + Future f = waitEither(never, ready); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::RaceCancel) { + Future f = waitEither(never, never); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::Quorum) { + promises.clear(); + promises.resize(3); + for (int i = 0; i < 3; ++i) { + inputs[i] = promises[i].getFuture(); + } + Future f = quorum(inputs, 2); + for (auto& promise : promises) { + promise.send(Void()); + } + benchmark::DoNotOptimize(f); + } else { + Promise signal; + if constexpr (pattern == ScalarPattern::WaitAfter) { + Future f = waitOne(signal.getFuture()); + signal.send(Void()); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::RaceAfter) { + Future f = waitEither(signal.getFuture(), never); + signal.send(Void()); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::WaitRaceAfter) { + Future f = waitOne(waitEither(signal.getFuture(), never)); + signal.send(Void()); + benchmark::DoNotOptimize(f); + } else if constexpr (pattern == ScalarPattern::FanoutRaceAfter) { + Future f = waitEither(signal.getFuture(), never); + Future first = waitOne(f); + Future second = waitOne(f); + signal.send(Void()); + benchmark::DoNotOptimize(first); + benchmark::DoNotOptimize(second); + } + } + benchmark::ClobberMemory(); + } + ASSERT_EQ(pending.getFutureReferenceCount(), 1); + state->SetItemsProcessed(state->iterations()); + co_return; +} + +template +void benchScalar(benchmark::State& state) { + onMainThread([&state] { return runScalar(&state); }).getBlocking(); +} + +enum class BatchPattern { + WaitFifo, + WaitLifo, + Race, + RaceSameInput, + NestedRace, + WaitRace, + FanoutRace, + FanoutWaitRace, + FanoutSnapshotRace +}; + +template +Future runBatch(benchmark::State* state) { + const int count = state->range(0); + constexpr bool fanout = pattern == BatchPattern::FanoutRace || pattern == BatchPattern::FanoutWaitRace || + pattern == BatchPattern::FanoutSnapshotRace; + Promise pending; + Future never = pending.getFuture(); + for (auto _ : *state) { + state->PauseTiming(); + { + std::vector> signals(count); + std::vector> first(count); + std::vector> second(fanout ? count : 0); + // Measure construction and completion with all frames live together; exclude input setup and final cleanup. + state->ResumeTiming(); + for (int i = 0; i < count; ++i) { + if constexpr (pattern == BatchPattern::WaitFifo || pattern == BatchPattern::WaitLifo) { + first[i] = waitOne(signals[i].getFuture()); + } else if constexpr (pattern == BatchPattern::Race) { + first[i] = waitEither(signals[i].getFuture(), never); + } else if constexpr (pattern == BatchPattern::RaceSameInput) { + first[i] = waitEither(signals[i].getFuture(), signals[i].getFuture()); + } else if constexpr (pattern == BatchPattern::NestedRace) { + first[i] = waitEither(waitEither(signals[i].getFuture(), never), never); + } else if constexpr (pattern == BatchPattern::WaitRace) { + first[i] = waitOne(waitEither(signals[i].getFuture(), never)); + } else if constexpr (pattern == BatchPattern::FanoutRace) { + Future f = waitEither(signals[i].getFuture(), never); + first[i] = waitOne(f); + second[i] = waitOne(f); + } else if constexpr (pattern == BatchPattern::FanoutWaitRace) { + Future f = waitEither(waitOne(signals[i].getFuture()), never); + first[i] = waitOne(f); + second[i] = waitOne(f); + } else if constexpr (pattern == BatchPattern::FanoutSnapshotRace) { + g_input = signals[i].getFuture(); + Future f = waitEither(waitSnapshot(), never); + g_input = f; + first[i] = waitSnapshot(); + second[i] = waitSnapshot(); + } + } + benchmark::DoNotOptimize(first.data()); + benchmark::DoNotOptimize(second.data()); + for (int i = 0; i < count; ++i) { + const int index = pattern == BatchPattern::WaitLifo ? count - 1 - i : i; + signals[index].send(Void()); + } + benchmark::ClobberMemory(); + state->PauseTiming(); + for (auto const& f : first) { + ASSERT(f.isReady() && !f.isError()); + } + for (auto const& f : second) { + ASSERT(f.isReady() && !f.isError()); + } + if constexpr (pattern == BatchPattern::FanoutSnapshotRace) { + g_input = Future(); + } + } + ASSERT_EQ(pending.getFutureReferenceCount(), 1); + state->ResumeTiming(); + } + state->SetItemsProcessed(state->iterations() * count); + co_return; +} + +template +void benchBatch(benchmark::State& state) { + onMainThread([&state] { return runBatch(&state); }).getBlocking(); +} + +Future runStreamSum(benchmark::State* state) { + const int count = state->range(0); + for (auto _ : *state) { + PromiseStream stream; + Future sum = sumStream(stream.getFuture()); + for (int i = 0; i < count; ++i) { + stream.send(1); + } + stream.sendError(end_of_stream()); + benchmark::DoNotOptimize(sum); + ASSERT_EQ(sum.get(), count); + } + state->SetItemsProcessed(state->iterations() * count); + co_return; +} + +void benchStreamSum(benchmark::State& state) { + onMainThread([&state] { return runStreamSum(&state); }).getBlocking(); +} + +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::EmptyUncancellable) + ->Name("actor_patterns/empty_uncancellable") + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::ReadyFuture)->Name("actor_patterns/ready_future")->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::WaitUncancellableReady) + ->Name("actor_patterns/wait_uncancellable_ready") + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::WaitReady)->Name("actor_patterns/wait_ready")->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::WaitCancel)->Name("actor_patterns/wait_construct_cancel")->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::WaitAfter)->Name("actor_patterns/wait_after")->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::RaceBothReady)->Name("actor_patterns/race_both_ready")->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::RaceFirstReady)->Name("actor_patterns/race_first_ready")->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::RaceSecondReady) + ->Name("actor_patterns/race_second_ready") + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::RaceCancel)->Name("actor_patterns/race_construct_cancel")->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::RaceAfter)->Name("actor_patterns/race_after")->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::WaitRaceAfter)->Name("actor_patterns/wait_race_after")->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::FanoutRaceAfter) + ->Name("actor_patterns/fanout_race_after") + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchScalar, ScalarPattern::Quorum)->Name("actor_patterns/quorum_2_of_3")->UseRealTime(); + +BENCHMARK_TEMPLATE(benchBatch, BatchPattern::WaitFifo) + ->Name("actor_patterns/batch_wait_fifo") + ->Arg(64) + ->Arg(4096) + ->Arg(65536) + ->Arg(1000000) + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchBatch, BatchPattern::WaitLifo) + ->Name("actor_patterns/batch_wait_lifo") + ->Arg(64) + ->Arg(4096) + ->Arg(65536) + ->Arg(1000000) + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchBatch, BatchPattern::Race) + ->Name("actor_patterns/batch_race") + ->Arg(64) + ->Arg(4096) + ->Arg(65536) + ->Arg(1000000) + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchBatch, BatchPattern::RaceSameInput) + ->Name("actor_patterns/batch_race_same_input") + ->Arg(64) + ->Arg(4096) + ->Arg(65536) + ->Arg(1000000) + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchBatch, BatchPattern::NestedRace) + ->Name("actor_patterns/batch_nested_race") + ->Arg(64) + ->Arg(4096) + ->Arg(65536) + ->Arg(1000000) + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchBatch, BatchPattern::WaitRace) + ->Name("actor_patterns/batch_wait_race") + ->Arg(64) + ->Arg(4096) + ->Arg(65536) + ->Arg(1000000) + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchBatch, BatchPattern::FanoutRace) + ->Name("actor_patterns/batch_fanout_race") + ->Arg(64) + ->Arg(4096) + ->Arg(65536) + ->Arg(1000000) + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchBatch, BatchPattern::FanoutWaitRace) + ->Name("actor_patterns/batch_fanout_wait_race") + ->Arg(64) + ->Arg(4096) + ->Arg(65536) + ->Arg(1000000) + ->UseRealTime(); +BENCHMARK_TEMPLATE(benchBatch, BatchPattern::FanoutSnapshotRace) + ->Name("actor_patterns/batch_fanout_snapshot_race") + ->Arg(64) + ->Arg(4096) + ->Arg(65536) + ->Arg(1000000) + ->UseRealTime(); +BENCHMARK(benchStreamSum) + ->Name("actor_patterns/stream_sum") + ->Arg(64) + ->Arg(4096) + ->Arg(65536) + ->Arg(1000000) + ->UseRealTime(); + +} // namespace diff --git a/flow/bench/README.md b/flow/bench/README.md index 348ec9f6cfd..67967ff3926 100644 --- a/flow/bench/README.md +++ b/flow/bench/README.md @@ -55,8 +55,26 @@ Existing Benchmarks - `bench_stream` measures the performance of writing to and reading from a `PromiseStream` - `bench_random` measures the performance of `DeterministicRandom`. - `bench_timer` measures the performance of FoundationDB timers. +- `actor_patterns` measures ready futures, coroutine waits and cancellation, races, nested and shared-input graphs, stream consumption, and quorum completion. - `Memcpy` compares `rte_memcpy_noinline` and `memcpy` across aligned/unaligned and cached/uncached copy cases. +Actor patterns +============== + +Run the actor-pattern suite with `bin/flow_bench --benchmark_filter='^actor_patterns/'`. +Use `--benchmark_repetitions=5 --benchmark_out=actor-patterns.json --benchmark_out_format=json` +to retain repeated measurements. Batch cases cover 64, 4096, 65536, and 1000000 inputs; +for a shorter run, select one size, for example `--benchmark_filter='^actor_patterns/batch_.*/4096/'`. + +These benchmarks report wall-clock time and items per second. One item is a complete +actor graph, including both outputs for fan-out cases, or one consumed value for +`stream_sum`. Scalar cases include construction, completion or cancellation, and +destruction. Batch cases measure graph construction and FIFO/LIFO completion with +all graphs outstanding together; input-promise/vector setup, result checks, and +final cleanup are excluded. Stream cases include consumer startup and end-of-stream +handling. `ready_future` returns an already-ready `Future` without a coroutine; +`empty_uncancellable` executes an uncancellable coroutine. + Future use cases ================ From f0e1a1fc92a38c76905783fb9b49caf776f9bb8c Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Wed, 16 Sep 2026 08:14:19 -0700 Subject: [PATCH 087/170] Wait for backup encryption keys before opening data files --- fdbclient/BackupContainerBlobStore.cpp | 38 ++++++++++++++++++--- fdbclient/BackupContainerLocalDirectory.cpp | 8 ++--- 2 files changed, 38 insertions(+), 8 deletions(-) diff --git a/fdbclient/BackupContainerBlobStore.cpp b/fdbclient/BackupContainerBlobStore.cpp index e76fb7881e8..9dc750dbf3d 100644 --- a/fdbclient/BackupContainerBlobStore.cpp +++ b/fdbclient/BackupContainerBlobStore.cpp @@ -24,6 +24,7 @@ #include "fdbrpc/AsyncFileEncrypted.h" #include "fdbrpc/AsyncFileReadAhead.h" #include "fdbrpc/HTTP.h" +#include "flow/Platform.h" #include "flow/UnitTest.h" class BackupContainerBlobStoreImpl { @@ -227,7 +228,10 @@ Future> BackupContainerBlobStore::readFile(const std::stri m_bstore->knobs.read_cache_blocks_per_file); } if (usesEncryption() && !StringRef(path).startsWith("properties/"_sr)) { - f = makeReference(f, AsyncFileEncrypted::Mode::READ_ONLY, encryptionBlockSize); + // Agents can open an existing container without calling create(), so key loading may still be in flight. + return map(encryptionSetupComplete(), [f, blockSize = encryptionBlockSize](Void) -> Reference { + return makeReference(f, AsyncFileEncrypted::Mode::READ_ONLY, blockSize); + }); } return f; } @@ -239,11 +243,16 @@ Future> BackupContainerBlobStore::listURLs(Reference> BackupContainerBlobStore::writeFile(const std::string& path) { - Reference f = makeReference(m_bstore, m_bucket, dataPath(path)); + Reference rawFile = makeReference(m_bstore, m_bucket, dataPath(path)); + Future> f = rawFile; if (usesEncryption() && !StringRef(path).startsWith("properties/"_sr)) { - f = makeReference(f, AsyncFileEncrypted::Mode::APPEND_ONLY, encryptionBlockSize); + f = map(encryptionSetupComplete(), [rawFile, blockSize = encryptionBlockSize](Void) -> Reference { + return makeReference(rawFile, AsyncFileEncrypted::Mode::APPEND_ONLY, blockSize); + }); } - return Future>(makeReference(path, f)); + return map(f, [path](Reference file) -> Reference { + return makeReference(path, file); + }); } Future BackupContainerBlobStore::writeEntireFile(const std::string& path, const std::string& fileContents) { @@ -281,6 +290,27 @@ std::string BackupContainerBlobStore::getPrefix() const { return m_prefix; } +TEST_CASE("/backup/containers/blobstore/encryptionSetup") { + std::string keyFile = joinPath(params.getDataDir(), "encryption-key"); + co_await BackupContainerFileSystem::createTestEncryptionKeyFile(keyFile); + std::string resource; + IBlobStoreEndpoint::ParametersT backupParams; + auto endpoint = IBlobStoreEndpoint::fromString( + "blobstore://localhost:9999/encryption-setup?bucket=test", {}, &resource, nullptr, &backupParams); + auto container = makeReference(endpoint, resource, backupParams, keyFile, 4096, true); + + // An agent reopening a container does not call create(). No blob requests are needed to open these handles. + auto read = container->readFile("range"); + auto write = container->writeFile("range"); + if (!container->encryptionSetupComplete().isReady()) { + ASSERT(!read.isReady()); + ASSERT(!write.isReady()); + } + co_await (success(read) && success(write)); + ASSERT(container->encryptionSetupComplete().isReady()); + ASSERT(!container->encryptionSetupComplete().isError()); +} + TEST_CASE("/backup/containers/blobstore/prefix") { // Normalization: leading/trailing slashes are stripped, empty selects the default layout. ASSERT(BackupContainerBlobStore::normalizePrefix("").empty()); diff --git a/fdbclient/BackupContainerLocalDirectory.cpp b/fdbclient/BackupContainerLocalDirectory.cpp index 4d3cc97f2ad..78f91eb9acf 100644 --- a/fdbclient/BackupContainerLocalDirectory.cpp +++ b/fdbclient/BackupContainerLocalDirectory.cpp @@ -260,9 +260,9 @@ Future> BackupContainerLocalDirectory::readFile(const std: // Skip encryption for properties/ folder if (usesEncryption() && !StringRef(path).startsWith("properties/"_sr)) { int encBlockSize = encryptionBlockSize; - f = map(f, [encBlockSize](Reference r) { + f = map(success(f) && encryptionSetupComplete(), [file = f, encBlockSize](Void) { return Reference( - makeReference(r, AsyncFileEncrypted::Mode::READ_ONLY, encBlockSize)); + makeReference(file.get(), AsyncFileEncrypted::Mode::READ_ONLY, encBlockSize)); }); } @@ -303,9 +303,9 @@ Future> BackupContainerLocalDirectory::writeFile(const st // Skip encryption for properties/ folder if (usesEncryption() && !StringRef(path).startsWith("properties/"_sr)) { int encBlockSize = encryptionBlockSize; - f = map(f, [encBlockSize](Reference r) { + f = map(success(f) && encryptionSetupComplete(), [file = f, encBlockSize](Void) { return Reference( - makeReference(r, AsyncFileEncrypted::Mode::APPEND_ONLY, encBlockSize)); + makeReference(file.get(), AsyncFileEncrypted::Mode::APPEND_ONLY, encBlockSize)); }); } return map(f, [=](Reference file) -> Reference { From b0e18d3ea39e2693af863548fe3ef4a91379628f Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Wed, 16 Sep 2026 08:35:11 -0700 Subject: [PATCH 088/170] Share status latency field emission --- fdbserver/clustercontroller/Status.cpp | 84 +++++++------------------- 1 file changed, 23 insertions(+), 61 deletions(-) diff --git a/fdbserver/clustercontroller/Status.cpp b/fdbserver/clustercontroller/Status.cpp index e4e52c0c6b5..cd91c939c5c 100644 --- a/fdbserver/clustercontroller/Status.cpp +++ b/fdbserver/clustercontroller/Status.cpp @@ -454,6 +454,16 @@ struct RolesInfo { return latencyStats; } + void appendLatencyStatistics(JsonBuilderObject& object, + EventMap const& metrics, + const char* eventName, + const char* jsonKey) { + TraceEventFields const& latencyMetrics = metrics.at(eventName); + if (latencyMetrics.size()) { + object[jsonKey] = addLatencyStatistics(latencyMetrics); + } + } + JsonBuilderObject addLatencyBandInfo(TraceEventFields const& metrics) { JsonBuilderObject latencyBands; std::map bands; @@ -547,10 +557,7 @@ struct RolesInfo { maxTLogVersion - version - SERVER_KNOBS->STORAGE_LOGGING_DELAY * SERVER_KNOBS->VERSIONS_PER_SECOND); } - TraceEventFields const& readLatencyMetrics = metrics.at("ReadLatencyMetrics"); - if (readLatencyMetrics.size()) { - obj["read_latency_statistics"] = addLatencyStatistics(readLatencyMetrics); - } + appendLatencyStatistics(obj, metrics, "ReadLatencyMetrics", "read_latency_statistics"); TraceEventFields const& readLatencyBands = metrics.at("ReadLatencyBands"); if (readLatencyBands.size()) { @@ -662,60 +669,22 @@ struct RolesInfo { obj["id"] = iface.id().shortString(); obj["role"] = role; try { - TraceEventFields const& commitLatencyMetrics = metrics.at("CommitLatencyMetrics"); - if (commitLatencyMetrics.size()) { - obj["commit_latency_statistics"] = addLatencyStatistics(commitLatencyMetrics); - } + appendLatencyStatistics(obj, metrics, "CommitLatencyMetrics", "commit_latency_statistics"); TraceEventFields const& commitLatencyBands = metrics.at("CommitLatencyBands"); if (commitLatencyBands.size()) { obj["commit_latency_bands"] = addLatencyBandInfo(commitLatencyBands); } - TraceEventFields const& commitBatchingWindowSize = metrics.at("CommitBatchingWindowSize"); - if (commitBatchingWindowSize.size()) { - obj["commit_batching_window_size"] = addLatencyStatistics(commitBatchingWindowSize); - } - - TraceEventFields const& commitBatchTransactions = metrics.at("CommitBatchTransactions"); - if (commitBatchTransactions.size()) { - obj["commit_batch_transactions"] = addLatencyStatistics(commitBatchTransactions); - } - - TraceEventFields const& commitBatchBytes = metrics.at("CommitBatchBytes"); - if (commitBatchBytes.size()) { - obj["commit_batch_bytes"] = addLatencyStatistics(commitBatchBytes); - } - - TraceEventFields const& commitBatchingWaiting = metrics.at("CommitBatchingWaiting"); - if (commitBatchingWaiting.size()) { - obj["commit_batching_waiting"] = addLatencyStatistics(commitBatchingWaiting); - } - - TraceEventFields const& commitPreresolutionLatency = metrics.at("CommitPreresolutionLatency"); - if (commitPreresolutionLatency.size()) { - obj["commit_preresolution_latency"] = addLatencyStatistics(commitPreresolutionLatency); - } - - TraceEventFields const& commitResolutionLatency = metrics.at("CommitResolutionLatency"); - if (commitResolutionLatency.size()) { - obj["commit_resolution_latency"] = addLatencyStatistics(commitResolutionLatency); - } - - TraceEventFields const& commitPostresolutionLatency = metrics.at("CommitPostresolutionLatency"); - if (commitPostresolutionLatency.size()) { - obj["commit_postresolution_latency"] = addLatencyStatistics(commitPostresolutionLatency); - } - - TraceEventFields const& commitTLogLoggingLatency = metrics.at("CommitTLogLoggingLatency"); - if (commitTLogLoggingLatency.size()) { - obj["commit_tlog_logging_latency"] = addLatencyStatistics(commitTLogLoggingLatency); - } - - TraceEventFields const& commitReplyLatency = metrics.at("CommitReplyLatency"); - if (commitReplyLatency.size()) { - obj["commit_reply_latency"] = addLatencyStatistics(commitReplyLatency); - } + appendLatencyStatistics(obj, metrics, "CommitBatchingWindowSize", "commit_batching_window_size"); + appendLatencyStatistics(obj, metrics, "CommitBatchTransactions", "commit_batch_transactions"); + appendLatencyStatistics(obj, metrics, "CommitBatchBytes", "commit_batch_bytes"); + appendLatencyStatistics(obj, metrics, "CommitBatchingWaiting", "commit_batching_waiting"); + appendLatencyStatistics(obj, metrics, "CommitPreresolutionLatency", "commit_preresolution_latency"); + appendLatencyStatistics(obj, metrics, "CommitResolutionLatency", "commit_resolution_latency"); + appendLatencyStatistics(obj, metrics, "CommitPostresolutionLatency", "commit_postresolution_latency"); + appendLatencyStatistics(obj, metrics, "CommitTLogLoggingLatency", "commit_tlog_logging_latency"); + appendLatencyStatistics(obj, metrics, "CommitReplyLatency", "commit_reply_latency"); } catch (Error& e) { if (e.code() != error_code_attribute_not_found) { throw e; @@ -735,15 +704,8 @@ struct RolesInfo { // GRV Latency metrics are grouped according to priority (currently batch or default). // Other priorities can be added in the future. - TraceEventFields const& grvLatencyMetrics = metrics.at("GRVLatencyMetrics"); - if (grvLatencyMetrics.size()) { - priorityStats["default"] = addLatencyStatistics(grvLatencyMetrics); - } - - TraceEventFields const& grvBatchMetrics = metrics.at("GRVBatchLatencyMetrics"); - if (grvBatchMetrics.size()) { - priorityStats["batch"] = addLatencyStatistics(grvBatchMetrics); - } + appendLatencyStatistics(priorityStats, metrics, "GRVLatencyMetrics", "default"); + appendLatencyStatistics(priorityStats, metrics, "GRVBatchLatencyMetrics", "batch"); // Add GRV Latency metrics (for all priorities) to parent node. if (!priorityStats.empty()) { From 62fa57816578d6eab3a0cd255877bbce0c8b1d22 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Wed, 16 Sep 2026 08:35:47 -0700 Subject: [PATCH 089/170] Remove unused and duplicate CMake symlink helpers --- cmake/FDBInstall.cmake | 79 ----------------------------------- cmake/InstallLayout.cmake | 88 --------------------------------------- 2 files changed, 167 deletions(-) diff --git a/cmake/FDBInstall.cmake b/cmake/FDBInstall.cmake index 5395a67a8a7..322b71ede97 100644 --- a/cmake/FDBInstall.cmake +++ b/cmake/FDBInstall.cmake @@ -6,85 +6,6 @@ function(fdb_install_dirs) set(FDB_INSTALL_DIRS ${ARGV} PARENT_SCOPE) endfunction() -function(install_symlink_impl) - if (NOT WIN32) - return() - endif() - set(options "") - set(one_value_options TO DESTINATION) - set(multi_value_options COMPONENTS) - cmake_parse_arguments(SYM "${options}" "${one_value_options}" "${multi_value_options}" "${ARGN}") - - file(MAKE_DIRECTORY ${CMAKE_CURRENT_BINARY_DIR}/symlinks) - get_filename_component(fname ${SYM_DESTINATION} NAME) - get_filename_component(dest_dir ${SYM_DESTINATION} DIRECTORY) - set(sl ${CMAKE_CURRENT_BINARY_DIR}/symlinks/${fname}) - execute_process(COMMAND ${CMAKE_COMMAND} -E create_symlink ${SYM_TO} ${sl}) - foreach(component IN LISTS SYM_COMPONENTS) - install(FILES ${sl} DESTINATION ${dest_dir} COMPONENT ${component}) - endforeach() -endfunction() - -function(install_symlink) - if(NOT WIN32 AND NOT OPEN_FOR_IDE) - return() - endif() - set(options "") - set(one_value_options COMPONENT LINK_DIR FILE_DIR LINK_NAME FILE_NAME) - set(multi_value_options "") - cmake_parse_arguments(IN "${options}" "${one_value_options}" "${multi_value_options}" "${ARGN}") - - set(rel_path "") - string(REGEX MATCHALL "\\/" slashes "${IN_LINK_NAME}") - foreach(ignored IN LISTS slashes) - set(rel_path "../${rel_path}") - endforeach() - if("${IN_FILE_DIR}" MATCHES "bin") - if("${IN_LINK_DIR}" MATCHES "lib") - install_symlink_impl( - TO "../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "lib/${IN_LINK_NAME}" - COMPONENTS "${IN_COMPONENT}-tgz") - install_symlink_impl( - TO "../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "usr/lib64/${IN_LINK_NAME}" - COMPONENTS - "${IN_COMPONENT}-el9" - "${IN_COMPONENT}-deb") - install_symlink_impl( - TO "../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "usr/lib64/${IN_LINK_NAME}" - COMPONENTS "${IN_COMPONENT}-deb") - elseif("${IN_LINK_DIR}" MATCHES "bin") - install_symlink_impl( - TO "../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "bin/${IN_LINK_NAME}" - COMPONENTS "${IN_COMPONENT}-tgz") - install_symlink_impl( - TO "../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "usr/bin/${IN_LINK_NAME}" - COMPONENTS - "${IN_COMPONENT}-el9" - "${IN_COMPONENT}-deb") - elseif("${IN_LINK_DIR}" MATCHES "fdbmonitor") - install_symlink_impl( - TO "../../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "lib/foundationdb/${IN_LINK_NAME}" - COMPONENTS "${IN_COMPONENT}-tgz") - install_symlink_impl( - TO "../../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "usr/lib/foundationdb/${IN_LINK_NAME}" - COMPONENTS - "${IN_COMPONENT}-el9" - "${IN_COMPONENT}-deb") - else() - message(FATAL_ERROR "Unknown LINK_DIR ${IN_LINK_DIR}") - endif() - else() - message(FATAL_ERROR "Unknown FILE_DIR ${IN_FILE_DIR}") - endif() -endfunction() - function(symlink_files) if (NOT WIN32) set(options "") diff --git a/cmake/InstallLayout.cmake b/cmake/InstallLayout.cmake index 616d2253e9e..f5e5a9e2c68 100644 --- a/cmake/InstallLayout.cmake +++ b/cmake/InstallLayout.cmake @@ -1,93 +1,5 @@ include(FDBInstall) -function(install_symlink_impl) - if (NOT WIN32) - set(options "") - set(one_value_options TO DESTINATION) - set(multi_value_options COMPONENTS) - cmake_parse_arguments(SYM "${options}" "${one_value_options}" "${multi_value_options}" "${ARGN}") - - file(MAKE_DIRECTORY ${CMAKE_CURRENT_BINARY_DIR}/symlinks) - get_filename_component(fname ${SYM_DESTINATION} NAME) - get_filename_component(dest_dir ${SYM_DESTINATION} DIRECTORY) - set(sl ${CMAKE_CURRENT_BINARY_DIR}/symlinks/${fname}) - execute_process(COMMAND ${CMAKE_COMMAND} -E create_symlink ${SYM_TO} ${sl}) - foreach(component IN LISTS SYM_COMPONENTS) - install(FILES ${sl} DESTINATION ${dest_dir} COMPONENT ${component}) - endforeach() - endif() -endfunction() - -function(install_symlink) - if(NOT WIN32 AND NOT OPEN_FOR_IDE) - set(options "") - set(one_value_options COMPONENT LINK_DIR FILE_DIR LINK_NAME FILE_NAME) - set(multi_value_options "") - cmake_parse_arguments(IN "${options}" "${one_value_options}" "${multi_value_options}" "${ARGN}") - - set(rel_path "") - string(REGEX MATCHALL "\\/" slashes "${IN_LINK_NAME}") - foreach(ignored IN LISTS slashes) - set(rel_path "../${rel_path}") - endforeach() - if("${IN_FILE_DIR}" MATCHES "bin") - if("${IN_LINK_DIR}" MATCHES "lib") - install_symlink_impl( - TO "../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "lib/${IN_LINK_NAME}" - COMPONENTS "${IN_COMPONENT}-tgz") - install_symlink_impl( - TO "../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "usr/lib64/${IN_LINK_NAME}" - COMPONENTS "${IN_COMPONENT}-el9" - "${IN_COMPONENT}-deb") - install_symlink_impl( - TO "../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "usr/lib64/${IN_LINK_NAME}" - COMPONENTS "${IN_COMPONENT}-deb") - elseif("${IN_LINK_DIR}" MATCHES "bin") - install_symlink_impl( - TO "../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "bin/${IN_LINK_NAME}" - COMPONENTS "${IN_COMPONENT}-tgz") - install_symlink_impl( - TO "../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "usr/bin/${IN_LINK_NAME}" - COMPONENTS "${IN_COMPONENT}-el9" - "${IN_COMPONENT}-deb") - elseif("${IN_LINK_DIR}" MATCHES "fdbmonitor") - install_symlink_impl( - TO "../../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "lib/foundationdb/${IN_LINK_NAME}" - COMPONENTS "${IN_COMPONENT}-tgz") - install_symlink_impl( - TO "../../${rel_path}bin/${IN_FILE_NAME}" - DESTINATION "usr/lib/foundationdb/${IN_LINK_NAME}" - COMPONENTS "${IN_COMPONENT}-el9" - "${IN_COMPONENT}-deb") - else() - message(FATAL_ERROR "Unknown LINK_DIR ${IN_LINK_DIR}") - endif() - else() - message(FATAL_ERROR "Unknown FILE_DIR ${IN_FILE_DIR}") - endif() - endif() -endfunction() - -function(symlink_files) - if (NOT WIN32) - set(options "") - set(one_value_options LOCATION SOURCE) - set(multi_value_options TARGETS) - cmake_parse_arguments(SYM "${options}" "${one_value_options}" "${multi_value_options}" "${ARGN}") - - file(MAKE_DIRECTORY ${CMAKE_BINARY_DIR}/${SYM_LOCATION}) - foreach(component IN LISTS SYM_TARGETS) - execute_process(COMMAND ${CMAKE_COMMAND} -E create_symlink ${SYM_SOURCE} ${CMAKE_BINARY_DIR}/${SYM_LOCATION}/${component} WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/${SYM_LOCATION}) - endforeach() - endif() -endfunction() - fdb_install_packages(TGZ DEB EL9 VERSIONED) fdb_install_dirs(BIN SBIN LIB INCLUDE ETC LOG DATA) message(STATUS "FDB_INSTALL_DIRS -> ${FDB_INSTALL_DIRS}") From 188064c0589e6ca787672bf06fd711fa97c0449d Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Wed, 16 Sep 2026 08:35:47 -0700 Subject: [PATCH 090/170] Remove unused bulk progress helpers --- fdbcli/BulkDumpCommand.cpp | 31 ------------------ fdbcli/BulkLoadCommand.cpp | 64 -------------------------------------- 2 files changed, 95 deletions(-) diff --git a/fdbcli/BulkDumpCommand.cpp b/fdbcli/BulkDumpCommand.cpp index 89c362b18fa..724683b5218 100644 --- a/fdbcli/BulkDumpCommand.cpp +++ b/fdbcli/BulkDumpCommand.cpp @@ -41,37 +41,6 @@ static const std::string BULK_DUMP_HELP_MESSAGE = std::string(BULK_DUMP_MODE_USAGE) + std::string(BULK_DUMP_DUMP_USAGE) + std::string(BULK_DUMP_STATUS_USAGE) + std::string(BULK_DUMP_CANCEL_USAGE); -Future getOngoingBulkDumpJob(Database cx) { - Transaction tr(cx); - while (true) { - Error err; - try { - Optional job = co_await getSubmittedBulkDumpJob(&tr); - if (job.present()) { - fmt::println("Running bulk dumping job: {}", job.get().getJobId().toString()); - co_return true; - } else { - fmt::println("No bulk dumping job is running"); - co_return false; - } - } catch (Error& e) { - err = e; - } - co_await tr.onError(err); - } -} - -Future getBulkDumpCompleteRanges(Database cx, KeyRange rangeToRead) { - try { - size_t finishCount = co_await getBulkDumpCompleteTaskCount(cx, rangeToRead); - fmt::println("Finished {} tasks", finishCount); - } catch (Error& e) { - if (e.code() == error_code_timed_out) { - fmt::println("timed out"); - } - } -} - Future bulkDumpCommandActor(Database cx, std::vector tokens) { BulkDumpState bulkDumpJob; if (tokencmp(tokens[1], "mode")) { diff --git a/fdbcli/BulkLoadCommand.cpp b/fdbcli/BulkLoadCommand.cpp index 55660c7643c..70a54efd2ea 100644 --- a/fdbcli/BulkLoadCommand.cpp +++ b/fdbcli/BulkLoadCommand.cpp @@ -86,70 +86,6 @@ Future printPastBulkLoadJob(Database cx) { } } -void printBulkLoadJobTotalTaskCount(Optional count) { - if (count.present()) { - fmt::println("Total {} tasks", count.get()); - } else { - fmt::println("Total task count is unknown"); - } - return; -} - -Future printBulkLoadJobProgress(Database cx, BulkLoadJobState job) { - Transaction tr(cx); - Key readBegin = job.getJobRange().begin; - Key readEnd = job.getJobRange().end; - UID jobId = job.getJobId(); - RangeResult rangeResult; - size_t completeTaskCount = 0; - size_t submitTaskCount = 0; - size_t errorTaskCount = 0; - Optional totalTaskCount = job.getTaskCount(); - while (readBegin < readEnd) { - Error err; - bool hasErr = false; - try { - rangeResult.clear(); - tr.setOption(FDBTransactionOptions::READ_SYSTEM_KEYS); - tr.setOption(FDBTransactionOptions::LOCK_AWARE); - rangeResult = co_await krmGetRanges(&tr, bulkLoadTaskPrefix, KeyRangeRef(readBegin, readEnd)); - for (int i = 0; i < rangeResult.size() - 1; ++i) { - if (rangeResult[i].value.empty()) { - continue; - } - BulkLoadTaskState bulkLoadTask = decodeBulkLoadTaskState(rangeResult[i].value); - if (bulkLoadTask.getJobId() != jobId) { - fmt::println("Submitted {} tasks", submitTaskCount); - fmt::println("Finished {} tasks", completeTaskCount); - fmt::println("Error {} tasks", errorTaskCount); - printBulkLoadJobTotalTaskCount(totalTaskCount); - if (bulkLoadTask.phase == BulkLoadPhase::Submitted && bulkLoadTask.getJobId().isValid()) { - fmt::println("Job {} has been cancelled or has completed", jobId.toString()); - } - co_return; - } - if (bulkLoadTask.phase == BulkLoadPhase::Complete) { - completeTaskCount = completeTaskCount + bulkLoadTask.getManifests().size(); - } else if (bulkLoadTask.phase == BulkLoadPhase::Error) { - errorTaskCount = errorTaskCount + bulkLoadTask.getManifests().size(); - } - submitTaskCount = submitTaskCount + bulkLoadTask.getManifests().size(); - } - readBegin = rangeResult.back().key; - } catch (Error& e) { - err = e; - hasErr = true; - } - if (hasErr) { - co_await tr.onError(err); - } - } - fmt::println("Submitted {} tasks", submitTaskCount); - fmt::println("Finished {} tasks", completeTaskCount); - fmt::println("Error {} tasks", errorTaskCount); - printBulkLoadJobTotalTaskCount(totalTaskCount); -} - Future bulkLoadCommandActor(Database cx, std::vector tokens) { if (tokencmp(tokens[1], "mode")) { if (tokens.size() == 2) { From 38d9bbeeea3133c13eb79bc8250468f98519975a Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Wed, 16 Sep 2026 08:35:47 -0700 Subject: [PATCH 091/170] Share TLS certificate and key path resolution --- flow/TLSConfig.cpp | 39 +++++++++++++++------------------------ 1 file changed, 15 insertions(+), 24 deletions(-) diff --git a/flow/TLSConfig.cpp b/flow/TLSConfig.cpp index 0ab43b54341..518a6b0507d 100644 --- a/flow/TLSConfig.cpp +++ b/flow/TLSConfig.cpp @@ -180,40 +180,31 @@ void ConfigureSSLStream(Reference policy, } } -std::string TLSConfig::getCertificatePathSync() const { - if (!tlsCertPath.empty()) { - return tlsCertPath; +static std::string resolveTLSMaterialPath(const std::string& configuredPath, + const char* envName, + const char* defaultFileName) { + if (!configuredPath.empty()) { + return configuredPath; } - std::string envCertPath; - if (platform::getEnvironmentVar("FDB_TLS_CERTIFICATE_FILE", envCertPath)) { - return envCertPath; + std::string envPath; + if (platform::getEnvironmentVar(envName, envPath)) { + return envPath; } - const char* defaultCertFileName = "cert.pem"; - if (fileExists(joinPath(platform::getDefaultConfigPath(), defaultCertFileName))) { - return joinPath(platform::getDefaultConfigPath(), defaultCertFileName); + if (fileExists(joinPath(platform::getDefaultConfigPath(), defaultFileName))) { + return joinPath(platform::getDefaultConfigPath(), defaultFileName); } return std::string(); } -std::string TLSConfig::getKeyPathSync() const { - if (!tlsKeyPath.empty()) { - return tlsKeyPath; - } - - std::string envKeyPath; - if (platform::getEnvironmentVar("FDB_TLS_KEY_FILE", envKeyPath)) { - return envKeyPath; - } - - const char* defaultKeyFileName = "key.pem"; - if (fileExists(joinPath(platform::getDefaultConfigPath(), defaultKeyFileName))) { - return joinPath(platform::getDefaultConfigPath(), defaultKeyFileName); - } +std::string TLSConfig::getCertificatePathSync() const { + return resolveTLSMaterialPath(tlsCertPath, "FDB_TLS_CERTIFICATE_FILE", "cert.pem"); +} - return std::string(); +std::string TLSConfig::getKeyPathSync() const { + return resolveTLSMaterialPath(tlsKeyPath, "FDB_TLS_KEY_FILE", "key.pem"); } std::string TLSConfig::getCAPathSync() const { From 825a3eade91e7537d8915d3d2da2b7bc0940093f Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Wed, 16 Sep 2026 08:36:30 -0700 Subject: [PATCH 092/170] Share single-key coalescing across key ownership variants --- fdbclient/include/fdbclient/KeyRangeMap.h | 90 ++++++++--------------- 1 file changed, 29 insertions(+), 61 deletions(-) diff --git a/fdbclient/include/fdbclient/KeyRangeMap.h b/fdbclient/include/fdbclient/KeyRangeMap.h index eda30829a2a..553023058c0 100644 --- a/fdbclient/include/fdbclient/KeyRangeMap.h +++ b/fdbclient/include/fdbclient/KeyRangeMap.h @@ -253,18 +253,16 @@ void insertCoalescedRange(Map, Metric>& } } -} // namespace KeyRangeMapImpl - -template -void CoalescedKeyRangeMap::insert(const KeyRangeRef& keys, const Val& value) { - KeyRangeMapImpl::insertCoalescedRange(this->map, this->mf, mapEnd, keys, value); -} - -template -void CoalescedKeyRangeMap::insert(const KeyRef& key, const Val& value) { +template +void insertCoalescedKey(Map, Metric>& map, + const MetricFunc& mf, + const Key& mapEnd, + const KeyRef& key, + const Val& value, + MakeKeyAfter makeKeyAfter) { ASSERT(key < mapEnd); - auto begin = RangeMap::map.lower_bound(key); + auto begin = map.lower_bound(key); auto end = begin; if (end->key == key) ++end; @@ -295,70 +293,40 @@ void CoalescedKeyRangeMap::insert(const KeyRef& key, co insertBegin = true; } - RangeMap::map.erase(begin, end); + map.erase(begin, end); if (insertEnd) { - MapPair p(keyAfter(key), endVal); - RangeMap::map.insert( - p, true, RangeMap::mf(p)); + MapPair p(makeKeyAfter(key), endVal); + map.insert(p, true, mf(p)); } if (insertBegin) { - MapPair p(key, value); - RangeMap::map.insert( - p, true, RangeMap::mf(p)); + MapPair p(key, value); + map.insert(p, true, mf(p)); } } +} // namespace KeyRangeMapImpl + template -void CoalescedKeyRefRangeMap::insert(const KeyRangeRef& keys, const Val& value) { +void CoalescedKeyRangeMap::insert(const KeyRangeRef& keys, const Val& value) { KeyRangeMapImpl::insertCoalescedRange(this->map, this->mf, mapEnd, keys, value); } template -void CoalescedKeyRefRangeMap::insert(const KeyRef& key, const Val& value, Arena& arena) { - ASSERT(key < mapEnd); - - auto begin = RangeMap::map.lower_bound(key); - auto end = begin; - if (end->key == key) - ++end; - - bool insertEnd = false; - bool insertBegin = false; - Val endVal; - - if (!equalsKeyAfter(key, end->key)) { - auto before_end = end; - before_end.decrementNonEnd(); - if (value != before_end->value) { - insertEnd = true; - endVal = before_end->value; - } - } - - if (!insertEnd && end->value == value && end->key != mapEnd) { - ++end; - } +void CoalescedKeyRangeMap::insert(const KeyRef& key, const Val& value) { + KeyRangeMapImpl::insertCoalescedKey( + this->map, this->mf, mapEnd, key, value, [](const KeyRef& key) -> Key { return keyAfter(key); }); +} - if (key == allKeys.begin) { - insertBegin = true; - } else { - auto before_begin = begin; - before_begin.decrementNonEnd(); - if (before_begin->value != value) - insertBegin = true; - } +template +void CoalescedKeyRefRangeMap::insert(const KeyRangeRef& keys, const Val& value) { + KeyRangeMapImpl::insertCoalescedRange(this->map, this->mf, mapEnd, keys, value); +} - RangeMap::map.erase(begin, end); - if (insertEnd) { - MapPair p(keyAfter(key, arena), endVal); - RangeMap::map.insert( - p, true, RangeMap::mf(p)); - } - if (insertBegin) { - MapPair p(key, value); - RangeMap::map.insert( - p, true, RangeMap::mf(p)); - } +template +void CoalescedKeyRefRangeMap::insert(const KeyRef& key, const Val& value, Arena& arena) { + KeyRangeMapImpl::insertCoalescedKey(this->map, this->mf, mapEnd, key, value, [&arena](const KeyRef& key) -> KeyRef { + return keyAfter(key, arena); + }); } #endif From 682304bc11ed8994c74ee7f25bebd2c849a6474f Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Wed, 16 Sep 2026 08:36:35 -0700 Subject: [PATCH 093/170] Share database lock operations across transaction handles --- fdbclient/ManagementAPI.cpp | 62 +++++++++++++------------------------ 1 file changed, 21 insertions(+), 41 deletions(-) diff --git a/fdbclient/ManagementAPI.cpp b/fdbclient/ManagementAPI.cpp index 6247bd1c697..c83b44233ed 100644 --- a/fdbclient/ManagementAPI.cpp +++ b/fdbclient/ManagementAPI.cpp @@ -2320,7 +2320,8 @@ Future timeKeeperSetDisable(Database cx) { } } -Future lockDatabase(Transaction* tr, UID id) { +template +static Future lockDatabaseImpl(TransactionHandle tr, UID id) { tr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); tr->setOption(FDBTransactionOptions::LOCK_AWARE); Optional val = co_await tr->get(databaseLockedKey); @@ -2340,24 +2341,12 @@ Future lockDatabase(Transaction* tr, UID id) { tr->addWriteConflictRange(normalKeys); } -Future lockDatabase(Reference tr, UID id) { - tr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); - tr->setOption(FDBTransactionOptions::LOCK_AWARE); - Optional val = co_await tr->get(databaseLockedKey); - - if (val.present()) { - if (BinaryReader::fromStringRef(val.get().substr(10), Unversioned()) == id) { - co_return; - } else { - //TraceEvent("DBA_LockLocked").detail("Expecting", id).detail("Lock", BinaryReader::fromStringRef(val.get().substr(10), Unversioned())); - throw database_locked(); - } - } +Future lockDatabase(Transaction* tr, UID id) { + return lockDatabaseImpl(tr, id); +} - tr->atomicOp(databaseLockedKey, - BinaryWriter::toValue(id, Unversioned()).withPrefix("0123456789"_sr).withSuffix("\x00\x00\x00\x00"_sr), - MutationRef::SetVersionstampedValue); - tr->addWriteConflictRange(normalKeys); +Future lockDatabase(Reference tr, UID id) { + return lockDatabaseImpl(std::move(tr), id); } Future lockDatabase(Database cx, UID id) { @@ -2380,7 +2369,8 @@ Future lockDatabase(Database cx, UID id) { } } -Future unlockDatabase(Transaction* tr, UID id) { +template +static Future unlockDatabaseImpl(TransactionHandle tr, UID id) { tr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); tr->setOption(FDBTransactionOptions::LOCK_AWARE); Optional val = co_await tr->get(databaseLockedKey); @@ -2396,20 +2386,12 @@ Future unlockDatabase(Transaction* tr, UID id) { tr->clear(singleKeyRange(databaseLockedKey)); } -Future unlockDatabase(Reference tr, UID id) { - tr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); - tr->setOption(FDBTransactionOptions::LOCK_AWARE); - Optional val = co_await tr->get(databaseLockedKey); - - if (!val.present()) - co_return; - - if (val.present() && BinaryReader::fromStringRef(val.get().substr(10), Unversioned()) != id) { - //TraceEvent("DBA_UnlockLocked").detail("Expecting", id).detail("Lock", BinaryReader::fromStringRef(val.get().substr(10), Unversioned())); - throw database_locked(); - } +Future unlockDatabase(Transaction* tr, UID id) { + return unlockDatabaseImpl(tr, id); +} - tr->clear(singleKeyRange(databaseLockedKey)); +Future unlockDatabase(Reference tr, UID id) { + return unlockDatabaseImpl(std::move(tr), id); } Future unlockDatabase(Database cx, UID id) { @@ -2429,7 +2411,8 @@ Future unlockDatabase(Database cx, UID id) { } } -Future checkDatabaseLock(Transaction* tr, UID id) { +template +static Future checkDatabaseLockImpl(TransactionHandle tr, UID id) { tr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); tr->setOption(FDBTransactionOptions::LOCK_AWARE); Optional val = co_await tr->get(databaseLockedKey); @@ -2440,15 +2423,12 @@ Future checkDatabaseLock(Transaction* tr, UID id) { } } -Future checkDatabaseLock(Reference tr, UID id) { - tr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); - tr->setOption(FDBTransactionOptions::LOCK_AWARE); - Optional val = co_await tr->get(databaseLockedKey); +Future checkDatabaseLock(Transaction* tr, UID id) { + return checkDatabaseLockImpl(tr, id); +} - if (val.present() && BinaryReader::fromStringRef(val.get().substr(10), Unversioned()) != id) { - //TraceEvent("DBA_CheckLocked").detail("Expecting", id).detail("Lock", BinaryReader::fromStringRef(val.get().substr(10), Unversioned())).backtrace(); - throw database_locked(); - } +Future checkDatabaseLock(Reference tr, UID id) { + return checkDatabaseLockImpl(std::move(tr), id); } Future advanceVersion(Database cx, Version v) { From 9f437774dd04f02c759eaf07669b9277350f7d95 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Wed, 16 Sep 2026 08:37:16 -0700 Subject: [PATCH 094/170] Share backup worker pause monitoring --- fdbserver/backupworker/BackupWorker.cpp | 18 ++++++---- fdbserver/backupworker/BackupWorkerPause.h | 31 +++++++++++++++++ .../RangePartitionedBackupWorker.cpp | 34 ++----------------- 3 files changed, 46 insertions(+), 37 deletions(-) create mode 100644 fdbserver/backupworker/BackupWorkerPause.h diff --git a/fdbserver/backupworker/BackupWorker.cpp b/fdbserver/backupworker/BackupWorker.cpp index 1ba6b8bfc14..d468b43b8c1 100644 --- a/fdbserver/backupworker/BackupWorker.cpp +++ b/fdbserver/backupworker/BackupWorker.cpp @@ -18,6 +18,7 @@ * limitations under the License. */ +#include "BackupWorkerPause.h" #include "fdbclient/BackupAgent.h" #include "fdbclient/BackupFileFormat.h" #include "fdbclient/BackupContainer.h" @@ -1013,8 +1014,12 @@ Future checkRemoved(Reference const> db, LogEpoch r } } -static Future monitorWorkerPause(BackupData* self) { - auto tr = makeReference(self->cx); +Future monitorBackupPause(Database cx, + UID workerId, + AsyncVar* pauseState, + const char* pausedEvent, + const char* resumedEvent) { + auto tr = makeReference(cx); Future watch; while (true) { @@ -1026,9 +1031,9 @@ static Future monitorWorkerPause(BackupData* self) { Optional value = co_await tr->get(backupPausedKey); bool paused = value.present() && value.get() == "1"_sr; - if (self->paused.get() != paused) { - TraceEvent(paused ? "BackupWorkerPaused" : "BackupWorkerResumed", self->myId).log(); - self->paused.set(paused); + if (pauseState->get() != paused) { + TraceEvent(paused ? pausedEvent : resumedEvent, workerId).log(); + pauseState->set(paused); } watch = tr->watch(backupPausedKey); @@ -1067,7 +1072,8 @@ Future backupWorker(BackupInterface interf, if (req.recruitedEpoch == req.backupEpoch && req.tag.id == 0) { addActor.send(monitorBackupProgress(&self)); } - addActor.send(monitorWorkerPause(&self)); + addActor.send( + monitorBackupPause(self.cx, self.myId, &self.paused, "BackupWorkerPaused", "BackupWorkerResumed")); // If the worker is on an old epoch and all backups starts a version >= the endVersion bool exitEarly = co_await shouldBackupWorkerExitEarly(&self); diff --git a/fdbserver/backupworker/BackupWorkerPause.h b/fdbserver/backupworker/BackupWorkerPause.h new file mode 100644 index 00000000000..f8786ab8907 --- /dev/null +++ b/fdbserver/backupworker/BackupWorkerPause.h @@ -0,0 +1,31 @@ +/* + * BackupWorkerPause.h + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#pragma once + +#include "fdbclient/NativeAPI.h" +#include "flow/genericactors.h" + +// The pause state and event names must outlive the returned future. +Future monitorBackupPause(Database cx, + UID workerId, + AsyncVar* pauseState, + const char* pausedEvent, + const char* resumedEvent); diff --git a/fdbserver/backupworker/RangePartitionedBackupWorker.cpp b/fdbserver/backupworker/RangePartitionedBackupWorker.cpp index 4e48f5699cb..bd655c2e12a 100644 --- a/fdbserver/backupworker/RangePartitionedBackupWorker.cpp +++ b/fdbserver/backupworker/RangePartitionedBackupWorker.cpp @@ -18,6 +18,7 @@ * limitations under the License. */ +#include "BackupWorkerPause.h" #include "fdbclient/BackupAgent.h" #include "fdbclient/BackupFileFormat.h" #include "fdbclient/BackupContainer.h" @@ -1051,36 +1052,6 @@ Future setBackupKeys(RangePartitionedBackupData* self, std::map monitorWorkerPause(RangePartitionedBackupData* self) { - auto tr = makeReference(self->cx); - Future watch; - - while (true) { - Error err; - try { - tr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); - tr->setOption(FDBTransactionOptions::LOCK_AWARE); - tr->setOption(FDBTransactionOptions::PRIORITY_SYSTEM_IMMEDIATE); - - Optional value = co_await tr->get(backupPausedKey); - bool paused = value.present() && value.get() == "1"_sr; - if (self->paused.get() != paused) { - TraceEvent(paused ? "RangePartitionedBWPaused" : "RangePartitionedBWResumed", self->myId).log(); - self->paused.set(paused); - } - - watch = tr->watch(backupPausedKey); - co_await tr->commit(); - co_await watch; - tr->reset(); - continue; - } catch (Error& e) { - err = e; - } - co_await tr->onError(err); - } -} - Future monitorRangePartitionedBackupProgress(RangePartitionedBackupData* self) { Future interval; @@ -1313,7 +1284,8 @@ Future rangePartitionedBackupWorker(BackupInterface interf, addActor.send(monitorRangePartitionedBackupProgress(&self)); } - addActor.send(monitorWorkerPause(&self)); + addActor.send(monitorBackupPause( + self.cx, self.myId, &self.paused, "RangePartitionedBWPaused", "RangePartitionedBWResumed")); // Must be sent before processPartitionMap so logSystem is populated before the partition-map peek. addActor.send(monitorLogSystemFromDbInfo(db, &self)); From 425a6688b0f3d451dd1ddf5e98de0445ecd5b8e0 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Wed, 16 Sep 2026 09:04:48 -0700 Subject: [PATCH 095/170] Clarify backup pause monitor event arguments --- fdbserver/backupworker/BackupWorker.cpp | 7 +++++-- fdbserver/backupworker/RangePartitionedBackupWorker.cpp | 7 +++++-- 2 files changed, 10 insertions(+), 4 deletions(-) diff --git a/fdbserver/backupworker/BackupWorker.cpp b/fdbserver/backupworker/BackupWorker.cpp index d468b43b8c1..54fe2f1be61 100644 --- a/fdbserver/backupworker/BackupWorker.cpp +++ b/fdbserver/backupworker/BackupWorker.cpp @@ -1072,8 +1072,11 @@ Future backupWorker(BackupInterface interf, if (req.recruitedEpoch == req.backupEpoch && req.tag.id == 0) { addActor.send(monitorBackupProgress(&self)); } - addActor.send( - monitorBackupPause(self.cx, self.myId, &self.paused, "BackupWorkerPaused", "BackupWorkerResumed")); + addActor.send(monitorBackupPause(self.cx, + self.myId, + &self.paused, + /*pausedEvent=*/"BackupWorkerPaused", + /*resumedEvent=*/"BackupWorkerResumed")); // If the worker is on an old epoch and all backups starts a version >= the endVersion bool exitEarly = co_await shouldBackupWorkerExitEarly(&self); diff --git a/fdbserver/backupworker/RangePartitionedBackupWorker.cpp b/fdbserver/backupworker/RangePartitionedBackupWorker.cpp index bd655c2e12a..e28c14b86ce 100644 --- a/fdbserver/backupworker/RangePartitionedBackupWorker.cpp +++ b/fdbserver/backupworker/RangePartitionedBackupWorker.cpp @@ -1284,8 +1284,11 @@ Future rangePartitionedBackupWorker(BackupInterface interf, addActor.send(monitorRangePartitionedBackupProgress(&self)); } - addActor.send(monitorBackupPause( - self.cx, self.myId, &self.paused, "RangePartitionedBWPaused", "RangePartitionedBWResumed")); + addActor.send(monitorBackupPause(self.cx, + self.myId, + &self.paused, + /*pausedEvent=*/"RangePartitionedBWPaused", + /*resumedEvent=*/"RangePartitionedBWResumed")); // Must be sent before processPartitionMap so logSystem is populated before the partition-map peek. addActor.send(monitorLogSystemFromDbInfo(db, &self)); From 895822830a1fb48f02e62fa36222baf3fedcd7ae Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Wed, 16 Sep 2026 13:59:46 -0700 Subject: [PATCH 096/170] Make native CDC client exclusivity setup deterministic --- fdbserver/workloads/NativeCdcEndToEnd.cpp | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/fdbserver/workloads/NativeCdcEndToEnd.cpp b/fdbserver/workloads/NativeCdcEndToEnd.cpp index ce7e6374b5a..28af73c6211 100644 --- a/fdbserver/workloads/NativeCdcEndToEnd.cpp +++ b/fdbserver/workloads/NativeCdcEndToEnd.cpp @@ -1082,8 +1082,10 @@ class NativeCdcEndToEndWorkload : public TestWorkload { // the tracked consumer so later workload phases do not retain a cursor behind acknowledgements made here. Reference idleConsumer = streams.front().consumer; const Version idleStartVersion = idleConsumer->position().lastConsumedVersion; - Future idleConsume; - co_await startBlockedConsume(cx, streamId, idleConsumer, *proxy, &idleConsume); + // Check client exclusivity before yielding: committed-version progress can complete a consume before + // a status request observes read demand, even when the client correctly rejects overlapping operations. + Future idleConsume = idleConsumer->consume(); + ASSERT(!idleConsume.isReady()); Future overlappingConsume = idleConsumer->consume(); ASSERT(overlappingConsume.isReady() && overlappingConsume.isError()); ASSERT_EQ(overlappingConsume.getError().code(), error_code_client_invalid_operation); From 4737e14fb906279fb5f6c8157d537d48797b9a82 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Wed, 16 Sep 2026 14:43:10 -0700 Subject: [PATCH 097/170] Fix CDC prefetch test capacity under buggified limits --- fdbserver/cdcproxy/CDCProxy.cpp | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index 94258f829e5..57c9b8e0f05 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -2662,6 +2662,8 @@ class CDCProxyPrefetchTest { ASSERT(second->readAhead.armedFor(199)); ASSERT_EQ(test.proxy.nextTagPrefetchVersion(test.tag).get(), 200); + // The next speculative pass can require the entire buffer under buggified limits. + test.proxy.clearBufferedMutations(first); auto secondCursor = makeReference(Void(), true, 200); co_await test.proxy.bufferTagCursor(test.tag, 200, secondCursor, Never(), Prefetch::True); ASSERT_EQ(secondCursor->fetchCount(), 1); @@ -2670,7 +2672,6 @@ class CDCProxyPrefetchTest { ASSERT(!second->readAhead.provesCursor(200, second->minVersion)); ASSERT(!test.proxy.nextTagPrefetchVersion(test.tag).present()); ASSERT_EQ(test.proxy.bufferLock.activePermits(), test.proxy.bufferedBytes); - test.proxy.clearBufferedMutations(first); test.proxy.clearBufferedMutations(second); ASSERT_EQ(test.proxy.bufferLock.activePermits(), 0); co_return; From 61f0910db9b733042d5c1fbce0269b41fb89c657 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Wed, 16 Sep 2026 16:17:42 -0700 Subject: [PATCH 098/170] Remove unused native CDC blocked-consume helper --- fdbserver/workloads/NativeCdcEndToEnd.cpp | 23 ----------------------- 1 file changed, 23 deletions(-) diff --git a/fdbserver/workloads/NativeCdcEndToEnd.cpp b/fdbserver/workloads/NativeCdcEndToEnd.cpp index 28af73c6211..87b8dfb93d1 100644 --- a/fdbserver/workloads/NativeCdcEndToEnd.cpp +++ b/fdbserver/workloads/NativeCdcEndToEnd.cpp @@ -1030,29 +1030,6 @@ class NativeCdcEndToEndWorkload : public TestWorkload { co_return proxyStatus.second; } - Future startBlockedConsume(Database cx, - CDCStreamId streamId, - Reference consumer, - CDCProxyInterface proxy, - Future* outstanding) { - *outstanding = consumer->consume(); - const double deadline = now() + operationTimeout; - while (true) { - CDCProxyBufferStatus status = co_await getCurrentProxyStatus(cx, streamId, &proxy); - if (outstanding->isReady()) { - co_await *outstanding; - co_await timeoutError(consumer->acknowledge(), operationTimeout); - *outstanding = consumer->consume(); - continue; - } - if (status.activeConsumeRequests > 0 && status.readDemand > 0) { - co_return; - } - ASSERT_LT(now(), deadline); - co_await delay(0.01); - } - } - Future waitForNoActiveConsumes(Database cx, CDCStreamId streamId, CDCProxyInterface* proxy) { const double deadline = now() + operationTimeout; while (true) { From a122a18a8cac4a432927c58e9b884b45ebe49cf5 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Wed, 16 Sep 2026 18:57:03 -0700 Subject: [PATCH 099/170] Log RocksDB checkpoint initialization before publishing readiness --- fdbserver/checkpoint/RocksDBCheckpointUtils.cpp | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/fdbserver/checkpoint/RocksDBCheckpointUtils.cpp b/fdbserver/checkpoint/RocksDBCheckpointUtils.cpp index 3fcb4b1ad08..83dbc625f35 100644 --- a/fdbserver/checkpoint/RocksDBCheckpointUtils.cpp +++ b/fdbserver/checkpoint/RocksDBCheckpointUtils.cpp @@ -508,10 +508,11 @@ void RocksDBColumnFamilyReader::Reader::action(RocksDBColumnFamilyReader::Reader return; } - a.done.send(Void()); TraceEvent(SevDebug, "RocksDBCheckpointReaderInitEnd", logId) .detail("Path", path) .detail("ColumnFamily", cf->GetName()); + // Readiness lets another thread close the column family, so finish accessing it before publishing. + a.done.send(Void()); } void RocksDBColumnFamilyReader::Reader::action(RocksDBColumnFamilyReader::Reader::CloseAction& a) { From 80567c6dda1978fe8c0cd67c48496aeccd613408 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Wed, 16 Sep 2026 18:57:33 -0700 Subject: [PATCH 100/170] Test single-key coalescing in both key range map variants --- fdbclient/KeyRangeMap.cpp | 54 +++++++++++++++++++++++ fdbclient/include/fdbclient/KeyRangeMap.h | 2 + 2 files changed, 56 insertions(+) diff --git a/fdbclient/KeyRangeMap.cpp b/fdbclient/KeyRangeMap.cpp index f521f46a0d7..fd59bdba9bd 100644 --- a/fdbclient/KeyRangeMap.cpp +++ b/fdbclient/KeyRangeMap.cpp @@ -25,6 +25,10 @@ #include "fdbclient/ReadYourWrites.h" #include "flow/UnitTest.h" +#include +#include +#include + void KeyRangeActorMap::getRangesAffectedByInsertion(const KeyRangeRef& keys, std::vector& affectedRanges) { auto s = map.rangeContaining(keys.begin); if (s.begin() != keys.begin && s.value().isValid() && !s.value().isReady()) @@ -318,6 +322,56 @@ Future krmSetRangeCoalescing(Reference const& t return holdWhile(tr, krmSetRangeCoalescing_(tr.getPtr(), mapPrefix, range, maxRange, value)); } +TEST_CASE("/keyrangemap/coalesced/singleKey") { + Arena arena; + const Key mapEnd = "z\x00"_sr; + CoalescedKeyRangeMap> owning(0, mapEnd); + CoalescedKeyRefRangeMap> arenaBacked(0, mapEnd); + + auto check = [&](const auto& map, std::initializer_list> expected) { + auto actual = map.ranges().begin(); + int expectedMetric = 0; + for (auto boundary = expected.begin(); boundary != expected.end(); ++boundary) { + ASSERT(actual != map.ranges().end()); + ASSERT(actual.begin() == boundary->first); + ASSERT(actual.value() == boundary->second); + auto next = boundary + 1; + ASSERT(actual.end() == (next == expected.end() ? mapEnd : next->first)); + using StoredKey = std::decay_t; + expectedMetric += boundary->first.size() + sizeof(MapPair); + ++actual; + } + ASSERT(actual == map.ranges().end()); + ASSERT(map.sumRange(allKeys.begin, KeyRef(mapEnd)) == expectedMetric); + }; + auto insertAndCheck = [&](KeyRef key, int value, std::initializer_list> expected) { + owning.insert(key, value); + arenaBacked.insert(key, value, arena); + check(owning, expected); + check(arenaBacked, expected); + }; + + // Split a range, repeat an insertion, and extend through the immediate successor. + insertAndCheck("b"_sr, 1, { { ""_sr, 0 }, { "b"_sr, 1 }, { "b\x00"_sr, 0 } }); + insertAndCheck("b"_sr, 1, { { ""_sr, 0 }, { "b"_sr, 1 }, { "b\x00"_sr, 0 } }); + insertAndCheck("b\x00"_sr, 1, { { ""_sr, 0 }, { "b"_sr, 1 }, { "b\x00\x00"_sr, 0 } }); + + // Overwrite a boundary, then coalesce with the left, right, and both neighbors. + insertAndCheck("b"_sr, 2, { { ""_sr, 0 }, { "b"_sr, 2 }, { "b\x00"_sr, 1 }, { "b\x00\x00"_sr, 0 } }); + insertAndCheck("b"_sr, 0, { { ""_sr, 0 }, { "b\x00"_sr, 1 }, { "b\x00\x00"_sr, 0 } }); + insertAndCheck("b"_sr, 1, { { ""_sr, 0 }, { "b"_sr, 1 }, { "b\x00\x00"_sr, 0 } }); + insertAndCheck("b\x00"_sr, 0, { { ""_sr, 0 }, { "b"_sr, 1 }, { "b\x00"_sr, 0 } }); + insertAndCheck("b"_sr, 0, { { ""_sr, 0 } }); + insertAndCheck("c"_sr, 0, { { ""_sr, 0 } }); + + // Keep the beginning and end sentinels, including when keyAfter(key) equals mapEnd. + insertAndCheck(""_sr, 1, { { ""_sr, 1 }, { "\x00"_sr, 0 } }); + insertAndCheck(""_sr, 0, { { ""_sr, 0 } }); + insertAndCheck("z"_sr, 1, { { ""_sr, 0 }, { "z"_sr, 1 } }); + insertAndCheck("z"_sr, 0, { { ""_sr, 0 } }); + return Void(); +} + TEST_CASE("/keyrangemap/decoderange/aligned") { Arena arena; Key prefix = "/prefix/"_sr; diff --git a/fdbclient/include/fdbclient/KeyRangeMap.h b/fdbclient/include/fdbclient/KeyRangeMap.h index 553023058c0..8b06de1e8e4 100644 --- a/fdbclient/include/fdbclient/KeyRangeMap.h +++ b/fdbclient/include/fdbclient/KeyRangeMap.h @@ -253,6 +253,8 @@ void insertCoalescedRange(Map, Metric>& } } +// Coalesces adjacent equal-valued ranges in memory. The transaction sequencing requirements of the +// database-backed krmSetRangeCoalescing operations do not apply to this synchronous update. template void insertCoalescedKey(Map, Metric>& map, const MetricFunc& mf, From 8d2c195f2d9fdb7a7b06e799b4f8d8a3915e5c8c Mon Sep 17 00:00:00 2001 From: Jonathan Klabunde Tomer <125505367+jkt-signal@users.noreply.github.com> Date: Wed, 16 Sep 2026 20:55:00 -0700 Subject: [PATCH 101/170] move StorageServer::watchMap below counters to fix UB at destruction (#14068) fixes #13911 - 23df2b07b9b9 --- fdbrpc/CountedSectionTest.cpp | 2 +- fdbserver/storageserver/storageserver.cpp | 16 ++++++++++------ 2 files changed, 11 insertions(+), 7 deletions(-) diff --git a/fdbrpc/CountedSectionTest.cpp b/fdbrpc/CountedSectionTest.cpp index 48fa71700bf..571b737b0a0 100644 --- a/fdbrpc/CountedSectionTest.cpp +++ b/fdbrpc/CountedSectionTest.cpp @@ -1,5 +1,5 @@ /* - * CountedSecitonTest.cpp + * CountedSectionTest.cpp * * This source file is part of the FoundationDB open source project * diff --git a/fdbserver/storageserver/storageserver.cpp b/fdbserver/storageserver/storageserver.cpp index 0af10cf204d..4a4fe2cf192 100644 --- a/fdbserver/storageserver/storageserver.cpp +++ b/fdbserver/storageserver/storageserver.cpp @@ -885,12 +885,6 @@ struct StorageServer : public IStorageMetricsService { VersionedData versionedData; std::map> mutationLog; // versions (durableVersion, version] - using WatchMapKey = Key; - using WatchMapKeyHasher = boost::hash; - using WatchMapValue = Reference; - using WatchMap_t = std::unordered_map; - WatchMap_t watchMap; // keep track of server watches - public: struct PendingNewShard { PendingNewShard(uint64_t shardId, KeyRangeRef range) : shardId(format("%016llx", shardId)), range(range) {} @@ -1195,6 +1189,8 @@ struct StorageServer : public IStorageMetricsService { Reference const> db; Database cx; + // counters must be declared before every member that can own an actor (actors, watchMap, …): cancelling those + // actors runs CountedSection destructors that touch these counters, so counters must outlive them struct Counters : CommonStorageCounters { Counter allQueries, systemKeyQueries, getKeyQueries, getValueQueries, getRangeQueries, getRangeSystemKeyQueries, @@ -1350,6 +1346,14 @@ struct StorageServer : public IStorageMetricsService { } } counters; +private: + using WatchMapKey = Key; + using WatchMapKeyHasher = boost::hash; + using WatchMapValue = Reference; + using WatchMap_t = std::unordered_map; + WatchMap_t watchMap; // keep track of server watches + +public: class GetValueQuery { public: GetValueQuery(GetValueRequest request, Counters& counters) From 776f3d95b48538ead625f309974fa51685b2985c Mon Sep 17 00:00:00 2001 From: Pierce Lopez Date: Thu, 17 Sep 2026 04:08:28 -0400 Subject: [PATCH 102/170] build: always compress debug section, cleanup debug level (#13904) Debug section compression "-gz" was disabled for FDB_RELEASE because of compatibility with cpack RPM gen. This was due to "debugedit" being used, but the version in rhel9 has compressed debuginfo support. Compression helps a lot, we want it. The "ggdb" flag meant basically "g3" with the highest level of dwarf supported. Instead, use gdwarf-4 or gdwarf-5 with level g3 or g2 or g1, explicitly. Use g3 only for FULL_DEBUG_SYMBOLS, g2 only for Debug / RelWithDebInfo. For FDB_RELEASE, use just g1 for the smallest reasonable debuginfo, so it is more practical to deploy for more regular use-cases. Also: * move ENABLE_LONG_RUNNING_TESTS to tests/ * move USE_SCCACHE to ConfigureCompiler.cmake * move FDB_RELEASE(_CANDIDATE) to main CMakeLists.txt * limit parallel link jobs to 4 to avoid OOM --- CMakeLists.txt | 14 +++---- cmake/ConfigureCompiler.cmake | 73 ++++++++++++++++++++--------------- cmake/InstallLayout.cmake | 20 +++++----- tests/CMakeLists.txt | 1 + 4 files changed, 59 insertions(+), 49 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 184e28a9c4c..c7bb603fe48 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -40,6 +40,9 @@ if("${PROJECT_SOURCE_DIR}" STREQUAL "${PROJECT_BINARY_DIR}") message(FATAL_ERROR "In-source builds are forbidden") endif() +set(FDB_RELEASE OFF CACHE BOOL "This is a building of a final release") +set(FDB_RELEASE_CANDIDATE OFF CACHE BOOL "This is a building of a release candidate") + set(OPEN_FOR_IDE OFF CACHE BOOL "Open this in an IDE (won't compile/link)") @@ -54,23 +57,18 @@ if(NOT CMAKE_BUILD_TYPE AND NOT CMAKE_CONFIGURATION_TYPES) Debug CACHE STRING "Choose the type of build" FORCE) else() - message(STATUS "Setting build type to 'Release' as none was specified") + message(STATUS "Setting build type to 'Release'") set(CMAKE_BUILD_TYPE Release CACHE STRING "Choose the type of build" FORCE) - set_property(CACHE CMAKE_BUILD_TYPE PROPERTY STRINGS "Debug" "Release" - "MinSizeRel" "RelWithDebInfo") endif() + set_property(CACHE CMAKE_BUILD_TYPE PROPERTY STRINGS "Debug" "Release" + "MinSizeRel" "RelWithDebInfo") endif() set(EXECUTABLE_OUTPUT_PATH ${PROJECT_BINARY_DIR}/bin) set(LIBRARY_OUTPUT_PATH ${PROJECT_BINARY_DIR}/lib) -option(USE_SCCACHE "Use sccache if found" ON) -if(USE_SCCACHE) - find_package(sccache) -endif() - option(FDB_MEMORY_TRACKER "Compile in the sampled per-call-site memory tracker (flow/MemoryTracker)" ON) if(NOT FDB_MEMORY_TRACKER) message(STATUS "Building FoundationDB with the memory tracker compiled out") diff --git a/cmake/ConfigureCompiler.cmake b/cmake/ConfigureCompiler.cmake index 7e0b26b0404..700372b5db6 100644 --- a/cmake/ConfigureCompiler.cmake +++ b/cmake/ConfigureCompiler.cmake @@ -10,32 +10,37 @@ env_set(USE_GCOV OFF BOOL "Compile with gcov instrumentation") env_set(USE_MSAN OFF BOOL "Compile with memory sanitizer. To avoid false positives you need to dynamically link to a msan-instrumented libc++ and libc++abi, which you must compile separately. See https://github.com/google/sanitizers/wiki/MemorySanitizerLibcxxHowTo#instrumented-libc.") env_set(USE_TSAN OFF BOOL "Compile with thread sanitizer. It is recommended to dynamically link to a tsan-instrumented libc++ and libc++abi, which you can compile separately.") env_set(USE_UBSAN OFF BOOL "Compile with undefined behavior sanitizer") -env_set(FDB_RELEASE_CANDIDATE OFF BOOL "This is a building of a release candidate") -env_set(FDB_RELEASE OFF BOOL "This is a building of a final release") -env_set(USE_CCACHE OFF BOOL "Use ccache for compilation if available") +env_set(USE_CCACHE OFF BOOL "Use ccache for compilation") +env_set(USE_SCCACHE ON BOOL "Use sccache if found") env_set(USE_CLANG_TIDY OFF BOOL "Run clang-tidy during C/C++ compilation") env_set(CLANG_TIDY "" STRING "Path to clang-tidy executable (empty to auto-detect)") env_set(CLANG_TIDY_EXTRA_ARGS "" STRING "Additional clang-tidy arguments (space-separated)") -env_set(RELATIVE_DEBUG_PATHS OFF BOOL "Use relative file paths in debug info") env_set(USE_WERROR OFF BOOL "Compile with -Werror. Recommended for local development and CI.") + default_linker(_use_ld) -env_set(USE_LD "${_use_ld}" STRING - "The linker to use for building: can be LD (system default and same as DEFAULT), BFD, GOLD, or LLD - will be LLD for Clang if available, DEFAULT otherwise") +env_set(USE_LD "${_use_ld}" STRING "The linker to use for building: can be LD (system default and same as DEFAULT), BFD, GOLD, or LLD - will be LLD for Clang if available, DEFAULT otherwise") use_libcxx(_use_libcxx) env_set(USE_LIBCXX "${_use_libcxx}" BOOL "Use libc++") static_link_libcxx(_static_link_libcxx) env_set(STATIC_LINK_LIBCXX "${_static_link_libcxx}" BOOL "Statically link libstdcpp/libc++") +env_set(MAX_LINK_JOBS "4" STRING "Maximum number of link jobs to run in parallel (to avoid OOM)") + env_set(TRACE_PC_GUARD_INSTRUMENTATION_LIB "" STRING "Path to a library containing an implementation for __sanitizer_cov_trace_pc_guard. See https://clang.llvm.org/docs/SanitizerCoverage.html for more info.") env_set(PROFILE_INSTR_GENERATE OFF BOOL "If set, build FDB as an instrumentation build to generate profiles") env_set(PROFILE_INSTR_USE "" STRING "If set, build FDB with profile") + +env_set(RELATIVE_DEBUG_PATHS OFF BOOL "Use relative file paths in debug info") env_set(FULL_DEBUG_SYMBOLS OFF BOOL "Generate full debug symbols") -env_set(ENABLE_LONG_RUNNING_TESTS OFF BOOL "Add a long running tests package") +env_set(COMPRESS_DEBUG_SYMBOLS ON BOOL "Compress debug symbols") set(is_swift_compile "$") set(is_cxx_compile "$,$>") set(is_swift_link "$") set(is_cxx_link "$,$>") +set_property(GLOBAL PROPERTY JOB_POOLS link_job_pool=${MAX_LINK_JOBS}) +set(CMAKE_JOB_POOL_LINK link_job_pool) + set(USE_SANITIZER OFF) if(USE_ASAN OR USE_VALGRIND OR USE_MSAN OR USE_TSAN OR USE_UBSAN) set(USE_SANITIZER ON) @@ -94,6 +99,11 @@ if (USE_CCACHE) set(CMAKE_CXX_COMPILER_LAUNCHER "${CCACHE_PROGRAM}") endif() +if(USE_SCCACHE) + find_package(sccache) +endif() + + include(CheckFunctionExists) set(CMAKE_REQUIRED_INCLUDES stdlib.h malloc.h) if(NOT WIN32) @@ -207,36 +217,37 @@ else() add_compile_options("$<${is_cxx_compile}:-fno-omit-frame-pointer>") + # The default DWARF 5 format does not play nicely with GNU Binutils 2.39 and earlier, resulting + # in tools like addr2line omitting line numbers. + # - Our rockylinux9 gcc compile uses rh/gcc--toolset-13 -> binutils 2.40 + # - Our rockylinux9 clang compile uses rhel9 default -> binutils 2.35.2 + # - MacOS/Darwin ld and lldb do not fully support dwarf-5 either (and also use clang) if(CLANG) - # The default DWARF 5 format does not play nicely with GNU Binutils 2.39 and earlier, resulting - # in tools like addr2line omitting line numbers. We can consider removing this once we are able - # to use a version that has a fix. add_compile_options("$<${is_cxx_compile}:-gdwarf-4>") - endif() - - if(FDB_RELEASE OR FULL_DEBUG_SYMBOLS OR CMAKE_BUILD_TYPE STREQUAL "Debug") - # Configure with FULL_DEBUG_SYMBOLS=ON to generate all symbols for debugging with gdb - # Also generating full debug symbols in release builds. CPack will strip them out - # and create a debuginfo rpm - add_compile_options("$<${is_cxx_compile}:-ggdb>") else() - # Generating minimal debug symbols by default. They are sufficient for testing purposes - add_compile_options("$<${is_cxx_compile}:-ggdb1>") - endif() - - if(CLANG) - # The default DWARF 5 format does not play nicely with GNU Binutils 2.39 and earlier, resulting - # in tools like addr2line omitting line numbers. We can consider removing this once we are able - # to use a version that has a fix. - add_compile_options("$<${is_cxx_compile}:-gdwarf-4>") + add_compile_options("$<${is_cxx_compile}:-gdwarf-5>") + endif() + + # Also generate debug symbols in release builds, + # CPack will strip them out and create a debuginfo rpm. + # Just -g1 by default because g2+ is huge, and cpack rpm debugedit is very slow. + if(FULL_DEBUG_SYMBOLS) + # As much as possible, including macros etc. + add_compile_options("$<${is_cxx_compile}:-g3>") + elseif(CMAKE_BUILD_TYPE STREQUAL "Debug" OR CMAKE_BUILD_TYPE STREQUAL "RelWithDebInfo") + # Reasonable for debugging, including function locals and c++ namespaces. + add_compile_options("$<${is_cxx_compile}:-g2>") + else() + # Minimal debug symbols: enough for backtraces with line numbers, but no locals. + add_compile_options("$<${is_cxx_compile}:-g1>") endif() - if(NOT FDB_RELEASE) - # Enable compression of the debug sections. This reduces the size of the binaries several times. - # We do not enable it release builds, because CPack fails to generate debuginfo packages when - # compression is enabled + # Enable compression of the debuginfo sections, reducing size to ~ 1/3 or less. + # CPack RPM gen used to have problems with compressed debuginfo due to "debugedit" + # but recent versions including the one in rhel9 support it. + if(COMPRESS_DEBUG_SYMBOLS) add_compile_options("$<${is_cxx_compile}:-gz>") - add_link_options("$<${is_cxx_compile}:-gz>") + add_link_options("$<${is_cxx_link}:-gz>") endif() if(TRACE_PC_GUARD_INSTRUMENTATION_LIB) diff --git a/cmake/InstallLayout.cmake b/cmake/InstallLayout.cmake index f5e5a9e2c68..ed4e6e8535b 100644 --- a/cmake/InstallLayout.cmake +++ b/cmake/InstallLayout.cmake @@ -44,7 +44,7 @@ set(CPACK_PROJECT_CONFIG_FILE "${CMAKE_BINARY_DIR}/packaging/CPackConfig.cmake") # User config ################################################################################ -set(GENERATE_DEBUG_PACKAGES "${FDB_RELEASE}" CACHE BOOL "Build debug rpm/deb packages (default: only ON for FDB_RELEASE)") +set(GENERATE_DEBUG_PACKAGES ON CACHE BOOL "Build debug rpm/deb packages") ################################################################################ # Alternatives config @@ -142,6 +142,7 @@ string(REPLACE "-" "_" FDB_PACKAGE_VERSION ${FDB_VERSION}) set(CPACK_RPM_PACKAGE_GROUP ${CURRENT_GIT_VERSION}) set(CPACK_RPM_PACKAGE_LICENSE "Apache 2.0") set(CPACK_RPM_PACKAGE_NAME "foundationdb") + set(CPACK_RPM_CLIENTS-EL9_PACKAGE_NAME "${CPACK_RPM_PACKAGE_NAME}-clients") set(CPACK_RPM_CLIENTS-EL9_FILE_NAME "${CPACK_RPM_CLIENTS-EL9_PACKAGE_NAME}-${FDB_PACKAGE_VERSION}${package_version_postfix}.el9.${CMAKE_SYSTEM_PROCESSOR}.rpm") set(CPACK_RPM_CLIENTS-EL9_DEBUGINFO_FILE_NAME "${CPACK_RPM_CLIENTS-EL9_PACKAGE_NAME}-${FDB_PACKAGE_VERSION}${package_version_postfix}.el9-debuginfo.${CMAKE_SYSTEM_PROCESSOR}.rpm") @@ -173,11 +174,13 @@ set(CPACK_RPM_SERVER-VERSIONED_PACKAGE_REQUIRES "${CPACK_COMPONENT_CL set(CPACK_RPM_SERVER-VERSIONED_POST_INSTALL_SCRIPT_FILE ${CMAKE_BINARY_DIR}/packaging/multiversion/server/postinst-rpm) set(CPACK_RPM_SERVER-VERSIONED_PRE_UNINSTALL_SCRIPT_FILE ${CMAKE_BINARY_DIR}/packaging/multiversion/server/prerm) -# Versioned packages must not own RPM's global build-id links. Two otherwise -# independent client packages can contain identical binaries, and those links -# would make the packages conflict outside their versioned install tree. -set(CPACK_RPM_SPEC_MORE_DEFINE - "%if \\\"%{name}\\\" == \\\"${CPACK_RPM_CLIENTS-VERSIONED_PACKAGE_NAME}\\\"\n%define _build_id_links none\n%endif") +# Avoid VERSIONED client package conflicting with main client package of same exact version, +# due to the build-id links in /usr/lib/.build-id/ +set(CPACK_RPM_SPEC_MORE_DEFINE " +%if \\\"%{name}\\\" == \\\"${CPACK_RPM_CLIENTS-VERSIONED_PACKAGE_NAME}\\\" +%define _build_id_links none +%endif +") file(MAKE_DIRECTORY "${CMAKE_BINARY_DIR}/packaging/emptydir") fdb_install(DIRECTORY "${CMAKE_BINARY_DIR}/packaging/emptydir/" DESTINATION data COMPONENT server) @@ -191,8 +194,6 @@ set(CPACK_RPM_EXCLUDE_FROM_AUTO_FILELIST_ADDITION "/usr/lib64/cmake" "/etc/foundationdb" "/usr/lib64/pkgconfig" - "/usr/lib64/python2.7" - "/usr/lib64/python2.7/site-packages" "/var" "/var/log" "/var/lib" @@ -204,10 +205,9 @@ set(CPACK_RPM_EXCLUDE_FROM_AUTO_FILELIST_ADDITION "/usr/lib/foundationdb" "/usr/lib/cmake" "/usr/lib/foundationdb-${FDB_VERSION}${FDB_BUILDTIME_STRING}/etc/foundationdb" - ) +) set(CPACK_RPM_BUILD_SOURCE_DIRS_PREFIX "/usr/src") set(CPACK_RPM_DEBUGINFO_PACKAGE ${GENERATE_DEBUG_PACKAGES}) -#set(CPACK_RPM_BUILD_SOURCE_FDB_INSTALL_DIRS_PREFIX /usr/src) set(CPACK_RPM_COMPONENT_INSTALL ON) ################################################################################ diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index b54bc98461b..1180958b6c3 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -21,6 +21,7 @@ set_tests_properties(test_venv_setup PROPERTIES RESOURCE_LOCK TEST_VENV_SETUP) # We need some variables to configure the test setup set(ENABLE_BUGGIFY ON CACHE BOOL "Enable buggify for tests") set(ENABLE_SIMULATION_TESTS OFF CACHE BOOL "Enable simulation tests (useful if you can't run Joshua)") +set(ENABLE_LONG_RUNNING_TESTS OFF CACHE BOOL "Add a long running tests package") set(RUN_IGNORED_TESTS OFF CACHE BOOL "Run tests that are marked for ignore") set(TEST_KEEP_LOGS "FAILED" CACHE STRING "Which logs to keep (NONE, FAILED, ALL)") set(TEST_KEEP_SIMDIR "NONE" CACHE STRING "Which simfdb directories to keep (NONE, FAILED, ALL)") From 7764cf0497155c4419316e07852d3dccf119c1d5 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Wed, 16 Sep 2026 23:51:07 -0700 Subject: [PATCH 103/170] Reset ClientMetric retry state between independent writes --- fdbserver/workloads/ClientMetric.cpp | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/fdbserver/workloads/ClientMetric.cpp b/fdbserver/workloads/ClientMetric.cpp index 4a0586a3105..7cbbafcb65f 100644 --- a/fdbserver/workloads/ClientMetric.cpp +++ b/fdbserver/workloads/ClientMetric.cpp @@ -145,12 +145,19 @@ struct ClientMetricWorkload : TestWorkload { Future writeRandomKeys(Database cx, int total) { int cnt = 0; Transaction tr(cx); + bool startNewTransaction = false; try { while (true) { Error err; try { co_await delay(0.001); - tr.reset(); + if (startNewTransaction) { + // Independent writes must not inherit the previous transaction's retry backoff. + tr.fullReset(); + startNewTransaction = false; + } else { + tr.reset(); + } tr.set(Key(deterministicRandom()->randomAlphaNumeric(10)), Value(Key(deterministicRandom()->randomAlphaNumeric(10)))); co_await tr.commit(); @@ -158,6 +165,7 @@ struct ClientMetricWorkload : TestWorkload { break; } ++cnt; + startNewTransaction = true; } catch (Error& e) { err = e; } From b19e622c9f5a77b3087f249b59bf4714c9172e18 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Thu, 17 Sep 2026 07:20:08 -0700 Subject: [PATCH 104/170] Support multi-range native CDC streams in Python --- bindings/python/fdb/__init__.py | 1 + bindings/python/fdb/impl.py | 77 +++++++++++++----- bindings/python/tests/native_cdc_tests.py | 90 +++++++++++++++++++++- documentation/sphinx/source/api-python.rst | 57 ++++++++++---- 4 files changed, 192 insertions(+), 33 deletions(-) diff --git a/bindings/python/fdb/__init__.py b/bindings/python/fdb/__init__.py index 64566ed5a96..dd790895aea 100644 --- a/bindings/python/fdb/__init__.py +++ b/bindings/python/fdb/__init__.py @@ -111,6 +111,7 @@ def api_version(ver): "StreamingMode", "CdcMutationType", "CdcCursor", + "CdcKeyRange", "CdcStreamInfo", "CdcMutation", "CdcVersionedMutations", diff --git a/bindings/python/fdb/impl.py b/bindings/python/fdb/impl.py index 8d009b72a21..856ae2dd874 100644 --- a/bindings/python/fdb/impl.py +++ b/bindings/python/fdb/impl.py @@ -907,11 +907,12 @@ def wait(self): CdcStreamInfo( ctypes.string_at(stream.name.key, stream.name.key_length), stream.stream_id, - ctypes.string_at( - stream.key_range.begin_key, stream.key_range.begin_key_length - ), - ctypes.string_at( - stream.key_range.end_key, stream.key_range.end_key_length + tuple( + CdcKeyRange( + ctypes.string_at(r.begin_key, r.begin_key_length), + ctypes.string_at(r.end_key, r.end_key_length), + ) + for r in stream.ranges[: stream.range_count] ), stream.min_version, ) @@ -1421,21 +1422,40 @@ def create_transaction(self): def get_client_status(self): return Key(self.capi.fdb_database_get_client_status(self.dpointer)) - def register_cdc_stream(self, name, begin_key, end_key): - """Register a named CDC range and return a future containing its stream ID.""" + def register_cdc_stream(self, name, begin_key=None, end_key=None, *, ranges=None): + """Register a named CDC range union and return its stream ID future. + + Supply either begin_key and end_key, or ranges as an iterable of + (begin_key, end_key) pairs. The native client canonicalizes the union. + """ _require_cdc_api_version() name = keyToBytes(name) - begin_key = keyToBytes(begin_key) - end_key = keyToBytes(end_key) + if ranges is None: + if begin_key is None or end_key is None: + raise TypeError("Supply both begin_key and end_key, or ranges") + ranges = ((begin_key, end_key),) + elif begin_key is not None or end_key is not None: + raise TypeError("ranges cannot be combined with begin_key or end_key") + # Keep converted key bytes alive until the C API has copied every range. + ranges = [(keyToBytes(begin), keyToBytes(end)) for begin, end in ranges] + native_ranges = (KeyRangeStruct * len(ranges))( + *( + KeyRangeStruct( + ctypes.cast(begin, ctypes.POINTER(ctypes.c_byte)), + len(begin), + ctypes.cast(end, ctypes.POINTER(ctypes.c_byte)), + len(end), + ) + for begin, end in ranges + ) + ) return FutureUInt64( self.capi.fdb_database_register_cdc_stream( self.dpointer, name, len(name), - begin_key, - len(begin_key), - end_key, - len(end_key), + native_ranges, + len(ranges), ) ) @@ -1515,15 +1535,35 @@ class CdcCursor(NamedTuple): last_consumed_version: int +class CdcKeyRange(NamedTuple): + """A half-open CDC key range with Python-owned endpoint bytes.""" + + begin_key: bytes + end_key: bytes + + class CdcStreamInfo(NamedTuple): """A registered stream and its durable minimum required version.""" name: bytes stream_id: int - begin_key: bytes - end_key: bytes + ranges: Tuple[CdcKeyRange, ...] min_version: int + @property + def begin_key(self): + """Return the begin key of a single-range stream.""" + if len(self.ranges) != 1: + raise ValueError("Use ranges for a multi-range CDC stream") + return self.ranges[0].begin_key + + @property + def end_key(self): + """Return the end key of a single-range stream.""" + if len(self.ranges) != 1: + raise ValueError("Use ranges for a multi-range CDC stream") + return self.ranges[0].end_key + class CdcMutation(NamedTuple): """One raw mutation with Python-owned parameter bytes.""" @@ -1723,7 +1763,8 @@ class CdcStreamInfoStruct(ctypes.Structure): _fields_ = [ ("name", KeyStruct), ("stream_id", ctypes.c_uint64), - ("key_range", KeyRangeStruct), + ("ranges", ctypes.POINTER(KeyRangeStruct)), + ("range_count", ctypes.c_int), ("min_version", ctypes.c_int64), ] @@ -2264,9 +2305,7 @@ def _init_cdc_c_api(): ctypes.c_void_p, ctypes.c_void_p, ctypes.c_int, - ctypes.c_void_p, - ctypes.c_int, - ctypes.c_void_p, + ctypes.POINTER(KeyRangeStruct), ctypes.c_int, ], ctypes.c_void_p, diff --git a/bindings/python/tests/native_cdc_tests.py b/bindings/python/tests/native_cdc_tests.py index ea2d3481f68..5614bbcb0f0 100644 --- a/bindings/python/tests/native_cdc_tests.py +++ b/bindings/python/tests/native_cdc_tests.py @@ -48,7 +48,7 @@ def test_packed_c_layout(self): pointer_size = ctypes.sizeof(ctypes.c_void_p) layouts = ( (impl.KeyRangeStruct, 2 * pointer_size + 8), - (impl.CdcStreamInfoStruct, 3 * pointer_size + 28), + (impl.CdcStreamInfoStruct, 2 * pointer_size + 24), (impl.CdcMutationStruct, 2 * pointer_size + 12), (impl.CdcVersionedMutationsStruct, pointer_size + 12), ) @@ -211,6 +211,7 @@ def test_stream_lifecycle_and_result_ownership(self): (info.name, info.stream_id, info.begin_key, info.end_key), (self.name, stream_id, self.begin, self.end), ) + self.assertEqual(info.ranges, (fdb.CdcKeyRange(self.begin, self.end),)) self.assertGreaterEqual(info.min_version, 0) with self.assertRaises(AttributeError): info.name = b"different" @@ -314,6 +315,78 @@ def write_raw_mutations(tr): self.name, [stream.name for stream in wait(self.db.list_cdc_streams())] ) + def test_multi_range_union_filtering_and_resume(self): + first = fdb.CdcKeyRange(self.prefix + b"a\x00", self.prefix + b"c\xff") + second = fdb.CdcKeyRange(self.prefix + b"x\x00", self.prefix + b"z\xff") + split = self.prefix + b"b" + ranges = ( + second, + (split, first.end_key), + (first.begin_key, split), + first, + ) + stream_id = wait( + self.db.register_cdc_stream(self.name, ranges=(r for r in ranges)) + ) + self.addCleanup(lambda: wait(self.db.remove_cdc_stream(self.name))) + self.assertEqual( + wait(self.db.register_cdc_stream(self.name, ranges=(first, second))), + stream_id, + ) + info = self.stream_info() + self.assertEqual(info.ranges, (first, second)) + for field in ("begin_key", "end_key"): + with self.assertRaisesRegex(ValueError, "Use ranges"): + getattr(info, field) + with self.assertRaises(AttributeError): + info.ranges[0].begin_key = b"different" + + with wait(self.db.create_cdc_consumer(self.name)) as consumer: + + def write_sets(tr): + tr[first.begin_key] = b"first\x00\xff" + tr[second.begin_key] = b"second\x00\xff" + tr[first.end_key] = b"excluded end" + tr[self.prefix + b"gap"] = b"excluded gap" + tr[second.end_key] = b"excluded end" + + version = self.commit(write_sets) + groups = self.consume_through(consumer, version) + self.assertCountEqual( + groups[version], + ( + fdb.CdcMutation( + fdb.CdcMutationType.SET_VALUE, first.begin_key, b"first\x00\xff" + ), + fdb.CdcMutation( + fdb.CdcMutationType.SET_VALUE, + second.begin_key, + b"second\x00\xff", + ), + ), + ) + wait(consumer.acknowledge()) + cursor = consumer.get_position() + self.assertEqual( + self.stream_info().min_version, cursor.last_consumed_version + 1 + ) + + with wait(self.db.resume_cdc_consumer(cursor)) as resumed: + self.assertEqual(resumed.get_position(), cursor) + version = self.commit( + lambda tr: tr.clear_range(self.prefix, self.prefix + b"\xff") + ) + groups = self.consume_through(resumed, version) + self.assertCountEqual( + groups[version], + tuple( + fdb.CdcMutation(fdb.CdcMutationType.CLEAR_RANGE, *r) + for r in (first, second) + ), + ) + wait(resumed.acknowledge()) + self.assertEqual(self.stream_info().ranges, (first, second)) + def test_native_errors_propagate(self): with self.assertRaises(fdb.FDBError): wait(self.db.create_cdc_consumer(self.name)) @@ -321,6 +394,20 @@ def test_native_errors_propagate(self): with self.assertRaises(fdb.FDBError): wait(self.db.register_cdc_stream(self.name, self.begin, self.end + b"\x00")) self.assertEqual(self.stream_info().end_key, self.end) + for ranges in ((), ((self.end, self.begin),), ((self.begin, self.begin),)): + with self.subTest(ranges=ranges): + with self.assertRaises(fdb.FDBError): + wait(self.db.register_cdc_stream(self.name, ranges=ranges)) + for kwargs in ( + {}, + {"begin_key": self.begin}, + {"end_key": self.end}, + {"begin_key": self.begin, "ranges": ((self.begin, self.end),)}, + {"end_key": self.end, "ranges": ((self.begin, self.end),)}, + ): + with self.subTest(kwargs=kwargs): + with self.assertRaises(TypeError): + self.db.register_cdc_stream(self.name, **kwargs) def test_missing_cdc_symbols_preserves_normal_database_use(self): impl = fdb.impl @@ -375,6 +462,7 @@ def test_normal_database_use_and_cdc_version_gate(self): self.assertEqual(self.db[key], b"normal database operations still work") operations = ( lambda: self.db.register_cdc_stream(b"legacy", b"a", b"z"), + lambda: self.db.register_cdc_stream(b"legacy", ranges=((b"a", b"z"),)), lambda: self.db.remove_cdc_stream(b"legacy"), lambda: self.db.list_cdc_streams(), lambda: self.db.create_cdc_consumer(b"legacy"), diff --git a/documentation/sphinx/source/api-python.rst b/documentation/sphinx/source/api-python.rst index cf1a51b9a79..3c8f8a6bcc7 100644 --- a/documentation/sphinx/source/api-python.rst +++ b/documentation/sphinx/source/api-python.rst @@ -461,13 +461,18 @@ Database options Native change data capture (CDC) ================================ -CDC provides durable, named streams of committed mutations for a non-empty, -half-open user-key range. Select API version 800 or later with +CDC provides durable, named streams of committed mutations for a non-empty +union of half-open user-key ranges. Select API version 800 or later with :func:`api_version` before using this interface. New stream registration also requires the cluster's ``ENABLE_NATIVE_CDC`` admission knob. Existing streams can still be listed or removed while admission is disabled; consumer creation, resume, consumption, and acknowledgement also remain available. Repeating an -existing same-name, same-range registration remains idempotent. +existing same-name, same-range-union registration remains idempotent. + +Native CDC is experimental. These bindings require the multi-range CDC C +library and server implementation; older single-range CDC binaries use an +incompatible interface even if they support API version 800. Drain and remove +existing streams before upgrading all CDC-capable clients and servers together. Unlike the synchronous database key-value methods, the CDC database methods and the consumer's ``consume()`` and ``acknowledge()`` methods return @@ -489,14 +494,32 @@ application writes by using :func:`transactional`. Stream management ----------------- -.. method:: Database.register_cdc_stream(name, begin_key, end_key) - - Registers the byte-string ``name`` for ``[begin_key, end_key)`` in normal - user key space. The name and range must be non-empty. Repeating the same - name and range is idempotent; reusing a name with another range fails. +.. method:: Database.register_cdc_stream(name, begin_key=None, end_key=None, *, ranges=None) + + Registers the byte-string ``name`` for a range union in normal user key + space. Supply either ``begin_key`` and ``end_key`` for one half-open range, + or the keyword argument ``ranges`` with an iterable of + ``(begin_key, end_key)`` pairs (including :class:`CdcKeyRange` records). + Combining the two forms raises ``TypeError``. The name, range collection, + and each range must be non-empty. Supply at most 1024 ranges before + canonicalization; the encoded union must fit within the native metadata + value-size limit. + + Registration sorts the ranges and merges overlaps, duplicates, and adjacent + intervals into a canonical union. Repeating the same name and canonical + union is idempotent, regardless of input order or partitioning; reusing a + name with a different union fails. All ranges share one stream ID, cursor, + and acknowledgement frontier. Mutations in gaps are excluded, and a clear + spanning multiple ranges produces one clipped clear per intersected range. Returns a future whose value is the unsigned 64-bit stream ID as a Python ``int``. Register long-lived streams rather than a stream per request. + For example:: + + stream_id = db.register_cdc_stream( + b"commerce", ranges=[(b"order/", b"order0"), (b"payment/", b"payment0")] + ).wait() + .. method:: Database.remove_cdc_stream(name) Removes the byte-string stream name and relinquishes its unread history. @@ -542,18 +565,26 @@ native result. They remain valid after the future or consumer is released. which mutations have been delivered. Both fields are Python ``int`` values. A delivered cursor is not proof of processing or acknowledgement. -.. class:: CdcStreamInfo(name, stream_id, begin_key, end_key, min_version) +.. class:: CdcKeyRange(begin_key, end_key) + + The inclusive begin and exclusive end byte strings of one registered range. + +.. class:: CdcStreamInfo(name, stream_id, ranges, min_version) - A stream's byte-string name, integer ID, half-open registered key range, - and durable minimum required version. ``min_version`` is a retention - frontier, not a snapshot version for the registered key range. + A stream's byte-string name, integer ID, canonical range union as a tuple + of :class:`CdcKeyRange` records, and durable minimum required version. + ``min_version`` is a retention frontier, not a snapshot version for the + registered ranges. The ``begin_key`` and ``end_key`` properties are + available when the canonical union contains exactly one range; accessing + them on a multi-range stream raises ``ValueError``. Use ``ranges`` to + inspect any stream without treating gaps as registered keys. .. class:: CdcMutation(type, param1, param2) One raw mutation. ``type`` is an integer, including for unrecognized mutation types; ``param1`` and ``param2`` are byte strings. For ``SET_VALUE`` they are the key and value; for ``CLEAR_RANGE`` they are - the begin and end keys, clipped to the registered range; for atomic + the begin and end keys, clipped to a registered range; for atomic mutations they are the key and operand. CDC returns raw mutation operations, not a materialized post-mutation value for every key. From eb5639a90ed506750b8929f57c316cd8be04382a Mon Sep 17 00:00:00 2001 From: Syed Paymaan Raza <1238752+spraza@users.noreply.github.com> Date: Thu, 17 Sep 2026 12:30:55 -0700 Subject: [PATCH 105/170] [main] Forward port stale peer fixes (#14050) * Fix stale peers for client facing roles (storage server, commit proxy, grv proxy) (#13912) Client processes can keep stale RequestStream references to dead storage-server/proxy peers (via DatabaseContext::locationCache and MonitorLeader::shrinkProxyList's sampled subset), causing connection-timeout churn. Adds opt-in draining of dead-peer connections: a per-DatabaseContext peer evictor driven by a persistent per-address connect-failed counter, a shrinkProxyList cache clear below the connection cap, eager proxy rebuild on clientInfo change, and an InterfaceTracker observability layer (all gated by knobs, default off in production). invalidateCacheByAddress(es) processes ranges in chunks and yields between chunks to avoid hogging the event loop on large caches. Includes StalePeerTest, a deterministic repro workload. Ported from upstream/release-7.3 PR #13367 (51e1c42eb9a8d2a5c174497e3de9fd7cf093075e) and PR #13643 (805c271fd496bcdf650e6b608642e0efc9ef1da2). Conflicts resolved in: tests/CMakeLists.txt, fdbclient/NativeAPI.actor.cpp, fdbclient/include/fdbclient/CommitProxyInterface.h, fdbclient/include/fdbclient/StorageServerInterface.h, fdbrpc/include/fdbrpc/FlowTransport.h, fdbserver/SimulatedCluster.actor.cpp. (cherry picked from commit 2963ee831440de9761bec4d79b4dedfc0c74b1ea) * Enable stale peer fixes by default on main/8.0 Flip the three stale-peer client-facing mitigation knobs from off to on by default now that they have had bake time on 7.3/7.4: - location_cache_peer_evictor_enabled - shrink_proxy_list_clear_cache_below_threshold - dbcontext_eager_proxy_update The forward-port landed these off by default (opt-in). By the 8.0 cut they will have had sufficient production bake time, so main should ship with them on. The simulation randomization (coinflip under randomize && isSimulated) is left unchanged, so simulation still exercises both on and off states; only the production default changes. The stale_peer_observability tracing knob is intentionally left off (it is a debug aid, not a mitigation, and carries overhead). * fix clang-format * fix clang-tidy --- fdbclient/ClientKnobs.cpp | 6 + fdbclient/DatabaseContext.cpp | 208 ++++++ fdbclient/MonitorLeader.cpp | 50 ++ fdbclient/StorageServerInterface.cpp | 4 + .../include/fdbclient/CommitProxyInterface.h | 20 + fdbclient/include/fdbclient/DatabaseContext.h | 5 + .../include/fdbclient/GrvProxyInterface.h | 21 + fdbclient/include/fdbclient/Knobs.h | 37 ++ .../fdbclient/StorageServerInterface.h | 16 + fdbrpc/FlowTransport.cpp | 311 ++++++++- fdbrpc/include/fdbrpc/FlowTransport.h | 118 ++++ fdbrpc/include/fdbrpc/fdbrpc.h | 91 ++- fdbserver/workloads/StalePeerTest.cpp | 610 ++++++++++++++++++ flow/Knobs.cpp | 2 + flow/include/flow/Knobs.h | 29 + flow/include/flow/flow.h | 53 +- tests/CMakeLists.txt | 1 + tests/fast/StalePeerTest.toml | 95 +++ 18 files changed, 1661 insertions(+), 16 deletions(-) create mode 100644 fdbserver/workloads/StalePeerTest.cpp create mode 100644 tests/fast/StalePeerTest.toml diff --git a/fdbclient/ClientKnobs.cpp b/fdbclient/ClientKnobs.cpp index 0d08fdbf8fe..214a88247b1 100644 --- a/fdbclient/ClientKnobs.cpp +++ b/fdbclient/ClientKnobs.cpp @@ -159,6 +159,8 @@ void ClientKnobs::initialize(Randomize randomize, IsSimulated isSimulated) { init( MAX_CLIENT_STATUS_AGE, 1.0 ); init( MAX_COMMIT_PROXY_CONNECTIONS, 5 ); if( randomize && buggify() ) MAX_COMMIT_PROXY_CONNECTIONS = 1; init( MAX_GRV_PROXY_CONNECTIONS, 3 ); if( randomize && buggify() ) MAX_GRV_PROXY_CONNECTIONS = 1; + init( SHRINK_PROXY_LIST_CLEAR_CACHE_BELOW_THRESHOLD, true ); if( randomize && isSimulated ) SHRINK_PROXY_LIST_CLEAR_CACHE_BELOW_THRESHOLD = deterministicRandom()->coinflip(); + init( DBCONTEXT_EAGER_PROXY_UPDATE, true ); if( randomize && isSimulated ) DBCONTEXT_EAGER_PROXY_UPDATE = deterministicRandom()->coinflip(); init( STATUS_IDLE_TIMEOUT, 120.0 ); init( STATUS_TIMEOUT, 30.0 ); init( GRPC_CTL_SERVICE_DEFAULT_TIMEOUT, 5.0 ); @@ -206,6 +208,10 @@ void ClientKnobs::initialize(Randomize randomize, IsSimulated isSimulated) { init( LOCATION_CACHE_EVICTION_SIZE_SIM, 10 ); if( randomize && buggify() ) LOCATION_CACHE_EVICTION_SIZE_SIM = 3; init( LOCATION_CACHE_ENDPOINT_FAILURE_GRACE_PERIOD, 60 ); init( LOCATION_CACHE_FAILED_ENDPOINT_RETRY_INTERVAL, 60 ); + init( LOCATION_CACHE_PEER_EVICTOR_ENABLED, true ); if( randomize && isSimulated ) LOCATION_CACHE_PEER_EVICTOR_ENABLED = deterministicRandom()->coinflip(); + init( LOCATION_CACHE_PEER_EVICTOR_DELAY, 60.0 ); + init( LOCATION_CACHE_PEER_EVICTOR_FAILED_THRESHOLD, 0 ); + init( LOCATION_CACHE_PEER_EVICTOR_SCAN_CHUNK, 1000000 ); if( randomize && buggify() ) LOCATION_CACHE_PEER_EVICTOR_SCAN_CHUNK = deterministicRandom()->randomInt(1, 11); init( GET_RANGE_SHARD_LIMIT, 2 ); init( WARM_RANGE_SHARD_LIMIT, 100 ); diff --git a/fdbclient/DatabaseContext.cpp b/fdbclient/DatabaseContext.cpp index 2d872de1a6d..134341154cd 100644 --- a/fdbclient/DatabaseContext.cpp +++ b/fdbclient/DatabaseContext.cpp @@ -26,6 +26,8 @@ #include #include #include +#include +#include #include #include "fdbclient/Knobs.h" @@ -34,6 +36,7 @@ #include "fdbclient/FDBOptions.g.h" #include "fdbclient/FDBTypes.h" #include "fdbrpc/MultiInterface.h" +#include "fdbrpc/FlowTransport.h" #include "fdbclient/ClusterInterface.h" #include "fdbclient/CoordinationInterface.h" @@ -919,6 +922,13 @@ static Future monitorClientDBInfoChange(DatabaseContext* cx, // Clear the version vector to ensure the latest commit versions are received. cx->ssVersionVectorCache.clear(); proxiesChangeTrigger->trigger(); + // Eagerly rebuild the published proxy ModelInterface the instant the + // proxy list changes, so a killed proxy's RequestStream is dropped from + // cx->commitProxies/grvProxies on clientInfo rotation rather than waiting + // for the next transaction's lazy getCommitProxies()/getGrvProxies(). + if (CLIENT_KNOBS->DBCONTEXT_EAGER_PROXY_UPDATE) { + cx->updateProxies(); + } } } else if (res.index() == 1) { UNSTOPPABLE_ASSERT(false); @@ -1125,6 +1135,200 @@ void DatabaseContext::initializeSpecialCounters() { specialCounter(cc, "WatchMapSize", [this] { return watchMap.size(); }); } +// Evicts cached ranges mapping to any server address in the input addresses set +// Yields every LOCATION_CACHE_PEER_EVICTOR_SCAN_CHUNK ranges +static Future invalidateCacheByAddresses(DatabaseContext* self, std::unordered_set addresses) { + // Initial checks + if (addresses.empty()) { + co_return; + } + int rangeChunkThreshold = CLIENT_KNOBS->LOCATION_CACHE_PEER_EVICTOR_SCAN_CHUNK; + ASSERT(rangeChunkThreshold >= 1); + + // State across phase 1 and phase 2 below + std::vector rangesToInvalidate; + double startT = now(); + + // Phase 1: scan the cache in chunks, and compute invalid ranges + TraceEvent("LocationCacheInvalidatedByAddresses_Phase1_Begin") + .detail("DbId", self->dbId) + .detail("AddressCount", addresses.size()); + Key cursor = allKeys.begin; + int phase1RangesScanned = 0; + int phase1Yields = 0; + + for (;;) { + TraceEvent("LocationCacheInvalidatedByAddresses_Phase1_ChunkIter") + .suppressFor(5.0) + .detail("DbId", self->dbId) + .detail("AddressCount", addresses.size()) + .detail("InvalidatedRanges", rangesToInvalidate.size()) + .detail("RangeChunkThreshold", rangeChunkThreshold) + .detail("Phase1RangesScanned", phase1RangesScanned) + .detail("Phase1Yields", phase1Yields); + + // Process as many ranges as possible within chunk threshold + auto iter = self->locationCache.rangeContaining(cursor); + auto endIter = self->locationCache.ranges().end(); + bool rangesRemaining = false; + int currRangesScanned = 0; + for (; iter != endIter; ++iter) { + if (currRangesScanned >= rangeChunkThreshold) { + rangesRemaining = true; + break; + } + ++currRangesScanned; + cursor = iter->end(); + if (!iter->value()) { + continue; + } + auto& loc = iter->value(); + for (int i = 0; i < loc->size(); ++i) { + if (addresses.contains(loc->getInterface(i).address())) { + rangesToInvalidate.push_back(KeyRange(KeyRangeRef(iter->begin(), iter->end()))); + break; + } + } + } + phase1RangesScanned += currRangesScanned; + + if (!rangesRemaining) { + TraceEvent("LocationCacheInvalidatedByAddresses_Phase1_End") + .detail("DbId", self->dbId) + .detail("AddressCount", addresses.size()) + .detail("InvalidatedRanges", rangesToInvalidate.size()) + .detail("RangeChunkThreshold", rangeChunkThreshold) + .detail("Phase1RangesScanned", phase1RangesScanned) + .detail("Phase1Yields", phase1Yields) + .detail("Phase1Duration", now() - startT); + break; + } + + ++phase1Yields; + TraceEvent("LocationCacheInvalidatedByAddresses_Phase1_Yield") + .suppressFor(5.0) + .detail("DbId", self->dbId) + .detail("AddressCount", addresses.size()) + .detail("InvalidatedRanges", rangesToInvalidate.size()) + .detail("RangeChunkThreshold", rangeChunkThreshold) + .detail("Phase1RangesScanned", phase1RangesScanned) + .detail("Phase1Yields", phase1Yields); + co_await yield(); + } + + // Phase 2: invalidate the cache based on invalid ranges computed in Phase 1 + TraceEvent("LocationCacheInvalidatedByAddresses_Phase2_Begin") + .detail("DbId", self->dbId) + .detail("AddressCount", addresses.size()) + .detail("InvalidatedRanges", rangesToInvalidate.size()); + int phase2Idx = 0; + int phase2Yields = 0; + double phase2StartT = now(); + for (; phase2Idx < rangesToInvalidate.size(); phase2Idx++) { + self->locationCache.insert(rangesToInvalidate[phase2Idx], Reference()); + if ((phase2Idx + 1) % rangeChunkThreshold == 0) { + ++phase2Yields; + TraceEvent("LocationCacheInvalidatedByAddresses_Phase2_Yield") + .suppressFor(5.0) + .detail("DbId", self->dbId) + .detail("AddressCount", addresses.size()) + .detail("InvalidatedRanges", rangesToInvalidate.size()) + .detail("RangeChunkThreshold", rangeChunkThreshold) + .detail("Phase2RangesScanned", phase2Idx + 1) + .detail("Phase2Yields", phase2Yields); + co_await yield(); + } + } + TraceEvent("LocationCacheInvalidatedByAddresses_Phase2_End") + .detail("DbId", self->dbId) + .detail("AddressCount", addresses.size()) + .detail("InvalidatedRanges", rangesToInvalidate.size()) + .detail("RangeChunkThreshold", rangeChunkThreshold) + .detail("Phase2RangesScanned", phase2Idx) + .detail("Phase2Yields", phase2Yields) + .detail("Phase2Duration", now() - phase2StartT) + .detail("OverallDuration", now() - startT); + + co_return; +} + +// Periodically samples FlowTransport's persistent per-address connect-failed +// counter and evicts any address whose count advanced since the previous tick +// (a "flap"). This is a direct ConnectionTimeout (CTO) signal: every connect failure increments +// the counter, and any positive delta within an evictor interval indicates an +// address that is still being targeted by RPCs but cannot establish a +// connection. +static Future locationCachePeerEvictorActor(DatabaseContext* cx) { + double evictorDelay = CLIENT_KNOBS->LOCATION_CACHE_PEER_EVICTOR_DELAY; + int evictorFailedThreshold = CLIENT_KNOBS->LOCATION_CACHE_PEER_EVICTOR_FAILED_THRESHOLD; + ASSERT(evictorDelay > 0); + ASSERT(evictorFailedThreshold >= 0); + // Per-address snapshot of FlowTransport's persistent connect-failed counter + // taken on the previous tick. The delta to the current count is the flap + // signal: a positive delta means the address is still being targeted by RPCs + // but cannot connect. + std::unordered_map lastConnectFailedSnapshot; + for (;;) { + try { + co_await delay(evictorDelay); + + std::unordered_set deadAddressSet; + const auto& persistent = FlowTransport::transport().getPersistentConnectFailedCounts(); + for (const auto& [addr, cur] : persistent) { + if (!addr.isValid()) { + continue; + } + int64_t prev = 0; + auto snapIt = lastConnectFailedSnapshot.find(addr); + if (snapIt != lastConnectFailedSnapshot.end()) { + prev = snapIt->second; + } + // If the persistent counter went backwards, the entry was TTL-pruned and re-added + // since the last sweep (its count reset to a small value). Count from zero in that + // case so a genuine post-reset connect failure isn't missed for a sweep. + int64_t delta = (cur.count >= prev) ? (cur.count - prev) : cur.count; + lastConnectFailedSnapshot[addr] = cur.count; + if (delta > evictorFailedThreshold) { + TraceEvent("LocationCachePeerEvictor_FoundDeadAddr") + .suppressFor(5.0) + .detail("DbId", cx->dbId) + .detail("Addr", addr) + .detail("ConnectFailedDelta", delta) + .detail("ConnectFailedTotal", cur.count); + deadAddressSet.insert(addr); + } + } + // Drop snapshot entries for addrs FlowTransport no longer reports a counter + // for, so this map stays bounded alongside the persistent one. + for (auto it = lastConnectFailedSnapshot.begin(); it != lastConnectFailedSnapshot.end();) { + if (persistent.find(it->first) == persistent.end()) { + TraceEvent("LocationCachePeerEvictor_ClearAddrInSnapshot") + .suppressFor(5.0) + .detail("DbId", cx->dbId) + .detail("Addr", it->first); + it = lastConnectFailedSnapshot.erase(it); + } else { + ++it; + } + } + if (!deadAddressSet.empty()) { + TraceEvent("LocationCachePeerEvictor_DeadAddrSummary") + .detail("DbId", cx->dbId) + .detail("DeadAddrSetSize", deadAddressSet.size()); + } + co_await invalidateCacheByAddresses(cx, deadAddressSet); + } catch (Error& e) { + // actor_cancelled must propagate so ~DatabaseContext can tear down the + // evictor; any other error should not kill the loop (that would stop the + // eviction sweep). + if (e.code() == error_code_actor_cancelled) { + throw; + } + TraceEvent(SevWarn, "LocationCachePeerEvictor_Error").error(e).detail("DbId", cx->dbId); + } + } +} + DatabaseContext::DatabaseContext(Reference>> connectionRecord, Reference> clientInfo, Reference> const> coordinator, @@ -1201,6 +1405,9 @@ DatabaseContext::DatabaseContext(ReferenceLOCATION_CACHE_PEER_EVICTOR_ENABLED) { + locationCachePeerEvictor = locationCachePeerEvictorActor(this); + } smoothMidShardSize.reset(CLIENT_KNOBS->INIT_MID_SHARD_BYTES); globalConfig = std::make_unique(this); @@ -1499,6 +1706,7 @@ DatabaseContext::~DatabaseContext() { clientStatusUpdater.actor.cancel(); throttleExpirer.cancel(); statusLeaderMon.cancel(); + locationCachePeerEvictor.cancel(); if (grvUpdateHandler.isValid()) { grvUpdateHandler.cancel(); diff --git a/fdbclient/MonitorLeader.cpp b/fdbclient/MonitorLeader.cpp index b7d435d9ad1..b70f6913a60 100644 --- a/fdbclient/MonitorLeader.cpp +++ b/fdbclient/MonitorLeader.cpp @@ -889,6 +889,9 @@ void shrinkProxyList(ClientDBInfo& ni, std::vector& lastGrvProxyUIDs, std::vector& lastGrvProxies) { if (ni.commitProxies.size() > CLIENT_KNOBS->MAX_COMMIT_PROXY_CONNECTIONS) { + // Cache the sampled subset to prevent proxy churn: while above MAX, we keep + // talking to the same subset across ClientDBInfo updates instead of reshuffling + // on every update. std::vector commitProxyUIDs; commitProxyUIDs.reserve(ni.commitProxies.size()); for (auto& commitProxy : ni.commitProxies) { @@ -905,8 +908,40 @@ void shrinkProxyList(ClientDBInfo& ni, } ni.firstCommitProxy = ni.commitProxies[0]; ni.commitProxies = lastCommitProxies; + } else if (CLIENT_KNOBS->SHRINK_PROXY_LIST_CLEAR_CACHE_BELOW_THRESHOLD) { + // Why clear here: say MAX=5 and we recruited 6 CPs, so we sampled+cached 5 of + // them above. Now a CP is killed, a recovery occurs, but the new recruited count + // is 5 (6 -> 5), which falls at/below MAX so we land in this branch and never + // re-enter the shrink path. The cached list still holds the pre-recovery sample + // -- including the killed CP's interface -- so it keeps that CP's RequestStreams + // (and their peer references), which causes stale peer issues + // (see StalePeerTest.toml with killRole=commit_proxy). + // + // Clearing is safe: + // - Below MAX the cache is not used at all (we talk to every recruited proxy), + // so clearing has zero effect on churn or perf here. + // - The one case where the cache could have helped is 6 -> 5 -> 6 returning + // to the *same* set: baseline would reuse, we re-sample. But crossing MAX + // again means a recovery, which recruits CPs with new UIDs, so the set is + // not the same and baseline would re-sample too. + // Either way it is a one-time reshuffle, not steady-state, and + // the knob means we can experiment and have it be off in case of performance concerns. + if (!lastCommitProxyUIDs.empty()) { + CODE_PROBE(true, + "commit proxy recruited count dropped to at/below MAX_COMMIT_PROXY_CONNECTIONS, clearing cache"); + TraceEvent("ShrinkProxyListCacheCleared") + .detail("PeerRole", "commit_proxy") + .detail("PrevCachedCount", lastCommitProxyUIDs.size()) + .detail("RecruitedCount", ni.commitProxies.size()) + .detail("MaxConnections", CLIENT_KNOBS->MAX_COMMIT_PROXY_CONNECTIONS); + lastCommitProxyUIDs.clear(); + lastCommitProxies.clear(); + } } if (ni.grvProxies.size() > CLIENT_KNOBS->MAX_GRV_PROXY_CONNECTIONS) { + // Cache the sampled subset to prevent proxy churn: while above MAX, we keep + // talking to the same subset across ClientDBInfo updates instead of reshuffling + // on every update. std::vector grvProxyUIDs; grvProxyUIDs.reserve(ni.grvProxies.size()); for (auto& grvProxy : ni.grvProxies) { @@ -922,6 +957,21 @@ void shrinkProxyList(ClientDBInfo& ni, } } ni.grvProxies = lastGrvProxies; + } else if (CLIENT_KNOBS->SHRINK_PROXY_LIST_CLEAR_CACHE_BELOW_THRESHOLD) { + // Same as the commit-proxy branch above (see that comment for the why and the + // no-perf-cost reasoning): a GP is killed, a recovery occurs, but the new + // recruited count drops to at/below MAX, so we clear the cache to avoid + // pinning the pre-recovery GP interfaces. + if (!lastGrvProxyUIDs.empty()) { + CODE_PROBE(true, "grv proxy recruited count dropped to at/below MAX_GRV_PROXY_CONNECTIONS, clearing cache"); + TraceEvent("ShrinkProxyListCacheCleared") + .detail("PeerRole", "grv_proxy") + .detail("PrevCachedCount", lastGrvProxyUIDs.size()) + .detail("RecruitedCount", ni.grvProxies.size()) + .detail("MaxConnections", CLIENT_KNOBS->MAX_GRV_PROXY_CONNECTIONS); + lastGrvProxyUIDs.clear(); + lastGrvProxies.clear(); + } } } diff --git a/fdbclient/StorageServerInterface.cpp b/fdbclient/StorageServerInterface.cpp index e7019c5bbb3..a58e6e78d18 100644 --- a/fdbclient/StorageServerInterface.cpp +++ b/fdbclient/StorageServerInterface.cpp @@ -97,6 +97,10 @@ void StorageServerInterface::initEndpoints() { streams.push_back(getCheckSum.getReceiver()); streams.push_back(bulkdump.getReceiver()); FlowTransport::transport().addEndpoints(streams); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + // streams[0] is `getValue` (base endpoint); streams[1..kNumAdjustedEndpoints] are adjusted endpoints. + ASSERT(streams.size() - 1 == kNumAdjustedEndpoints); + } } // if size + hex of checksum is shorter than value, record that instead of actual value. break-even point is 12 diff --git a/fdbclient/include/fdbclient/CommitProxyInterface.h b/fdbclient/include/fdbclient/CommitProxyInterface.h index 64ee80d50e1..e497a0e662d 100644 --- a/fdbclient/include/fdbclient/CommitProxyInterface.h +++ b/fdbclient/include/fdbclient/CommitProxyInterface.h @@ -36,6 +36,7 @@ struct CommitProxyInterface { constexpr static FileIdentifier file_identifier = 8954922; + constexpr static int kNumAdjustedEndpoints = 10; enum { LocationAwareLoadBalance = 1 }; enum { AlwaysFresh = 1 }; @@ -65,6 +66,17 @@ struct CommitProxyInterface { NetworkAddress address() const { return commit.getEndpoint().getPrimaryAddress(); } NetworkAddressList addresses() const { return commit.getEndpoint().addresses; } + std::vector getEndpointTokens() const { + // Token at index 0 is the base `commit` endpoint; adjusted endpoints start at 1. + std::vector tokens; + tokens.reserve(kNumAdjustedEndpoints + 1); + tokens.push_back(commit.getEndpoint().token); + for (int i = 1; i <= kNumAdjustedEndpoints; ++i) { + tokens.push_back(commit.getEndpoint().getAdjustedEndpoint(i).token); + } + return tokens; + } + template void serialize(Archive& ar) { serializer(ar, processId, provisional, commit); @@ -85,6 +97,10 @@ struct CommitProxyInterface { PublicRequestStream(commit.getEndpoint().getAdjustedEndpoint(9)); setThrottledShard = RequestStream(commit.getEndpoint().getAdjustedEndpoint(10)); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && g_network && g_network->global(INetwork::enFlowTransport)) { + FlowTransport::transport().interfaceTracker.created( + commit.getEndpoint().getPrimaryAddress(), "CP", getEndpointTokens()); + } } } @@ -103,6 +119,10 @@ struct CommitProxyInterface { streams.push_back(expireIdempotencyId.getReceiver()); streams.push_back(setThrottledShard.getReceiver()); FlowTransport::transport().addEndpoints(streams); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + // streams[0] is `commit` (base endpoint); streams[1..kNumAdjustedEndpoints] are adjusted endpoints. + ASSERT(streams.size() - 1 == kNumAdjustedEndpoints); + } } }; diff --git a/fdbclient/include/fdbclient/DatabaseContext.h b/fdbclient/include/fdbclient/DatabaseContext.h index b4827051bce..2bcf3c15127 100644 --- a/fdbclient/include/fdbclient/DatabaseContext.h +++ b/fdbclient/include/fdbclient/DatabaseContext.h @@ -31,6 +31,7 @@ #include #include #include +#include #pragma once #include "fdbclient/FDBTypes.h" @@ -422,6 +423,10 @@ class DatabaseContext : public ReferenceCounted, public FastAll std::map server_interf; + // Periodically samples FlowTransport per-address connect-failed counts and evicts + // location-cache entries for any address whose count advanced (a dead/flapping peer). + Future locationCachePeerEvictor; + // map from ssid -> tss interface std::unordered_map tssMapping; // map from tssid -> metrics for that tss pair diff --git a/fdbclient/include/fdbclient/GrvProxyInterface.h b/fdbclient/include/fdbclient/GrvProxyInterface.h index 3c57708ce75..5523f65a2a5 100644 --- a/fdbclient/include/fdbclient/GrvProxyInterface.h +++ b/fdbclient/include/fdbclient/GrvProxyInterface.h @@ -218,6 +218,7 @@ struct GlobalConfigRefreshRequest { // information of the cluster, and handles proxied GlobalConfig requests. struct GrvProxyInterface { constexpr static FileIdentifier file_identifier = 8743216; + constexpr static int kNumAdjustedEndpoints = 3; enum { LocationAwareLoadBalance = 1 }; enum { AlwaysFresh = 1 }; @@ -240,6 +241,17 @@ struct GrvProxyInterface { NetworkAddress address() const { return getConsistentReadVersion.getEndpoint().getPrimaryAddress(); } NetworkAddressList addresses() const { return getConsistentReadVersion.getEndpoint().addresses; } + std::vector getEndpointTokens() const { + // Token at index 0 is the base `getConsistentReadVersion` endpoint; adjusted endpoints start at 1. + std::vector tokens; + tokens.reserve(kNumAdjustedEndpoints + 1); + tokens.push_back(getConsistentReadVersion.getEndpoint().token); + for (int i = 1; i <= kNumAdjustedEndpoints; ++i) { + tokens.push_back(getConsistentReadVersion.getEndpoint().getAdjustedEndpoint(i).token); + } + return tokens; + } + template void serialize(Archive& ar) { serializer(ar, processId, provisional, getConsistentReadVersion); @@ -250,6 +262,10 @@ struct GrvProxyInterface { getConsistentReadVersion.getEndpoint().getAdjustedEndpoint(2)); refreshGlobalConfig = PublicRequestStream( getConsistentReadVersion.getEndpoint().getAdjustedEndpoint(3)); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && g_network && g_network->global(INetwork::enFlowTransport)) { + FlowTransport::transport().interfaceTracker.created( + getConsistentReadVersion.getEndpoint().getPrimaryAddress(), "GP", getEndpointTokens()); + } } } @@ -260,6 +276,11 @@ struct GrvProxyInterface { streams.push_back(getHealthMetrics.getReceiver()); streams.push_back(refreshGlobalConfig.getReceiver()); FlowTransport::transport().addEndpoints(streams); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + // streams[0] is `getConsistentReadVersion` (base endpoint); streams[1..kNumAdjustedEndpoints] are adjusted + // endpoints. + ASSERT(streams.size() - 1 == kNumAdjustedEndpoints); + } } }; diff --git a/fdbclient/include/fdbclient/Knobs.h b/fdbclient/include/fdbclient/Knobs.h index 47468759a89..24b1dc89513 100644 --- a/fdbclient/include/fdbclient/Knobs.h +++ b/fdbclient/include/fdbclient/Knobs.h @@ -56,6 +56,18 @@ class SWIFT_CXX_IMMORTAL_SINGLETON_TYPE ClientKnobs : public KnobsImplcommitProxies/grvProxies immediately so a + // killed proxy's RequestStream is dropped on rotation rather than on the next transaction's + // lazy getCommitProxies()/getGrvProxies(). Off by default; randomized in simulation; + // StalePeerTest forces it on. + bool DBCONTEXT_EAGER_PROXY_UPDATE; double STATUS_IDLE_TIMEOUT; double STATUS_TIMEOUT; double GRPC_CTL_SERVICE_DEFAULT_TIMEOUT; // Default timeout for gRPC FDBCTL service requests @@ -105,6 +117,31 @@ class SWIFT_CXX_IMMORTAL_SINGLETON_TYPE ClientKnobs : public KnobsImpl 0, evictor will assert otherwise. + double LOCATION_CACHE_PEER_EVICTOR_DELAY; + // In the locationCachePeerEvictor sweep, evict an address whose persistent connect-failed + // count advanced by more than this since the previous sweep. + // 0 = any new connect failure counts. + // Value must be >= 0, evictor will assert otherwise. + // + // NOTE: this threshold must be less than the expected number of connect failures + // a dead endpoint generates in one evictor interval, otherwise eviction never triggers. + // Expected failures per interval ~ LOCATION_CACHE_PEER_EVICTOR_DELAY / + // (SERVER_REQUEST_INTERVAL + CONNECTION_MONITOR_TIMEOUT) + int LOCATION_CACHE_PEER_EVICTOR_FAILED_THRESHOLD; + // Number of location cache ranges the address-based invalidation scan processes + // before yielding to the main event loop. + // Value must be >= 1 (evictor will assert otherwise). + // If this value is > LOCATION_CACHE_EVICTION_SIZE, then this will boil down to blocking + // invalidation without any yields in between. + int LOCATION_CACHE_PEER_EVICTOR_SCAN_CHUNK; + int GET_RANGE_SHARD_LIMIT; int WARM_RANGE_SHARD_LIMIT; int STORAGE_METRICS_SHARD_LIMIT; diff --git a/fdbclient/include/fdbclient/StorageServerInterface.h b/fdbclient/include/fdbclient/StorageServerInterface.h index 4f290630d00..037b6854954 100644 --- a/fdbclient/include/fdbclient/StorageServerInterface.h +++ b/fdbclient/include/fdbclient/StorageServerInterface.h @@ -88,6 +88,7 @@ struct UpdateCommitCostRequest { struct StorageServerInterface { constexpr static FileIdentifier file_identifier = 15302073; + constexpr static int kNumAdjustedEndpoints = 26; enum { BUSY_ALLOWED = 0, BUSY_FORCE = 1, BUSY_LOCAL = 2 }; enum { LocationAwareLoadBalance = 1 }; @@ -143,6 +144,17 @@ struct StorageServerInterface { UID id() const { return uniqueID; } bool isAcceptingRequests() const { return acceptingRequests; } void startAcceptingRequests() { acceptingRequests = true; } + + std::vector getEndpointTokens() const { + // Token at index 0 is the base `getValue` endpoint; adjusted endpoints start at 1. + std::vector tokens; + tokens.reserve(kNumAdjustedEndpoints + 1); + tokens.push_back(getValue.getEndpoint().token); + for (int i = 1; i <= kNumAdjustedEndpoints; ++i) { + tokens.push_back(getValue.getEndpoint().getAdjustedEndpoint(i).token); + } + return tokens; + } void stopAcceptingRequests() { acceptingRequests = false; } bool isTss() const { return tssPairID.present(); } std::string toString() const { return id().shortString(); } @@ -160,6 +172,10 @@ struct StorageServerInterface { if (Ar::isDeserializing) { initEndpointsFromGetValue(); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && g_network && g_network->global(INetwork::enFlowTransport)) { + FlowTransport::transport().interfaceTracker.created( + getValue.getEndpoint().getPrimaryAddress(), "SS", getEndpointTokens()); + } } } bool operator==(StorageServerInterface const& s) const { return uniqueID == s.uniqueID; } diff --git a/fdbrpc/FlowTransport.cpp b/fdbrpc/FlowTransport.cpp index ec217c7c35e..7be8960b9e3 100644 --- a/fdbrpc/FlowTransport.cpp +++ b/fdbrpc/FlowTransport.cpp @@ -430,6 +430,10 @@ class TransportData { NetworkAddressCachedString localAddresses; std::vector> listeners; std::unordered_map> peers; + + std::unordered_map persistentConnectFailedCount; + double persistentConnectFailedLastPrune = 0; + // FIXME: explain what the std::pair represent: std::unordered_map> closedPeers; HealthMonitor healthMonitor; @@ -975,13 +979,56 @@ Future connectionKeeper(Reference self, } } catch (Error& e) { ++self->connectFailedCount; + // Track per-address cumulative connect failures + last-failure time. The map + // potentially has unbounded list of peers as they're having connection issues. + // To make the map bounded, evict based on TTL. + // + // Note: this prune only runs here, inside the connect-failure path, so it is + // failure-driven rather than on a timer. If connect failures stop across all + // addresses, the scan does not run again and stale entries can linger past their + // TTL until the next failure on any address. The map is still bounded -- by the + // set of addresses ever contacted, and the next failure prunes the stale ones -- + // so this is not unbounded growth, just not strictly time-based eviction. + { + double ttl = FLOW_KNOBS->PERSISTENT_CONNECT_FAILED_COUNT_TTL; + double tNow = now(); + auto& failCounts = self->transport->persistentConnectFailedCount; + auto& info = failCounts[self->destination]; + info.count++; + info.lastFailed = tNow; + TraceEvent("PersistentConnectFailed") + .suppressFor(5.0) + .detail("PeerAddr", self->destination) + .detail("ConnectFailedTotal", info.count); + if (ttl > 0 && tNow - self->transport->persistentConnectFailedLastPrune >= ttl) { + self->transport->persistentConnectFailedLastPrune = tNow; + for (auto it = failCounts.begin(); it != failCounts.end();) { + if (tNow - it->second.lastFailed >= ttl) { + TraceEvent("PersistentConnectFailedPrune") + .suppressFor(5.0) + .detail("PeerAddr", it->first) + .detail("ConnectFailedTotal", it->second.count); + it = failCounts.erase(it); + } else { + ++it; + } + } + } + } if (e.code() != error_code_connection_failed) { throw; } + TraceEvent("ConnectionTimedOut", conn ? conn->getDebugID() : UID()) .suppressFor(1.0) .detail("PeerAddr", self->destination) - .detail("PeerAddress", self->destination); + .detail("PeerAddress", self->destination) + .detail("PeerReferences", self->peerReferences) + .detail("ReliableEmpty", self->reliable.empty()) + .detail("UnsentEmpty", self->unsent.empty()) + .detail("OutstandingReplies", self->outstandingReplies) + .detail("ConnectFailedCount", self->connectFailedCount) + .detail("Connected", self->connected); throw; } @@ -1130,7 +1177,13 @@ Future connectionKeeper(Reference self, .errorUnsuppressed(e) .suppressFor(1.0) .detail("PeerAddr", self->destination) - .detail("PeerAddress", self->destination); + .detail("PeerAddress", self->destination) + .detail("PeerReferences", self->peerReferences) + .detail("ReliableEmpty", self->reliable.empty()) + .detail("UnsentEmpty", self->unsent.empty()) + .detail("OutstandingReplies", self->outstandingReplies) + .detail("ConnectFailedCount", self->connectFailedCount) + .detail("Connected", self->connected); self->connect.cancel(); self->transport->peers.erase(self->destination); self->transport->orderedAddresses.erase(self->destination); @@ -1946,6 +1999,225 @@ static Future multiVersionCleanupWorker(TransportData* self) { } } +// ==== InterfaceTracker ==== +// Bookkeeping is gated entirely on FLOW_KNOBS->STALE_PEER_OBSERVABILITY: every +// mutating method early-returns when the knob is off, every accessor returns +// empty/zero. Callers may invoke these unconditionally. + +void InterfaceTracker::created(const NetworkAddress& dstAddr, + const std::string& dstRole, + const std::vector& tokens) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return; + } + auto& entry = map[Key{ dstAddr, dstRole }]; + entry.numCreated += tokens.size(); + int64_t id = nextCreateId++; + entry.createRecords.push_back( + { id, g_network ? g_network->now() : 0.0, platform::get_backtrace(), (int)tokens.size() }); + for (const auto& tok : tokens) { + tokenToInfo[TokenKey{ dstAddr, tok }] = TokenInfo{ dstRole, id }; + } +} + +void InterfaceTracker::peerRefAdded(const NetworkAddress& addr) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return; + } + peerRefCounts[addr].added++; +} + +void InterfaceTracker::peerRefRemovedRaw(const NetworkAddress& addr) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return; + } + peerRefCounts[addr].removed++; +} + +void InterfaceTracker::peerRefRemoved(const NetworkAddress& addr, const UID& token) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return; + } + auto it = tokenToInfo.find(TokenKey{ addr, token }); + if (it != tokenToInfo.end()) { + ASSERT(map.contains(Key{ addr, it->second.role })); + auto& entry = map[Key{ addr, it->second.role }]; + entry.numDeleted++; + for (auto& rec : entry.createRecords) { + if (rec.id == it->second.createId) { + rec.numStreamsDeleted++; + break; + } + } + } +} + +int64_t InterfaceTracker::getDelta(const NetworkAddress& addr, const std::string& role) const { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return 0; + } + auto it = map.find(Key{ addr, role }); + if (it == map.end()) + return 0; + return it->second.numCreated - it->second.numDeleted; +} + +int64_t InterfaceTracker::flowReceiverCreated(const NetworkAddress& addr, const UID& token) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return -1; + } + int64_t id = nextFlowReceiverId++; + flowReceiverRecords[id] = FlowReceiverRecord{ + id, addr, token, g_network ? g_network->now() : 0.0, platform::get_backtrace(), currentCallerTag + }; + return id; +} + +void InterfaceTracker::flowReceiverDestroyed(const NetworkAddress& addr, int64_t id) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return; + } + flowReceiverRecords.erase(id); +} + +int64_t InterfaceTracker::promiseRefAdded(const NetworkAddress& addr, const UID& token) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return -1; + } + int64_t id = nextRefId++; + refRecords[id] = RefRecord{ id, addr, token, g_network ? g_network->now() : 0.0, platform::get_backtrace(), false }; + return id; +} + +void InterfaceTracker::promiseRefReleased(int64_t id) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return; + } + if (id < 0) + return; + refRecords.erase(id); +} + +int64_t InterfaceTracker::futureRefAdded(const NetworkAddress& addr, const UID& token) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return -1; + } + int64_t id = nextRefId++; + refRecords[id] = RefRecord{ id, addr, token, g_network ? g_network->now() : 0.0, platform::get_backtrace(), true }; + return id; +} + +// Clone the (addr, token) of an existing tracked future ref into a brand-new +// tracked ref. Used by FutureStream's copy ctor/assignment: a copy is a +// distinct ref and must get its own id so it is released independently, rather +// than going untracked (which would undercount live future refs). Copy the +// fields out before inserting -- the insert may rehash and invalidate `it`. +int64_t InterfaceTracker::futureRefCopied(int64_t srcId) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return -1; + } + if (srcId < 0) { + return -1; + } + auto it = refRecords.find(srcId); + if (it == refRecords.end()) { + return -1; + } + NetworkAddress addr = it->second.addr; + UID token = it->second.token; + int64_t id = nextRefId++; + refRecords[id] = RefRecord{ id, addr, token, g_network ? g_network->now() : 0.0, platform::get_backtrace(), true }; + return id; +} + +void InterfaceTracker::futureRefReleased(int64_t id) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return; + } + if (id < 0) + return; + refRecords.erase(id); +} + +void InterfaceTracker::prettyPrintLeakedReceivers(const NetworkAddress& srcAddr, + const std::vector& filterAddrs) const { + for (const auto& [id, rec] : flowReceiverRecords) { + for (const auto& filterAddr : filterAddrs) { + if (rec.addr == filterAddr) { + TraceEvent("FlowReceiverLeaked") + .detail("SrcProcess", srcAddr) + .detail("CallerTag", rec.callerTag) + .detail("DstAddress", rec.addr) + .detail("Token", rec.token) + .detail("ReceiverId", rec.id) + .detail("CreateTime", format("%.6f", rec.createTime)) + .detail("Backtrace", rec.backtrace); + } + } + } +} + +void InterfaceTracker::prettyPrintLeakedRefs(const NetworkAddress& srcAddr, + const std::vector& filterAddrs) const { + for (const auto& [id, rec] : refRecords) { + for (const auto& filterAddr : filterAddrs) { + if (rec.addr == filterAddr) { + TraceEvent(rec.isFutureRef ? "FutureRefLeaked" : "PromiseRefLeaked") + .detail("SrcProcess", srcAddr) + .detail("DstAddress", rec.addr) + .detail("Token", rec.token) + .detail("RefId", rec.id) + .detail("RefCreateTime", format("%.6f", rec.time)) + .detail("RefBacktrace", rec.backtrace); + break; + } + } + } +} + +void InterfaceTracker::prettyPrint(const NetworkAddress& srcAddr, + const std::vector& filterAddrs) const { + for (const auto& filterAddr : filterAddrs) { + for (const auto& [key, entry] : map) { + if (key.dstAddress == filterAddr) { + TraceEvent("InterfaceTrackerDump") + .detail("SrcProcess", srcAddr) + .detail("DstAddress", key.dstAddress) + .detail("DstRole", key.dstRole) + .detail("NumCreated", entry.numCreated) + .detail("NumDeleted", entry.numDeleted) + .detail("Delta", entry.numCreated - entry.numDeleted); + if (entry.numCreated > entry.numDeleted) { + for (const auto& rec : entry.createRecords) { + if (rec.numStreamsDeleted < rec.numStreams) { + TraceEvent("InterfaceTrackerLeaked") + .detail("SrcProcess", srcAddr) + .detail("DstAddress", key.dstAddress) + .detail("DstRole", key.dstRole) + .detail("CreateId", rec.id) + .detail("CreateTime", format("%.6f", rec.time)) + .detail("NumStreams", rec.numStreams) + .detail("NumStreamsDeleted", rec.numStreamsDeleted) + .detail("Backtrace", rec.backtrace); + } + } + } + } + } + } + for (const auto& filterAddr : filterAddrs) { + auto it = peerRefCounts.find(filterAddr); + if (it != peerRefCounts.end()) { + TraceEvent("InterfaceTrackerPeerRefRaw") + .detail("SrcProcess", srcAddr) + .detail("DstAddress", filterAddr) + .detail("TotalAdded", it->second.added) + .detail("TotalRemoved", it->second.removed) + .detail("RawDelta", it->second.added - it->second.removed); + } + } +} + FlowTransport::FlowTransport(uint64_t transportId, int maxWellKnownEndpoints, IPAllowList const* allowList) : self(new TransportData(transportId, maxWellKnownEndpoints, allowList)) { self->multiVersionCleanup = multiVersionCleanupWorker(self); @@ -1954,6 +2226,23 @@ FlowTransport::FlowTransport(uint64_t transportId, int maxWellKnownEndpoints, IP self->publicKeys.emplace(keyName, key.toPublic()); } } + g_futureRefReleasedCallback = [](int64_t id) { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return; + } + if (g_network && g_network->global(INetwork::enFlowTransport)) { + FlowTransport::transport().interfaceTracker.futureRefReleased(id); + } + }; + g_futureRefCopiedCallback = [](int64_t srcId) -> int64_t { + if (!FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + return -1; + } + if (g_network && g_network->global(INetwork::enFlowTransport)) { + return FlowTransport::transport().interfaceTracker.futureRefCopied(srcId); + } + return -1; + }; } FlowTransport::~FlowTransport() { @@ -1980,6 +2269,10 @@ const std::unordered_map>& FlowTransport::getAll return self->peers; } +const std::unordered_map& FlowTransport::getPersistentConnectFailedCounts() const { + return self->persistentConnectFailedCount; +} + std::vector FlowTransport::consumeReportableIncompatiblePeers() { for (auto it = self->incompatiblePeers.begin(); it != self->incompatiblePeers.end();) { if (self->multiVersionConnections.contains(it->second.first)) { @@ -2039,6 +2332,7 @@ void FlowTransport::addPeerReference(const Endpoint& endpoint, bool isStream) { } else { peer->peerReferences++; } + interfaceTracker.peerRefAdded(endpoint.getPrimaryAddress()); } void FlowTransport::removePeerReference(const Endpoint& endpoint, bool isStream) { @@ -2047,6 +2341,19 @@ void FlowTransport::removePeerReference(const Endpoint& endpoint, bool isStream) Reference peer = self->getPeer(endpoint.getPrimaryAddress()); if (peer) { peer->peerReferences--; + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY) { + interfaceTracker.peerRefRemovedRaw(endpoint.getPrimaryAddress()); + interfaceTracker.peerRefRemoved(endpoint.getPrimaryAddress(), endpoint.token); + // Per-token backtrace of which code path is releasing this peer ref. + // platform::get_backtrace() is expensive and this fires on every + // removePeerReference; suppressFor caps the per-event rate. + TraceEvent("PeerRefRemovedBacktrace") + .suppressFor(2.0) + .detail("PeerAddr", endpoint.getPrimaryAddress()) + .detail("Token", endpoint.token) + .detail("PeerReferences", peer->peerReferences) + .detail("Backtrace", platform::get_backtrace()); + } if (peer->peerReferences < 0) { TraceEvent(SevError, "InvalidPeerReferences") .detail("References", peer->peerReferences) diff --git a/fdbrpc/include/fdbrpc/FlowTransport.h b/fdbrpc/include/fdbrpc/FlowTransport.h index f67e706f267..41f756b9167 100644 --- a/fdbrpc/include/fdbrpc/FlowTransport.h +++ b/fdbrpc/include/fdbrpc/FlowTransport.h @@ -27,6 +27,7 @@ #include #include #include +#include #include "fdbrpc/DDSketch.h" #include "flow/genericactors.h" @@ -35,8 +36,111 @@ #include "flow/ProtocolVersion.h" #include "flow/Net2Packet.h" #include "flow/Arena.h" +#include "flow/Platform.h" #include "flow/PKey.h" +// Tracks creation and destruction of interface objects (StorageServerInterface, +// TLogInterface, etc.) per process. Used for debugging stale peer references. +// +// All bookkeeping methods are no-ops when FLOW_KNOBS->STALE_PEER_OBSERVABILITY +// is false. Method bodies live in fdbrpc/FlowTransport.actor.cpp to keep this +// header light. +struct InterfaceTracker { + struct Key { + NetworkAddress dstAddress; + std::string dstRole; + bool operator==(const Key& other) const { return dstAddress == other.dstAddress && dstRole == other.dstRole; } + }; + struct KeyHash { + size_t operator()(const Key& k) const { + return std::hash()(k.dstAddress) ^ std::hash()(k.dstRole); + } + }; + struct Entry { + int64_t numCreated = 0; + int64_t numDeleted = 0; + struct CreateRecord { + int64_t id; + double time; + std::string backtrace; + int numStreams; + int numStreamsDeleted = 0; + }; + std::vector createRecords; + }; + struct TokenKey { + NetworkAddress addr; + UID token; + bool operator==(const TokenKey& other) const { return addr == other.addr && token == other.token; } + }; + struct TokenKeyHash { + size_t operator()(const TokenKey& k) const { return k.addr.hash() ^ k.token.hash(); } + }; + struct TokenInfo { + std::string role; + int64_t createId; + }; + struct PeerRefCount { + int64_t added = 0; + int64_t removed = 0; + }; + struct FlowReceiverRecord { + int64_t id; + NetworkAddress addr; + UID token; + double createTime; + std::string backtrace; + std::string callerTag; + }; + struct RefRecord { + int64_t id; + NetworkAddress addr; + UID token; + double time; + std::string backtrace; + bool isFutureRef = false; + }; + + // Per (address, role): created/deleted interface counts + per-creation backtraces (the Delta source). + std::unordered_map map; + // Per (address, token): which (role, create-record) a stream token belongs to, for matching deletes to creates. + std::unordered_map tokenToInfo; + // Per address: running count of peer references added vs removed (raw peer-ref accounting). + std::unordered_map peerRefCounts; + // Live FlowReceiver records by id: who created each receiver + backtrace, erased on destroy. + std::unordered_map flowReceiverRecords; + // Live promise/future ref records by id: creation backtrace per outstanding ref, erased on release. + std::unordered_map refRecords; + // Monotonic id generators for the records above (id 0 unused; -1 means "not tracked"). + int64_t nextCreateId = 0; + int64_t nextFlowReceiverId = 0; + int64_t nextRefId = 0; + // Optional tag identifying the current call site, attached to new FlowReceiver records. + std::string currentCallerTag; + + void created(const NetworkAddress& dstAddr, const std::string& dstRole, const std::vector& tokens); + + void peerRefAdded(const NetworkAddress& addr); + void peerRefRemovedRaw(const NetworkAddress& addr); + void peerRefRemoved(const NetworkAddress& addr, const UID& token); + + int64_t getDelta(const NetworkAddress& addr, const std::string& role) const; + + int64_t flowReceiverCreated(const NetworkAddress& addr, const UID& token); + void flowReceiverDestroyed(const NetworkAddress& addr, int64_t id); + + int64_t promiseRefAdded(const NetworkAddress& addr, const UID& token); + void promiseRefReleased(int64_t id); + int64_t futureRefAdded(const NetworkAddress& addr, const UID& token); + int64_t futureRefCopied(int64_t srcId); + void futureRefReleased(int64_t id); + + void prettyPrintLeakedReceivers(const NetworkAddress& srcAddr, + const std::vector& filterAddrs) const; + void prettyPrintLeakedRefs(const NetworkAddress& srcAddr, const std::vector& filterAddrs) const; + void prettyPrint(const NetworkAddress& srcAddr, const std::vector& filterAddrs) const; +}; + class IConnection; // Applications own IDs starting at WLTOKEN_FIRST_AVAILABLE and reserve their @@ -202,6 +306,12 @@ struct Peer : public ReferenceCounted { class IPAllowList; +// Per-address connect-failure tracking on TransportData. +struct ConnectFailedInfo { + int64_t count = 0; // cumulative number of failed connect attempts + double lastFailed = 0; // now() of the most recent failure +}; + // FIXME: describe what FlowTransport represents. Is it everything // for a given process? Is it some subset of what a process uses? class FlowTransport : NonCopyable { @@ -242,6 +352,12 @@ class FlowTransport : NonCopyable { // Peers later recognized as multi-version connections are discarded without reporting. std::vector consumeReportableIncompatiblePeers(); + // Returns a per-address cumulative count of connect failures. + // The map lives on TransportData and persists across Peer destruction, + // so it remains a reliable signal for catching short-lived peers that + // all point to the same address + const std::unordered_map& getPersistentConnectFailedCounts() const; + // Returns when an incompatible peer has persisted long enough to report. Future onIncompatibleChanged(); @@ -320,6 +436,8 @@ class FlowTransport : NonCopyable { // Periodically read JWKS (RFC 7517) public key file to refresh public key set. void watchPublicKeyFile(const std::string& publicKeyFilePath); + InterfaceTracker interfaceTracker; + private: class TransportData* self; }; diff --git a/fdbrpc/include/fdbrpc/fdbrpc.h b/fdbrpc/include/fdbrpc/fdbrpc.h index f5924bb8b44..d59fe4c419a 100644 --- a/fdbrpc/include/fdbrpc/fdbrpc.h +++ b/fdbrpc/include/fdbrpc/fdbrpc.h @@ -39,13 +39,19 @@ class FlowReceiver : public NetworkMessageReceiver, public NonCopyable { Endpoint endpoint; bool m_isLocalEndpoint; bool m_stream; + int64_t m_flowReceiverId; protected: - FlowReceiver() : m_isLocalEndpoint(false), m_stream(false) {} + FlowReceiver() : m_isLocalEndpoint(false), m_stream(false), m_flowReceiverId(-1) {} FlowReceiver(Endpoint const& remoteEndpoint, bool stream) - : endpoint(remoteEndpoint), m_isLocalEndpoint(false), m_stream(stream) { + : endpoint(remoteEndpoint), m_isLocalEndpoint(false), m_stream(stream), m_flowReceiverId(-1) { FlowTransport::transport().addPeerReference(endpoint, m_stream); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && m_stream && endpoint.getPrimaryAddress().isValid() && + endpoint.getPrimaryAddress().isPublic()) { + m_flowReceiverId = FlowTransport::transport().interfaceTracker.flowReceiverCreated( + endpoint.getPrimaryAddress(), endpoint.token); + } } ~FlowReceiver() { @@ -53,6 +59,10 @@ class FlowReceiver : public NetworkMessageReceiver, public NonCopyable { FlowTransport::transport().removeEndpoint(endpoint, this); } else { FlowTransport::transport().removePeerReference(endpoint, m_stream); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && m_flowReceiverId >= 0) { + FlowTransport::transport().interfaceTracker.flowReceiverDestroyed(endpoint.getPrimaryAddress(), + m_flowReceiverId); + } } } @@ -66,6 +76,11 @@ class FlowReceiver : public NetworkMessageReceiver, public NonCopyable { endpoint = remoteEndpoint; m_stream = stream; FlowTransport::transport().addPeerReference(endpoint, m_stream); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && m_stream && endpoint.getPrimaryAddress().isValid() && + endpoint.getPrimaryAddress().isPublic()) { + m_flowReceiverId = FlowTransport::transport().interfaceTracker.flowReceiverCreated( + endpoint.getPrimaryAddress(), endpoint.token); + } } // If already a remote endpoint, returns that. Otherwise makes this @@ -908,35 +923,86 @@ class RequestStream { return getReplyUnlessFailedFor(ReplyPromise(), sustainedFailureDuration, sustainedFailureSlope); } - explicit RequestStream(const Endpoint& endpoint) : queue(new NetNotifiedQueue(0, 1, endpoint)) {} + explicit RequestStream(const Endpoint& endpoint) + : queue(new NetNotifiedQueue(0, 1, endpoint)), m_promiseRefTrackingId(-1) { + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && queue->isRemoteEndpoint() && g_network && + g_network->global(INetwork::enFlowTransport)) { + m_promiseRefTrackingId = FlowTransport::transport().interfaceTracker.promiseRefAdded( + endpoint.getPrimaryAddress(), endpoint.token); + } + } SWIFT_CXX_IMPORT_UNSAFE FutureStream getFuture() const { queue->addFutureRef(); - return FutureStream(queue); + int64_t trackingId = -1; + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && queue->isRemoteEndpoint() && g_network && + g_network->global(INetwork::enFlowTransport)) { + const auto& ep = queue->getEndpoint(TaskPriority::DefaultEndpoint); + trackingId = FlowTransport::transport().interfaceTracker.futureRefAdded(ep.getPrimaryAddress(), ep.token); + } + return FutureStream(queue, trackingId); } - RequestStream() : queue(new NetNotifiedQueue(0, 1)) {} + RequestStream() : queue(new NetNotifiedQueue(0, 1)), m_promiseRefTrackingId(-1) {} explicit RequestStream(PeerCompatibilityPolicy policy) : RequestStream() { queue->setPeerCompatibilityPolicy(policy); } - RequestStream(const RequestStream& rhs) : queue(rhs.queue) { queue->addPromiseRef(); } - RequestStream(RequestStream&& rhs) noexcept : queue(rhs.queue) { rhs.queue = 0; } + RequestStream(const RequestStream& rhs) : queue(rhs.queue), m_promiseRefTrackingId(-1) { + queue->addPromiseRef(); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && queue->isRemoteEndpoint() && g_network && + g_network->global(INetwork::enFlowTransport)) { + const auto& ep = queue->getEndpoint(TaskPriority::DefaultEndpoint); + m_promiseRefTrackingId = + FlowTransport::transport().interfaceTracker.promiseRefAdded(ep.getPrimaryAddress(), ep.token); + } + } + RequestStream(RequestStream&& rhs) noexcept : queue(rhs.queue), m_promiseRefTrackingId(rhs.m_promiseRefTrackingId) { + rhs.queue = 0; + // Transfer promise-ref tracking ownership to the moved-to object: clear rhs's id so its + // destructor does not release a tracking record now owned by *this (avoids double-release). + rhs.m_promiseRefTrackingId = -1; + } void operator=(const RequestStream& rhs) { rhs.queue->addPromiseRef(); - if (queue) + int64_t newTrackingId = -1; + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && rhs.queue->isRemoteEndpoint() && g_network && + g_network->global(INetwork::enFlowTransport)) { + const auto& ep = rhs.queue->getEndpoint(TaskPriority::DefaultEndpoint); + newTrackingId = + FlowTransport::transport().interfaceTracker.promiseRefAdded(ep.getPrimaryAddress(), ep.token); + } + if (queue) { queue->delPromiseRef(); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && m_promiseRefTrackingId >= 0 && g_network && + g_network->global(INetwork::enFlowTransport)) { + FlowTransport::transport().interfaceTracker.promiseRefReleased(m_promiseRefTrackingId); + } + } queue = rhs.queue; + m_promiseRefTrackingId = newTrackingId; } void operator=(RequestStream&& rhs) noexcept { if (queue != rhs.queue) { - if (queue) + if (queue) { queue->delPromiseRef(); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && m_promiseRefTrackingId >= 0 && g_network && + g_network->global(INetwork::enFlowTransport)) { + FlowTransport::transport().interfaceTracker.promiseRefReleased(m_promiseRefTrackingId); + } + } queue = rhs.queue; + m_promiseRefTrackingId = rhs.m_promiseRefTrackingId; rhs.queue = 0; + rhs.m_promiseRefTrackingId = -1; } } ~RequestStream() { - if (queue) + if (queue) { queue->delPromiseRef(); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && m_promiseRefTrackingId >= 0 && g_network && + g_network->global(INetwork::enFlowTransport)) { + FlowTransport::transport().interfaceTracker.promiseRefReleased(m_promiseRefTrackingId); + } + } // queue = (NetNotifiedQueue*)0xdeadbeef; } @@ -959,6 +1025,11 @@ class RequestStream { private: NetNotifiedQueue* queue; + // InterfaceTracker id for this stream's promise-ref, or -1 when not tracked. It stays -1 + // whenever STALE_PEER_OBSERVABILITY is off (the gated *RefAdded blocks never run), so the + // plain transfers of this member in the copy/move/assignment paths are no-ops in that case + // and need no knob guard; only the tracker calls (promiseRefAdded/promiseRefReleased) are gated. + int64_t m_promiseRefTrackingId; }; // Public request streams require T::verify() and reject unauthorized messages. diff --git a/fdbserver/workloads/StalePeerTest.cpp b/fdbserver/workloads/StalePeerTest.cpp new file mode 100644 index 00000000000..b8bf5eb2026 --- /dev/null +++ b/fdbserver/workloads/StalePeerTest.cpp @@ -0,0 +1,610 @@ +/* + * StalePeerTest.actor.cpp + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +// Test that verifies stale peer references are cleaned up after a process +// running a specific role is killed. Configurable via dstKillRole parameter. + +#include "fdbserver/tester/workloads.h" +#include "fdbserver/core/ServerDBInfo.h" +#include "fdbserver/core/QuietDatabase.h" +#include "fdbserver/core/FDBSimulatorProcessInfo.h" +#include "fdbrpc/SimulatorProcessInfo.h" +#include "fdbrpc/simulator.h" +#include "fdbrpc/FlowTransport.h" +#include "fdbclient/CoordinationInterface.h" +#include "fdbclient/ManagementAPI.h" +#include +#include "flow/CoroUtils.h" + +struct StalePeerTestWorkload : TestWorkload { + static constexpr auto NAME = "StalePeerTest"; + + double waitAfterKill; + std::string dstKillRole; + // Which source roles we inspect for stale peer references to the killed + // destination. "any" = every live process (server roles included). + // "tester_client" = only customer-client processes (ProcessClass::TesterClass, + // the tester processes running the workload). Server roles use the client + // library internally so their peer refs are affected by the client-side + // fixes too, but the contract we care about is that a pure customer client + // holds no stale ref to a killed client-facing role. + std::string srcCheckRole; + bool testPassed = false; + bool skippedNoTargets = false; + bool skippedClusterUnhealthy = false; + // Set when the chosen target was not referenced by any inspected source at + // kill time, so a post-kill Delta==0 would prove nothing (inconclusive). We + // skip rather than count such a run as a meaningful pass. + bool skippedVacuous = false; + // How many inspected source processes held a reference (tracked interface + // copy or live peer ref) to the killed address just before the kill. Logged + // for audit; the post-kill check is gated on this being > 0. + int preKillRefSources = 0; + + std::vector oldKillAddresses; + // Tracker role string for the killed interface, used purely for + // diagnostic per-role Delta reporting in the failure event. The pass + // criterion is the raw peer->peerReferences count, not the per-role + // Delta. Strings: + // "TLog" (tlog or log_router), "SS" (ss), + // "CP" (commit_proxy), "GP" (grv_proxy), + // "MS" (master), "RV" (resolver), + // "DD" (dd), "RK" (rk), "CC" (cluster_controller). + // Empty for coordinator (no service interface registered with the + // tracker -- diagnostic Delta dump is skipped). + std::string trackedDstRole; + + StalePeerTestWorkload(WorkloadContext const& wcx) : TestWorkload(wcx), testPassed(true) { + waitAfterKill = getOption(options, "waitAfterKill"_sr, 60.0); + dstKillRole = getOption(options, "dstKillRole"_sr, "clientFacing"_sr).toString(); + srcCheckRole = getOption(options, "srcCheckRole"_sr, "any"_sr).toString(); + // dstKillRole="clientFacing" resolves (per-run, deterministically) to one + // of the client-facing destination roles -- the roles a client + // (DatabaseContext / its peer connections) directly addresses: coordinator, + // cluster controller, commit proxy, grv proxy, storage server. Same + // seed/config picks the same role, so per-run determinism is preserved + // while a single bulk ensemble covers all client-facing dst kills. + if (dstKillRole == "clientFacing") { + static const std::vector clientDstChoices = { + "coordinator", "cluster_controller", "commit_proxy", "grv_proxy", "ss" + }; + dstKillRole = clientDstChoices[deterministicRandom()->randomInt(0, clientDstChoices.size())]; + TraceEvent("StalePeerTestPickedClientDstRole").detail("DstKillRole", dstKillRole); + } + static const std::set validRoles = { + "tlog", "ss", "commit_proxy", "grv_proxy", "master", "resolver", "dd", + "rk", "coordinator", "log_router", "cluster_controller" + }; + if (!validRoles.contains(dstKillRole)) { + TraceEvent(SevError, "StalePeerTestInvalidDstKillRole") + .detail("DstKillRole", dstKillRole) + .detail("ValidOptions", + "clientFacing, tlog, ss, commit_proxy, grv_proxy, master, resolver, dd, rk, coordinator, " + "log_router, cluster_controller"); + ASSERT(false); + } + static const std::set validSrcChecks = { "any", "tester_client" }; + if (!validSrcChecks.contains(srcCheckRole)) { + TraceEvent(SevError, "StalePeerTestInvalidSrcCheckRole") + .detail("SrcCheckRole", srcCheckRole) + .detail("ValidOptions", "any, tester_client"); + ASSERT(false); + } + } + + Future setup(Database const& cx) override { return Void(); } + + Future start(Database const& cx) override { + if (clientId != 0) + return Void(); + return _start(this, cx); + } + + void disableFailureInjectionWorkloads(std::set& out) const override { out.insert("all"); } + + // Push addr into `out` only if it's valid and not protected by the simulator. + // sim2 silently refuses to kill addresses in protectedAddresses (coordinator + // majority, HTTP servers, etc.), and a silent no-op kill would trip the + // stuck-peer-ref check on an alive process. + void addIfKillable(std::vector& out, NetworkAddress addr) { + if (addr.isValid() && !g_simulator->isProtectedAddress(addr)) { + out.push_back(addr); + } + } + + // InterfaceTracker role string for the killed dst, used for the per-role + // Delta check and pre-kill reference snapshot. Deterministic from + // dstKillRole. Empty for coordinator: its endpoints use well-known tokens + // (ClientLeaderRegInterface) that are not registered with the tracker. + std::string computeTrackerRole() const { + if (dstKillRole == "tlog" || dstKillRole == "log_router") + return "TLog"; + if (dstKillRole == "ss") + return "SS"; + if (dstKillRole == "commit_proxy") + return "CP"; + if (dstKillRole == "grv_proxy") + return "GP"; + if (dstKillRole == "master") + return "MS"; + if (dstKillRole == "resolver") + return "RV"; + if (dstKillRole == "dd") + return "DD"; + if (dstKillRole == "rk") + return "RK"; + if (dstKillRole == "cluster_controller") + return "CC"; + return ""; // coordinator + } + + // Should this source process be inspected for stale refs? Mirrors the + // srcCheckRole filter used by both the pre-kill snapshot and the post-kill + // check so the two are always over the same population. "tester_client" + // keeps only customer-client (TesterClass) processes, excluding the + // simulator's internal TestSystem driver (IP 1.1.1.1), which is TesterClass + // but makes no workload transactions. + bool isInspectedSource(ISimulator::ProcessInfo* proc) const { + if (proc->failed || proc->rebooting) + return false; + if (srcCheckRole == "tester_client" && (getSimulatorProcessClass(proc) != ProcessClass::TesterClass || + proc->address.ip == IPAddress(0x01010101))) { + return false; + } + return proc->global(INetwork::enFlowTransport) != nullptr; + } + + // Count inspected source processes that currently hold a reference to `addr`: + // either a live tracked interface copy (per-role Delta > 0) or a live peer + // reference. Used pre-kill to establish non-vacuity (the target is actually + // referenced) and to pick the most-referenced storage server to kill. + int countSourcesReferencing(const NetworkAddress& addr, const std::string& trackerRole) const { + int n = 0; + for (auto* proc : g_simulator->getAllProcesses()) { + if (!isInspectedSource(proc)) + continue; + auto* transport = static_cast((void*)proc->global(INetwork::enFlowTransport)); + bool hasRef = !trackerRole.empty() && transport->interfaceTracker.getDelta(addr, trackerRole) > 0; + if (!hasRef) { + const auto& allPeers = transport->getAllPeers(); + auto it = allPeers.find(addr); + hasRef = (it != allPeers.end() && it->second->peerReferences > 0); + } + if (hasRef) + ++n; + } + return n; + } + + // Find a process address for the given role. + std::vector findAddressesForRole(Database const& cx) { + std::vector result; + const auto& info = dbInfo->get(); + if (dstKillRole == "tlog") { + for (const auto& tlogset : info.logSystemConfig.tLogs) { + if (!tlogset.isLocal) + continue; + for (const auto& log : tlogset.tLogs) { + if (log.present()) + addIfKillable(result, log.interf().address()); + } + } + } else if (dstKillRole == "log_router") { + // Log routers live on remote DCs in fearless configs. In a + // single-region cluster there are none, in which case the + // empty-target skip path below will treat this as a no-op. + for (const auto& tlogset : info.logSystemConfig.tLogs) { + for (const auto& lr : tlogset.logRouters) { + if (lr.present()) + addIfKillable(result, lr.interf().address()); + } + } + } else if (dstKillRole == "ss") { + // SS targets are resolved in _start from the authoritative recruited + // set via getStorageServers(cx) (see the ss path there), not here. + // Scanning simulator processes by StorageClass/UnsetClass could pick a + // process that hosts no recruited SS, in which case no SS interface is + // ever tracked at that address and the per-role Delta check passes + // vacuously. Leaving this empty; _start handles ss before calling. + } else if (dstKillRole == "commit_proxy") { + for (const auto& cp : info.client.commitProxies) { + addIfKillable(result, cp.address()); + } + trackedDstRole = "CP"; + } else if (dstKillRole == "grv_proxy") { + for (const auto& gp : info.client.grvProxies) { + addIfKillable(result, gp.address()); + } + trackedDstRole = "GP"; + } else if (dstKillRole == "master") { + addIfKillable(result, info.master.address()); + } else if (dstKillRole == "resolver") { + for (const auto& rv : info.resolvers) { + addIfKillable(result, rv.address()); + } + } else if (dstKillRole == "dd") { + if (info.distributor.present()) + addIfKillable(result, info.distributor.get().address()); + } else if (dstKillRole == "rk") { + if (info.ratekeeper.present()) + addIfKillable(result, info.ratekeeper.get().address()); + } else if (dstKillRole == "cluster_controller") { + addIfKillable(result, info.clusterInterface.address()); + } else if (dstKillRole == "coordinator") { + // sim2 keeps a majority of coordinators in protectedAddresses. Any + // un-protected coordinator is a valid kill target. With + // coordinators=3 (set in StalePeerTest.toml), 2 are protected and + // the third is killable. Resolve hostnames too -- simulation + // connection strings use hostnames (e.g. fakeCoordinatorDC0M0:1) + // rather than direct NetworkAddresses, so cs.coords is typically + // empty and the candidates live in cs.hostnames. + auto connRecord = cx->getConnectionRecord(); + if (connRecord) { + auto cs = connRecord->getConnectionString(); + for (const auto& addr : cs.coords) { + addIfKillable(result, addr); + } + for (const auto& hn : cs.hostnames) { + Optional resolved = hn.resolveBlocking(); + if (resolved.present()) { + addIfKillable(result, resolved.get()); + } + } + } + } + return result; + } + + static Future _start(StalePeerTestWorkload* self, Database cx) { + // Wait for cluster to stabilize + co_await delay(10.0); + while (self->dbInfo->get().recoveryState < RecoveryState::FULLY_RECOVERED) { + co_await self->dbInfo->onChange(); + } + + // Tracker role for the killed dst (deterministic from dstKillRole), used + // for both the pre-kill reference snapshot and the post-kill Delta check. + std::string trackerRole = self->computeTrackerRole(); + + // Resolve kill targets. + std::vector targetAddresses; + if (self->dstKillRole == "ss") { + // #1 non-vacuity: target an ACTUAL recruited storage server from the + // cluster's authoritative serverList, not just any StorageClass + // process. Killing a process that hosts no recruited SS would leave no + // SS interface tracked at that address, so the per-role Delta check + // would pass without ever exercising the client-side eviction path. + std::vector ssis = co_await getStorageServers(cx); + for (const auto& ssi : ssis) { + self->addIfKillable(targetAddresses, ssi.address()); + } + } else { + targetAddresses = self->findAddressesForRole(cx); + } + + if (targetAddresses.empty()) { + // No killable addresses for this role (e.g. every coordinator / + // proxy is protected, or log_routers don't exist in a + // single-region cluster). Skip the test -- there's nothing to + // leak refs for. + TraceEvent("StalePeerTestNoTargets") + .detail("Role", self->dstKillRole) + .detail("Note", "No killable addresses found; treating as skip"); + self->skippedNoTargets = true; + co_return; + } + + TraceEvent("StalePeerTestStarting") + .detail("Role", self->dstKillRole) + .detail("TargetsFound", targetAddresses.size()); + + // Pick the target to kill. For SS, choose the recruited server that the + // most inspected sources currently reference (cached interface / live + // peer ref) so the kill actually exercises client-side eviction, polling + // briefly (bounded) to let caches warm under the ReadWrite workload. For + // other roles the candidates are interchangeable, so take the first. In + // all cases snapshot how many inspected sources reference the chosen + // target right before the kill -- that count is the non-vacuity signal. + NetworkAddress oldAddr; + if (self->dstKillRole == "ss") { + double warmDeadline = now() + 30.0; + for (;;) { + NetworkAddress best; + int bestN = -1; + for (const auto& a : targetAddresses) { + int n = self->countSourcesReferencing(a, trackerRole); + if (n > bestN) { + bestN = n; + best = a; + } + } + if (bestN > 0 || now() >= warmDeadline) { + oldAddr = best; + self->preKillRefSources = bestN; + break; + } + co_await delay(2.0); + } + } else { + oldAddr = targetAddresses[0]; + self->preKillRefSources = self->countSourcesReferencing(oldAddr, trackerRole); + } + + // Non-vacuity gate: if no inspected source referenced the target at kill + // time, a post-kill Delta==0 would prove nothing. Mark the run + // inconclusive (skip) rather than recording it as a meaningful pass. + // Coordinator is exempt -- its endpoints use well-known tokens that are + // not tracked here, and its meaningful signal is connection-string + // removal (handled below), not a peer-ref drain. + if (self->dstKillRole != "coordinator" && self->preKillRefSources <= 0) { + TraceEvent(SevWarnAlways, "StalePeerTestVacuousNoPreKillRef") + .detail("Role", self->dstKillRole) + .detail("Address", oldAddr) + .detail("TrackerRole", trackerRole) + .detail("Note", "No inspected source referenced the target at kill time; skipping as inconclusive"); + self->skippedVacuous = true; + co_return; + } + + ISimulator::ProcessInfo* proc = g_simulator->getProcessByAddress(oldAddr); + if (!proc || proc->failed) { + TraceEvent(SevError, "StalePeerTestProcessNotFound").detail("Address", oldAddr); + self->testPassed = false; + co_return; + } + + self->oldKillAddresses.push_back(oldAddr); + + TraceEvent("StalePeerTestKilling") + .detail("Role", self->dstKillRole) + .detail("Address", oldAddr) + .detail("PreKillRefSources", self->preKillRefSources) + .detail("ProcessClass", getSimulatorProcessClass(proc).toString()) + .detail("Zone", proc->locality.zoneId()); + + g_simulator->killProcess(proc, ISimulator::KillType::KillInstantly); + TraceEvent("StalePeerTestKillDone") + .detail("Address", oldAddr) + .detail("Role", self->dstKillRole) + .detail("ProcFailedFlag", proc->failed); + + // Guard: sim2 silently refuses to kill protected addresses. If our + // candidate filter missed one, fail fast with a clear error rather + // than letting the later stale-peer-ref check attribute the refs on + // an alive process to a real leak. + if (!proc->failed) { + TraceEvent(SevError, "StalePeerTestKillIneffective") + .detail("Address", oldAddr) + .detail("Protected", g_simulator->isProtectedAddress(oldAddr)) + .detail("Rebooting", proc->rebooting); + self->testPassed = false; + co_return; + } + + // Wait for the kill's recovery + cleanup. Right after killProcess the + // broadcast dbInfo still carries the pre-kill FULLY_RECOVERED for a few + // seconds, so a bare wait-for-FULLY_RECOVERED races straight through + // without actually waiting for the kill's recovery -- then the later + // peer-ref check lands mid-recovery and skips as "cluster unhealthy". + // First wait (bounded) for the kill to drop recoveryState below + // FULLY_RECOVERED (recovery triggered); roles whose loss is handled + // without a master recovery (ss/dd/rk) simply hit this timeout. Then + // wait (bounded) for the recovery to complete. Both bounds fall through + // to the post-waitAfterKill cluster-health guard if exceeded, so this + // can never hang. + TraceEvent("StalePeerTestWaitingForRecovery"); + double recoveryTriggerDeadline = now() + 30.0; + while (self->dbInfo->get().recoveryState >= RecoveryState::FULLY_RECOVERED && now() < recoveryTriggerDeadline) { + co_await race(self->dbInfo->onChange(), delay(1.0)); + } + double recoveryCompleteDeadline = now() + 180.0; + while (self->dbInfo->get().recoveryState < RecoveryState::FULLY_RECOVERED && now() < recoveryCompleteDeadline) { + co_await race(self->dbInfo->onChange(), delay(5.0)); + } + TraceEvent("StalePeerTestRecovered").detail("RecoveryState", (int)self->dbInfo->get().recoveryState); + + // Coordinator-only: the dead coordinator stays in the cluster's + // connection string indefinitely, so every live process keeps a + // long-term LeaderMonitor connection to its address (PeerRef = 1 + // per process). That isn't a stale-ref leak -- it's the design + // pattern for cluster identity. Trigger an auto quorum change to + // swap the dead coordinator for a healthy candidate, mirroring + // what `coordinators auto` does in fdbcli. Once the dead address + // is no longer in the connection string the LeaderMonitor refs + // drop and the strict peerRefs == 0 check is meaningful. + if (self->dstKillRole == "coordinator") { + TraceEvent("StalePeerTestCoordinatorAutoChange"); + int autoChangeAttempts = 0; + for (;;) { + ++autoChangeAttempts; + CoordinatorsResult res = co_await changeQuorum(cx, autoQuorumChange()); + TraceEvent("StalePeerTestCoordinatorAutoChangeResult") + .detail("Attempt", autoChangeAttempts) + .detail("Result", (int)res); + if (res == CoordinatorsResult::SUCCESS || res == CoordinatorsResult::SAME_NETWORK_ADDRESSES) { + break; + } + if (autoChangeAttempts >= 20) { + TraceEvent(SevWarn, "StalePeerTestCoordinatorAutoChangeGivingUp").detail("LastResult", (int)res); + break; + } + co_await delay(1.0); + } + } + + TraceEvent("StalePeerTestWaiting").detail("WaitSeconds", self->waitAfterKill); + co_await delay(self->waitAfterKill); + + // Cluster-health guard: if the cluster never recovered to + // FULLY_RECOVERED within the wait window (e.g. perpetual_storage_wiggle + // + ssd-sharded-rocksdb configs that produce RkSSListFetchTimeout + // before the kill, or a degenerate recovery loop after the kill), the + // peer-ref check is meaningless -- the leak signal will be dominated + // by the cluster meltdown rather than the kill we're testing. Skip + // rather than fail: the contract this test verifies is "kill of role X + // drains its peer refs in a healthy cluster", not "every cluster + // configuration recovers within 240s". + if (self->dbInfo->get().recoveryState < RecoveryState::FULLY_RECOVERED) { + TraceEvent("StalePeerTestClusterUnhealthy") + .detail("RecoveryState", (int)self->dbInfo->get().recoveryState) + .detail("KillRole", self->dstKillRole) + .detail("Note", "Cluster did not return to FULLY_RECOVERED; skipping peer-ref check"); + self->skippedClusterUnhealthy = true; + co_return; + } + + // Pass criterion: per-role InterfaceTracker Delta == 0 at the killed + // address. Delta counts leaked COPIES of the killed role's interface + // RequestStreams still pinned somewhere -- i.e. the actual stale-interface + // leak our client-side fixes target. We deliberately do NOT use raw + // peer->peerReferences: a tester client legitimately retains + // connection-level / well-known-token refs to the killed address (the + // draining TCP connection, ping/leader-monitor endpoints, and refs to a + // co-resident role the process also hosted -- sim co-locates many roles + // per process). Those are real, expected peers (peerReferences > 0 with + // Delta == 0), not the staleness bug. peerReferences is logged as a + // diagnostic. Delta requires the InterfaceTracker, so this mode needs + // stale_peer_observability = true (deterministic: toggling it does not + // change simulator ordering). + // + // Coordinator is special -- its endpoints use well-known tokens + // (ClientLeaderRegInterface) and aren't tracked, and every process keeps + // a long-term LeaderMonitor connection to each coordinator address. The + // meaningful signal for a coordinator kill is whether the cluster removed + // the dead coord from the connection string; that is the autoQuorumChange + // result above, so no peer-ref check is done for coord. + // + // KNOWN COVERAGE GAP (intentional): a coordinator kill therefore asserts + // only that the dead coordinator was swapped out of the connection string + // (the autoQuorumChange succeeded) -- it does NOT assert that every source + // drained its peer ref to the dead address. The old refs do get dropped, but + // asynchronously: they live in clientLeaderServers (a state vector in + // monitorProxiesOneGeneration) and only go away when that generation rolls over + // to the new connection string. We check right after autoQuorumChange, which + // doesn't guarantee that rollover has happened yet, so a lingering ref could be present + // becaus of timing. + // Since dstKillRole="clientFacing" resolves uniformly across {coordinator, + // cluster_controller, commit_proxy, grv_proxy, ss}, roughly one in five + // clientFacing runs lands on coordinator and exercises only this weaker + // connection-string check. The per-role stale-interface drain is covered + // by the other four roles. (Tightening the coordinator path to a strict + // peer-ref assertion -- by waiting for the generation rollover -- is + // a possible follow-up.) + if (self->dstKillRole == "coordinator") { + TraceEvent("StalePeerTestChecking").detail("Mode", "coordinator/skip-peer-refs"); + co_return; + } + TraceEvent("StalePeerTestChecking"); + + // trackerRole was computed once at the top of _start (computeTrackerRole) + // and used for the pre-kill snapshot; reuse it here. It is non-empty for + // every role that reaches this point (coordinator returned above). + ASSERT(!trackerRole.empty()); + + auto allProcesses = g_simulator->getAllProcesses(); + int clientSourcesChecked = 0; // source processes inspected (post-filter) + int clientSourcesWithLeak = 0; // ...of which had a per-role Delta > 0 (a leak) + for (auto* proc : allProcesses) { + // Same source population as the pre-kill snapshot (see + // isInspectedSource): drops failed/rebooting and, in "tester_client" + // mode, keeps only customer-client (TesterClass) processes while + // excluding the simulator's internal TestSystem driver (IP 1.1.1.1). + if (!self->isInspectedSource(proc)) + continue; + auto* transport = static_cast((void*)proc->global(INetwork::enFlowTransport)); + ++clientSourcesChecked; + for (const auto& oldKillAddr : self->oldKillAddresses) { + auto& allPeers = transport->getAllPeers(); + auto it = allPeers.find(oldKillAddr); + const int peerRefs = (it != allPeers.end()) ? it->second->peerReferences : 0; + const int64_t delta = transport->interfaceTracker.getDelta(oldKillAddr, trackerRole); + if (delta > 0) { + ++clientSourcesWithLeak; + } + if (delta > 0) { + transport->interfaceTracker.prettyPrint(proc->address, self->oldKillAddresses); + transport->interfaceTracker.prettyPrintLeakedReceivers(proc->address, self->oldKillAddresses); + transport->interfaceTracker.prettyPrintLeakedRefs(proc->address, self->oldKillAddresses); + TraceEvent(SevError, "StalePeerTestFailed") + .detail("CheckedProcess", proc->address) + .detail("CheckedProcessClass", getSimulatorProcessClass(proc).toString()) + .detail("SrcCheckRole", self->srcCheckRole) + .detail("OldAddress", oldKillAddr) + .detail("KillRole", self->dstKillRole) + .detail("TrackerRole", trackerRole) + .detail("TrackerDelta", delta) + .detail("PeerReferences", peerRefs); + self->testPassed = false; + } + } + } + // Coverage visibility: how many source processes we examined, and how + // many showed a per-role interface leak (Delta > 0). This is logged for + // every run (pass or fail) so the aggregate leak signal is visible even + // when the run passes. PreKillRefSources records how many sources held a + // reference to the target before the drain -- it is > 0 here by + // construction (the non-vacuity gate above skips the run otherwise). + TraceEvent("StalePeerTestCheckCoverage") + .detail("SrcCheckRole", self->srcCheckRole) + .detail("KillRole", self->dstKillRole) + .detail("PreKillRefSources", self->preKillRefSources) + .detail("ClientSourcesChecked", clientSourcesChecked) + .detail("ClientSourcesWithLeak", clientSourcesWithLeak); + + // Non-vacuity assertion: a real check must have inspected at least one + // source. Reaching here means we killed a target that was referenced + // pre-kill (preKillRefSources > 0), so the inspected population cannot be + // empty. If it somehow is, the run validated nothing -- fail loudly + // rather than report a hollow pass. + if (clientSourcesChecked == 0) { + TraceEvent(SevError, "StalePeerTestNoSourcesInspected") + .detail("KillRole", self->dstKillRole) + .detail("SrcCheckRole", self->srcCheckRole) + .detail("PreKillRefSources", self->preKillRefSources); + self->testPassed = false; + } + + co_return; + } + + Future check(Database const& cx) override { + if (clientId != 0) { + return true; + } + if (oldKillAddresses.empty() && !skippedNoTargets && !skippedVacuous) { + TraceEvent(SevError, "StalePeerTestNoProcessKilled").detail("KillRole", dstKillRole); + testPassed = false; + } + TraceEvent("StalePeerTestResult") + .detail("Passed", testPassed) + .detail("KillRole", dstKillRole) + .detail("SrcCheckRole", srcCheckRole) + .detail("SkippedNoTargets", skippedNoTargets) + .detail("SkippedClusterUnhealthy", skippedClusterUnhealthy) + .detail("SkippedVacuous", skippedVacuous) + .detail("PreKillRefSources", preKillRefSources) + .detail("ProcessesKilled", oldKillAddresses.size()); + return testPassed; + } + + void getMetrics(std::vector& m) override {} +}; + +WorkloadFactory StalePeerTestWorkloadFactory; diff --git a/flow/Knobs.cpp b/flow/Knobs.cpp index 9932b696f04..377d542b955 100644 --- a/flow/Knobs.cpp +++ b/flow/Knobs.cpp @@ -56,7 +56,9 @@ void FlowKnobs::initialize(Randomize randomize, IsSimulated isSimulated) { init( ENABLE_COORDINATOR_DNS_CACHE, false ); if( randomize && buggify() ) ENABLE_COORDINATOR_DNS_CACHE = true; init( COORDINATOR_DNS_CACHE_REFRESH_INTERVAL, 3.0 ); init( COORDINATOR_DNS_CACHE_TTL, 30.0 ); + init( STALE_PEER_OBSERVABILITY, false ); init( CACHE_REFRESH_INTERVAL_WHEN_ALL_ALTERNATIVES_FAILED, 1.0 ); + init( PERSISTENT_CONNECT_FAILED_COUNT_TTL, isSimulated ? 120.0 : 600.0 ); init( DELAY_JITTER_OFFSET, 0.9 ); init( DELAY_JITTER_RANGE, 0.2 ); diff --git a/flow/include/flow/Knobs.h b/flow/include/flow/Knobs.h index c003d8905bc..bf56084d39f 100644 --- a/flow/include/flow/Knobs.h +++ b/flow/include/flow/Knobs.h @@ -108,6 +108,35 @@ class FlowKnobs : public KnobsImpl { double COORDINATOR_DNS_CACHE_TTL; double CACHE_REFRESH_INTERVAL_WHEN_ALL_ALTERNATIVES_FAILED; + // When true, enables tracing to debug stale peer issues (see StalePeerTest.toml). + // Not recommended for production due to potential performance overhead. + // + // Debugging workflow: if a simulation test fails with a dangling peer reference + // error, re-run the same seed with stale_peer_observability=true. This causes + // interfaceTracker to emit trace events recording creation and destruction at + // different layers of the RPC stack: interface, request stream, flow receiver. + // See InterfaceTracker-related traces in FlowTransport.cpp. These traces include + // backtraces that can be used to identify where the leak originated. From there, + // reason about what the code is doing, why it runs into the stale peer issue, + // and what can be done to prune the stale peers. The fix will vary case by case. + bool STALE_PEER_OBSERVABILITY; + + // Used for two purposes off the same value: (1) eviction age -- an address is + // pruned from TransportData::persistentConnectFailedCount once it has had no + // connect failure for this many seconds; and (2) sweep cadence -- the prune + // scan runs at most once per this interval (and only when triggered by a + // connect failure). Bounds that map and the locationCachePeerEvictor + // snapshot/streak maps it feeds. 0 disables pruning. Coupling both onto one + // knob is a deliberate simplification: sweep about as often as the eviction + // horizon. + // + // NOTE: this value should be set to a multiple of LOCATION_CACHE_PEER_EVICTOR_DELAY + // so that a dead address is not pruned from this map before the evictor has a + // chance to observe its failure delta and evict the corresponding location cache entries. + // It should also be a multiple of the reconnect interval, so that an address that is + // still being targeted by RPCs re-fails and refreshes its timestamp before the TTL, + // and is never pruned while actively dead. + double PERSISTENT_CONNECT_FAILED_COUNT_TTL; double DELAY_JITTER_OFFSET; double DELAY_JITTER_RANGE; double BUSY_WAIT_THRESHOLD; diff --git a/flow/include/flow/flow.h b/flow/include/flow/flow.h index 86926a3c56d..500f17729b7 100644 --- a/flow/include/flow/flow.h +++ b/flow/include/flow/flow.h @@ -1281,6 +1281,20 @@ struct NotifiedQueue : private SingleCallback bool shouldFireImmediately() { return SingleCallback::next != this; } }; +// Global callback for futureRef tracking release, set by FlowTransport at init +inline void (*g_futureRefReleasedCallback)(int64_t id) = nullptr; + +// Global callback for futureRef tracking copy, set by FlowTransport at init. +// Given the source FutureStream's tracking id, registers a brand-new tracked +// ref (cloning the source's addr/token) and returns its id. A copied +// FutureStream is a distinct ref (it bumps the queue's future ref count), so it +// needs its own tracking id rather than sharing the source's -- otherwise +// copies would go untracked (the move ctor propagates the id, but a copy can't +// steal it). The flow layer has no endpoint context of its own, so this defers +// to FlowTransport, which holds the source record's addr/token. Returns -1 when +// the source isn't tracked or observability is off. +inline int64_t (*g_futureRefCopiedCallback)(int64_t srcId) = nullptr; + template class SWIFT_SENDABLE FutureStream { public: @@ -1294,26 +1308,53 @@ class SWIFT_SENDABLE FutureStream { void addCallbackAndClear(SingleCallback* cb) { queue->addCallbackAndDelFutureRef(cb); queue = nullptr; + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && m_futureRefTrackingId >= 0 && g_futureRefReleasedCallback) { + g_futureRefReleasedCallback(m_futureRefTrackingId); + } + m_futureRefTrackingId = -1; + } + FutureStream() : queue(nullptr), m_futureRefTrackingId(-1) {} + FutureStream(const FutureStream& rhs) : queue(rhs.queue), m_futureRefTrackingId(-1) { + queue->addFutureRef(); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && rhs.m_futureRefTrackingId >= 0 && g_futureRefCopiedCallback) { + m_futureRefTrackingId = g_futureRefCopiedCallback(rhs.m_futureRefTrackingId); + } + } + FutureStream(FutureStream&& rhs) noexcept : queue(rhs.queue), m_futureRefTrackingId(rhs.m_futureRefTrackingId) { + rhs.queue = 0; + rhs.m_futureRefTrackingId = -1; } - FutureStream() : queue(nullptr) {} - FutureStream(const FutureStream& rhs) : queue(rhs.queue) { queue->addFutureRef(); } - FutureStream(FutureStream&& rhs) noexcept : queue(rhs.queue) { rhs.queue = 0; } ~FutureStream() { if (queue) queue->delFutureRef(); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && m_futureRefTrackingId >= 0 && g_futureRefReleasedCallback) { + g_futureRefReleasedCallback(m_futureRefTrackingId); + } } void operator=(const FutureStream& rhs) { rhs.queue->addFutureRef(); if (queue) queue->delFutureRef(); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && m_futureRefTrackingId >= 0 && g_futureRefReleasedCallback) { + g_futureRefReleasedCallback(m_futureRefTrackingId); + } queue = rhs.queue; + m_futureRefTrackingId = -1; + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && rhs.m_futureRefTrackingId >= 0 && g_futureRefCopiedCallback) { + m_futureRefTrackingId = g_futureRefCopiedCallback(rhs.m_futureRefTrackingId); + } } void operator=(FutureStream&& rhs) noexcept { if (rhs.queue != queue) { if (queue) queue->delFutureRef(); + if (FLOW_KNOBS->STALE_PEER_OBSERVABILITY && m_futureRefTrackingId >= 0 && g_futureRefReleasedCallback) { + g_futureRefReleasedCallback(m_futureRefTrackingId); + } queue = rhs.queue; + m_futureRefTrackingId = rhs.m_futureRefTrackingId; rhs.queue = nullptr; + rhs.m_futureRefTrackingId = -1; } } bool operator==(const FutureStream& rhs) { return rhs.queue == queue; } @@ -1326,10 +1367,14 @@ class SWIFT_SENDABLE FutureStream { return queue->error; } - explicit FutureStream(NotifiedQueue* queue) : queue(queue) {} + explicit FutureStream(NotifiedQueue* queue) : queue(queue), m_futureRefTrackingId(-1) {} + FutureStream(NotifiedQueue* queue, int64_t trackingId) : queue(queue), m_futureRefTrackingId(trackingId) {} + + int64_t getFutureRefTrackingId() const { return m_futureRefTrackingId; } private: NotifiedQueue* queue; + int64_t m_futureRefTrackingId; }; template diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 78d3fc1d508..14ff76617bb 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -248,6 +248,7 @@ if(WITH_PYTHON) add_fdb_test(TEST_FILES fast/RESTUnit.toml IGNORE) add_fdb_test(TEST_FILES fast/SelectorCorrectness.toml) add_fdb_test(TEST_FILES fast/ShardedRocksNondeterministicTest.toml) + add_fdb_test(TEST_FILES fast/StalePeerTest.toml) add_fdb_test(TEST_FILES fast/Sideband.toml) add_fdb_test(TEST_FILES fast/SidebandSingle.toml) add_fdb_test(TEST_FILES fast/SidebandWithStatus.toml) diff --git a/tests/fast/StalePeerTest.toml b/tests/fast/StalePeerTest.toml new file mode 100644 index 00000000000..c3e0d60e648 --- /dev/null +++ b/tests/fast/StalePeerTest.toml @@ -0,0 +1,95 @@ +[configuration] +buggify = false +machineCount = 10 +desiredTLogCount = 6 +generateFearless = false +config = 'triple' + +# Exclude ShardedRocksDB (enum 5): release-7.4 weights simulation storage-engine +# selection heavily toward RocksDB-family engines (PROBABILITY_FACTOR_SHARDED_ROCKSDB_ +# ENGINE_SELECTED_SIM=100, vs no such knob on 7.3), and ShardedRocksDB's slower post-kill +# shard relocation occasionally exceeds this test's fixed waitAfterKill window, which is +# unrelated to the stale-peer eviction logic under test. Same exclusion precedent as +# tests/fast/DDPipelineSaturation.toml (excluded there for a different reason, OOM). +# Plain RocksDB (enum 4) is intentionally kept in scope. +storageEngineExcludeTypes = [5] +tenantModes = ['disabled'] +encryptModes = ['disabled'] +disableTss = true + +# Force 3 coordinators so dstKillRole=coordinator always has a non-protected target. +# With sim2's default coordinator count and protectedAddresses (a majority is protected), small +# clusters can end up with all coordinators in protectedAddresses, in which case +# the test trivially skips and we get no signal. +coordinators = 3 + +[[knobs]] + +# Tune knobs to aggressively evict potentially dead storage servers from NativeAPI's LocationCache +location_cache_peer_evictor_enabled = true +location_cache_peer_evictor_failed_threshold = 0 +location_cache_peer_evictor_delay = 5.0 +location_cache_peer_evictor_scan_chunk = 2 + +# Ensure stale commit/grv proxies are also cleaned up +shrink_proxy_list_clear_cache_below_threshold = true +dbcontext_eager_proxy_update = true + +# The pass criterion is the per-role InterfaceTracker delta (leaked copies of +# the killed role's interface RequestStreams), which requires this bookkeeping, +# so it must be ON. Raw peerReferences is NOT the criterion (a client keeps +# expected connection-level refs to the killed address). +stale_peer_observability = true + +[[test]] +testTitle = 'StalePeerTest' + +# The purpose of this test is purely measuring stale peer references +# and ensuring there are none for a given (srcCheckRole, dstKillRole). +# These checks in general are run outside this test with stale fix features +# turned on e.g. location_cache_peer_evictor_enabled. +runConsistencyCheck = false +waitForQuiescenceBegin = false +waitForQuiescenceEnd = false +clearAfterTest = false + + [[test.workload]] + testName = 'StalePeerTest' + # dstKillRole: which role's process to kill (the "destination" whose stale + # peer refs we hunt). Options: 'tlog', 'ss', 'commit_proxy', 'grv_proxy', + # 'master', 'resolver', 'dd', 'rk', 'coordinator', 'log_router', + # 'cluster_controller', and the special value 'clientFacing'. + # 'clientFacing' resolves per-run (deterministically by seed) to one of the + # roles a client directly addresses: coordinator, cluster_controller, + # commit_proxy, grv_proxy, ss. Use it for bulk validation that all + # client-facing dst-role kills are stale-ref clean in a single ensemble. + dstKillRole = 'clientFacing' + # srcCheckRole: which source processes we inspect for a lingering peer ref + # to the killed dst. 'any' = every live process (server roles included). + # 'tester_client' = only customer-client processes (ProcessClass::TesterClass, + # the testers running the workload). The client-side fixes target what a + # pure customer client sees; server roles drive the client lib internally + # but are out of scope, so we check only tester clients here. + srcCheckRole = 'tester_client' + # Wait after the kill before checking. For the client contract the drain is + # fast (clientInfo rotation for proxies/CC; location-cache eviction for SS), + # so a short wait suffices. + waitAfterKill = 120.0 + + # Drive real client->dst connections so the tester_client check is not + # vacuous: without sustained reads/writes a tester client may never open a + # connection to the killed dst (especially an SS). ReadWrite has each client + # populate + read/write a keyspace, so clients hold genuine peer refs to + # commit/grv proxies and storage servers and we observe them drop the dead + # dst while keeping the live ones. + [[test.workload]] + testName = 'ReadWrite' + testDuration = 180.0 + transactionsPerSecond = 100 + nodeCount = 1000 + readsPerTransactionA = 10 + writesPerTransactionA = 1 + readsPerTransactionB = 0 + writesPerTransactionB = 0 + alpha = 0.0 + setup = true From 475371deed88a91651c046af39e90d9ea6c561ac Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Thu, 17 Sep 2026 20:43:39 -0700 Subject: [PATCH 106/170] Simplify ClientMetric reset after successful writes --- fdbserver/workloads/ClientMetric.cpp | 12 +++--------- 1 file changed, 3 insertions(+), 9 deletions(-) diff --git a/fdbserver/workloads/ClientMetric.cpp b/fdbserver/workloads/ClientMetric.cpp index 7cbbafcb65f..6cb90929a8d 100644 --- a/fdbserver/workloads/ClientMetric.cpp +++ b/fdbserver/workloads/ClientMetric.cpp @@ -145,19 +145,12 @@ struct ClientMetricWorkload : TestWorkload { Future writeRandomKeys(Database cx, int total) { int cnt = 0; Transaction tr(cx); - bool startNewTransaction = false; try { while (true) { Error err; try { co_await delay(0.001); - if (startNewTransaction) { - // Independent writes must not inherit the previous transaction's retry backoff. - tr.fullReset(); - startNewTransaction = false; - } else { - tr.reset(); - } + tr.reset(); tr.set(Key(deterministicRandom()->randomAlphaNumeric(10)), Value(Key(deterministicRandom()->randomAlphaNumeric(10)))); co_await tr.commit(); @@ -165,7 +158,8 @@ struct ClientMetricWorkload : TestWorkload { break; } ++cnt; - startNewTransaction = true; + // Independent writes must not inherit the previous transaction's retry backoff. + tr.fullReset(); } catch (Error& e) { err = e; } From e6203279f4f827e9e9e78c337634ed58a278c5a0 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Thu, 17 Sep 2026 21:34:05 -0700 Subject: [PATCH 107/170] Stop CDC proxy rebalancing when disabled --- fdbserver/clustercontroller/ClusterController.cpp | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index 906b02bed81..8be5d50799f 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -2484,7 +2484,11 @@ Future monitorCDCProxyAssignments(ClusterControllerData* self) { Future rebalanceCDCProxyAssignments(ClusterControllerData* self) { while (true) { co_await delay(std::max(1.0, SERVER_KNOBS->CDC_PROXY_REBALANCE_INTERVAL)); - if (!SERVER_KNOBS->CDC_PROXY_REBALANCE_ENABLED || !self->db.recoveryData.isValid() || + if (!SERVER_KNOBS->CDC_PROXY_REBALANCE_ENABLED) { + TraceEvent("CDCProxyRebalanceDisabled", self->id); + co_return; + } + if (!self->db.recoveryData.isValid() || self->db.serverInfo->get().recoveryState != RecoveryState::FULLY_RECOVERED || !self->db.clientInfo->get().nativeCdcEnabled) { continue; From 4ea1106a9daf336d00a166cfa3683c2f205c8f81 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Thu, 17 Sep 2026 21:47:52 -0700 Subject: [PATCH 108/170] Update rebalance test after blocked-consume helper removal --- fdbserver/workloads/NativeCdcEndToEnd.cpp | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/fdbserver/workloads/NativeCdcEndToEnd.cpp b/fdbserver/workloads/NativeCdcEndToEnd.cpp index a1e4d4e820c..99c087bb035 100644 --- a/fdbserver/workloads/NativeCdcEndToEnd.cpp +++ b/fdbserver/workloads/NativeCdcEndToEnd.cpp @@ -829,9 +829,8 @@ class NativeCdcEndToEndWorkload : public TestWorkload { } ASSERT(primed); // Leave this stream unacknowledged so its old tag data must remain readable by the new owner. - Future pending; - co_await timeoutError(startBlockedConsume(cx, firstId, streams[0].consumer, source, &pending), - operationTimeout); + Future pending = streams[0].consumer->consume(); + ASSERT(!pending.isReady()); std::vector availableProxies{ proxies[0].id(), proxies[1].id() }; ASSERT(co_await timeoutError(rebalanceNativeCdcProxyAssignments(cx, availableProxies, [] { return true; }), From d8ff6bf120ddf8ee7646f86eaa0425db0e1cfea8 Mon Sep 17 00:00:00 2001 From: Pierce Lopez Date: Thu, 17 Sep 2026 15:52:03 -0400 Subject: [PATCH 109/170] bindings/c: upgrade tests only support linux These tests assume linux and download linux binaries from github releases, in tests/TestRunner/fdb_test_runner/binary_download.py --- bindings/c/CMakeLists.txt | 18 ++++++++++-------- 1 file changed, 10 insertions(+), 8 deletions(-) diff --git a/bindings/c/CMakeLists.txt b/bindings/c/CMakeLists.txt index e5b6a8ec0fb..2969bcd90e2 100644 --- a/bindings/c/CMakeLists.txt +++ b/bindings/c/CMakeLists.txt @@ -354,15 +354,17 @@ if(NOT WIN32) ) endforeach() - add_python_venv_test(NAME fdb_c_upgrade_to_future_version - COMMAND python -m fdb_test_runner.upgrade_test - --build-dir ${CMAKE_BINARY_DIR} - --test-file ${CMAKE_SOURCE_DIR}/bindings/c/test/apitester/tests/upgrade/MixedApiWorkloadMultiThr.toml - --upgrade-path "${FDB_CURRENT_VERSION}" "${FDB_FUTURE_VERSION}" "${FDB_CURRENT_VERSION}" - --process-number 3 - ) + if(CMAKE_SYSTEM_NAME STREQUAL "Linux") + add_python_venv_test(NAME fdb_c_upgrade_to_future_version + COMMAND python -m fdb_test_runner.upgrade_test + --build-dir ${CMAKE_BINARY_DIR} + --test-file ${CMAKE_SOURCE_DIR}/bindings/c/test/apitester/tests/upgrade/MixedApiWorkloadMultiThr.toml + --upgrade-path "${FDB_CURRENT_VERSION}" "${FDB_FUTURE_VERSION}" "${FDB_CURRENT_VERSION}" + --process-number 3 + ) + endif() - if(CMAKE_SYSTEM_PROCESSOR STREQUAL "x86_64" AND NOT USE_SANITIZER) + if(CMAKE_SYSTEM_NAME STREQUAL "Linux" AND CMAKE_SYSTEM_PROCESSOR STREQUAL "x86_64" AND NOT USE_SANITIZER) add_python_venv_test(NAME fdb_c_client_config_tests COMMAND python ${CMAKE_CURRENT_SOURCE_DIR}/test/fdb_c_client_config_tests.py --build-dir ${CMAKE_BINARY_DIR} From af5e1af848203ee709150659ff3f05483a91e333 Mon Sep 17 00:00:00 2001 From: Michael Stack Date: Fri, 18 Sep 2026 11:08:08 -0700 Subject: [PATCH 110/170] BulkLoad: replace the NarrowFleet test with unit tests for the split (#14082) Deletes tests/slow/BackupS3BlobBulkLoadRestoreNarrowFleet.toml and covers the invariants it was meant to protect with a unit test instead. bulkload task by splitting it, rather than failing the whole restore. Placement fails because a destination team must be disjoint from the task's source, so a task range spanning enough of the fleet has no legal team, and re-dispatch cannot clear that -- every attempt recomputes the same source. That fix shipped with this test, whose job was to show the rescue working end to end by starving the restore of storage machines until placement had to fail. The disjointness requirement is a current limitation of the bulkload implementation rather than something inherent, to be addressed separately. Ingestion runs inside fetchKeys, and changeServerKeys only starts fetchKeys for a range the server does not already hold, so a server present in both source and destination would keep its stale copy while the others ingest. Ingesting onto a server that already owns the range would retire this whole class of unplaceable task; until then, narrowing is the only escape. The premise does not hold. Source is the union of the owners of every shard the task range spans, and the test clears its target keyspace before restoring, so the range spans one or two shards and source stays at three or four servers however wide the task is. There is no fleet size between "cannot place" and "places immediately": pinned at six machines, three Joshua seeds failed with restore_error, all C(6,3) teams built and healthy but source carrying four servers; pinned at eight, the same seed passed having logged no placement failure and no split at all. Widening the task cannot help either, since a task cannot hold more manifests than the dump produced. The condition measured on the 100M validation cluster needs a restore range of thousands of shards, which this dataset cannot produce. Coverage does not depend on it. BulkLoadingDestTeamFailure and BulkLoadingDestTeamFailureExhausted exercise the data distribution half -- retry within budget, and give-up once exhausted -- on a topology that can place a disjoint team, so the load is tested rather than stalled. The end-to-end restore has four sibling tests. So cover the invariants where they are deterministic. splitBulkLoadTask splits into deriveSplitBulkLoadTasks(), which derives the replacement tasks, and the transaction that installs them. The new TEST_CASE covers tiling, manifest partition, both declines, halving termination, and as a regression the edge-clipped case, where a task at a job-range edge holds manifests whose midpoint lies outside its own range and cutting there would hand a child key space the task never owned. --- .../datadistributor/DataDistribution.cpp | 149 ++++++++++++++-- tests/CMakeLists.txt | 1 - ...ackupS3BlobBulkLoadRestoreNarrowFleet.toml | 163 ------------------ 3 files changed, 131 insertions(+), 182 deletions(-) delete mode 100644 tests/slow/BackupS3BlobBulkLoadRestoreNarrowFleet.toml diff --git a/fdbserver/datadistributor/DataDistribution.cpp b/fdbserver/datadistributor/DataDistribution.cpp index 739a35f514f..9365b10bbfa 100644 --- a/fdbserver/datadistributor/DataDistribution.cpp +++ b/fdbserver/datadistributor/DataDistribution.cpp @@ -1111,34 +1111,31 @@ Future> triggerBulkLoadTask(Reference splitBulkLoadTask(Reference self, BulkLoadTaskState parent) { +// The children's ranges tile the parent's, so the caller must write both in a single transaction: then no +// version exists in which the parent's range is unowned, or owned by anything but tasks whose union is the +// parent -- that is the difference from erasing the task and relying on something to rebuild it, which drops +// the range's data if nothing does. +Optional> deriveSplitBulkLoadTasks(const BulkLoadTaskState& parent, UID logId) { std::vector manifests = parent.getManifests(); if (manifests.size() < 2) { // A single manifest is as narrow as a task gets, and a manifest can span an arbitrarily wide range, // so this is reachable with a range covering the whole key space. Nothing here can place it: the // cluster needs servers outside src, or the manifest needs to have been dumped more finely. - TraceEvent(SevWarnAlways, "DDBulkLoadTaskSplitDeclined", self->ddId) + TraceEvent(SevWarnAlways, "DDBulkLoadTaskSplitDeclined", logId) .detail("Reason", "Task holds a single manifest and cannot be narrowed") .detail("TaskRange", parent.getRange()) .detail("TaskID", parent.getTaskId()); - co_return false; + return {}; } std::sort(manifests.begin(), manifests.end(), [](BulkLoadManifest const& a, BulkLoadManifest const& b) { return a.getBeginKey() < b.getBeginKey(); @@ -1169,12 +1166,12 @@ Future splitBulkLoadTask(Reference self, BulkLoadTaskStat } if (insideBoundaries.empty()) { // Every manifest boundary is outside the parent's clipped range, so the range cannot be cut at one. - TraceEvent(SevWarnAlways, "DDBulkLoadTaskSplitDeclined", self->ddId) + TraceEvent(SevWarnAlways, "DDBulkLoadTaskSplitDeclined", logId) .detail("Reason", "No manifest split point lies inside the task's range") .detail("TaskRange", parent.getRange()) .detail("TaskID", parent.getTaskId()) .detail("ManifestCount", manifests.size()); - co_return false; + return {}; } int const half = insideBoundaries[insideBoundaries.size() / 2]; // The parent's range starts at or after its first manifest's begin key, so that key can never be @@ -1201,7 +1198,7 @@ Future splitBulkLoadTask(Reference self, BulkLoadTaskStat ASSERT(children[0].getRange().begin == parent.getRange().begin); ASSERT(children[0].getRange().end == children[1].getRange().begin); ASSERT(children[1].getRange().end == parent.getRange().end); - // The writes below must be issued in ascending key order, so keep the guard next to the reason. + // The caller must issue the writes in ascending key order, so keep the guard next to the reason. // krmSetRange reads oldValue at Snapshot::True on a plain Transaction, which has no read-your-writes, // so each call is blind to the previous one's mutations and only their order makes the result correct. // Each call emits clear(range); set(begin, value); set(end, oldValue). Ascending, the second call's @@ -1210,6 +1207,18 @@ Future splitBulkLoadTask(Reference self, BulkLoadTaskStat // lands last and republishes the parent over the second child's range -- precisely the state the // tiling comment above says cannot exist. ASSERT(children[0].getRange().begin < children[1].getRange().begin); + return children; +} + +// Install the derived children in place of the parent, in one transaction. Returns false without writing +// anything if the task cannot be narrowed, and also if the parent turns out to be no longer ours, which is +// not a statement about the range. +Future splitBulkLoadTask(Reference self, BulkLoadTaskState parent) { + Optional> derived = deriveSplitBulkLoadTasks(parent, self->ddId); + if (!derived.present()) { + co_return false; + } + std::vector const& children = derived.get(); Database cx = self->txnProcessor->context(); Transaction tr(cx); @@ -1230,7 +1239,7 @@ Future splitBulkLoadTask(Reference self, BulkLoadTaskStat .detail("CommitVersion", tr.getCommittedVersion()) .detail("TaskRange", parent.getRange()) .detail("TaskID", parent.getTaskId()) - .detail("ManifestCount", manifests.size()) + .detail("ManifestCount", parent.getManifests().size()) .detail("FirstRange", children[0].getRange()) .detail("FirstTaskID", children[0].getTaskId()) .detail("SecondRange", children[1].getRange()) @@ -5962,3 +5971,107 @@ TEST_CASE("/DataDistribution/Initialization/ResumeFromShard") { self->shardsAffectedByTeamFailure->check(); co_return; } + +namespace { + +// Only the key range matters to deriveSplitBulkLoadTasks(). The remaining fields are whatever satisfies +// BulkLoadManifest::isValid(), which the constructor asserts. +BulkLoadManifest splitTestManifest(KeyRef begin, KeyRef end) { + return BulkLoadManifest(BulkLoadFileSet("root", "relative", "0-manifest.txt", "0-data.sst", "", {}), + begin, + end, + /*version=*/1, + /*bytes=*/1, + /*keyCount=*/1, + BulkLoadByteSampleSetting(0, "hashlittle2", 250, 100, 0.5), + BulkLoadType::SST, + BulkLoadTransportMethod::CP); +} + +BulkLoadTaskState splitTestTask(const std::vector& manifestRanges, const KeyRange& taskRange) { + BulkLoadManifestSet set(manifestRanges.size()); + for (const auto& range : manifestRanges) { + ASSERT(set.addManifest(splitTestManifest(range.begin, range.end))); + } + return BulkLoadTaskState(deterministicRandom()->randomUniqueID(), set, taskRange); +} + +} // namespace + +TEST_CASE("/DataDistribution/BulkLoad/DeriveSplitTasks") { + // The children tile the parent and partition its manifests, and inherit its job. + { + auto parent = splitTestTask({ KeyRangeRef("a"_sr, "c"_sr), + KeyRangeRef("c"_sr, "e"_sr), + KeyRangeRef("e"_sr, "g"_sr), + KeyRangeRef("g"_sr, "i"_sr) }, + KeyRangeRef("a"_sr, "i"_sr)); + auto children = deriveSplitBulkLoadTasks(parent, UID()).get(); + ASSERT_EQ(children.size(), 2); + ASSERT(children[0].getRange().begin == parent.getRange().begin); + ASSERT(children[0].getRange().end == children[1].getRange().begin); + ASSERT(children[1].getRange().end == parent.getRange().end); + ASSERT_EQ(children[0].getManifests().size() + children[1].getManifests().size(), parent.getManifests().size()); + ASSERT(children[0].getJobId() == parent.getJobId()); + ASSERT(children[1].getJobId() == parent.getJobId()); + } + + // The cut falls on a manifest boundary, so neither child is handed a range whose data lives in the + // other's manifests. + { + auto parent = + splitTestTask({ KeyRangeRef("a"_sr, "c"_sr), KeyRangeRef("c"_sr, "e"_sr) }, KeyRangeRef("a"_sr, "e"_sr)); + auto children = deriveSplitBulkLoadTasks(parent, UID()).get(); + ASSERT(children[0].getRange() == KeyRangeRef("a"_sr, "c"_sr)); + ASSERT(children[1].getRange() == KeyRangeRef("c"_sr, "e"_sr)); + } + + // A single manifest is as narrow as a task gets. Declining is terminal for the task -- the caller marks + // it Error rather than re-dispatching it. + { + auto parent = splitTestTask({ KeyRangeRef("a"_sr, "z"_sr) }, KeyRangeRef("a"_sr, "z"_sr)); + ASSERT(!deriveSplitBulkLoadTasks(parent, UID()).present()); + } + + // REGRESSION: a task at a job-range edge holds manifests whose boundaries lie outside its clipped + // range. The cut must come from the boundaries strictly inside that range: here the manifest midpoint + // is "c", below the parent's own begin key. + { + auto parent = + splitTestTask({ KeyRangeRef("a"_sr, "c"_sr), KeyRangeRef("c"_sr, "e"_sr), KeyRangeRef("e"_sr, "g"_sr) }, + KeyRangeRef("d"_sr, "f"_sr)); + auto children = deriveSplitBulkLoadTasks(parent, UID()).get(); + ASSERT(children[0].getRange() == KeyRangeRef("d"_sr, "e"_sr)); + ASSERT(children[1].getRange() == KeyRangeRef("e"_sr, "f"_sr)); + } + + // Every boundary outside the clipped range leaves nothing to cut at, so the task declines rather than + // producing an empty child. + { + auto parent = + splitTestTask({ KeyRangeRef("a"_sr, "e"_sr), KeyRangeRef("e"_sr, "i"_sr) }, KeyRangeRef("f"_sr, "h"_sr)); + ASSERT(!deriveSplitBulkLoadTasks(parent, UID()).present()); + } + + // Halving terminates: each child holds strictly fewer manifests than its parent, so repeated splitting + // of the lower child reaches a single manifest and declines. + { + StringRef const boundaries = "abcdefghi"_sr; + std::vector ranges; + for (int i = 0; i + 1 < boundaries.size(); i++) { + ranges.push_back(KeyRangeRef(boundaries.substr(i, 1), boundaries.substr(i + 1, 1))); + } + BulkLoadTaskState task = splitTestTask(ranges, KeyRangeRef(ranges.front().begin, ranges.back().end)); + while (true) { + Optional> children = deriveSplitBulkLoadTasks(task, UID()); + if (!children.present()) { + break; + } + ASSERT_LT(children.get()[0].getManifests().size(), task.getManifests().size()); + task = children.get()[0]; + } + ASSERT_EQ(task.getManifests().size(), 1); + } + + return Void(); +} diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 14ff76617bb..17a46f04af2 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -177,7 +177,6 @@ if(WITH_PYTHON) add_fdb_test(TEST_FILES slow/BackupS3BlobBulkLoadRestore.toml) add_fdb_test(TEST_FILES slow/BackupS3BlobBulkLoadRestoreWithChaos.toml) add_fdb_test(TEST_FILES slow/BackupS3BlobBulkLoadRestoreMultiRange.toml) - add_fdb_test(TEST_FILES slow/BackupS3BlobBulkLoadRestoreNarrowFleet.toml) add_fdb_test(TEST_FILES slow/BackupS3BlobBulkLoadRestoreJobIncomplete.toml) add_fdb_test(TEST_FILES fast/BulkLoading.toml) add_fdb_test(TEST_FILES fast/BulkLoadingDestTeamFailure.toml) diff --git a/tests/slow/BackupS3BlobBulkLoadRestoreNarrowFleet.toml b/tests/slow/BackupS3BlobBulkLoadRestoreNarrowFleet.toml deleted file mode 100644 index adf1bce339e..00000000000 --- a/tests/slow/BackupS3BlobBulkLoadRestoreNarrowFleet.toml +++ /dev/null @@ -1,163 +0,0 @@ -# BulkLoad restore onto a fleet with barely room for a disjoint destination team -# -# Variant of BackupS3BlobBulkLoadRestore.toml that cuts the extra storage machines its sibling tests rely -# on down to the minimum that still permits a legal destination team, so bulk-load task placement -# genuinely fails and the restore has to recover by narrowing task ranges rather than by having spare -# capacity. See the comment on the configuration below for where that minimum comes from. -# -# Reproduces the condition measured on a 100M-key validation cluster of ~40 storage servers: every -# candidate destination team is rejected for overlapping the source, so the team selector reports -# ValidTeamSize 0 with no unhealthy or ineligible teams involved. On that cluster the recovery path ran -# end to end - GetTeamFailedToFindValidTeam, then a relocation declared stuck, then a task split - and the -# restore completed with the consistency check passing. -# -# A run that exercises the fix logs GetTeamFailedToFindValidTeam followed by BulkLoad task splits, and no -# SplitDeclined. Exact counts depend on the dataset knobs below, so they are deliberately not asserted -# here; a fleet large enough to place a disjoint team makes the test vacuous rather than failing it. -# -# The rest of this file is inherited from BackupS3BlobBulkLoadRestore.toml. -# -# BulkLoad Validation Test -# Tests that BulkLoad restore produces identical results to traditional restore -# -# This test validates BulkLoad produces the same data as traditional restore: -# - Backup creates BOTH range files AND SST files (snapshotMode=2) -# - Range files are used by traditional restore -# - SST files are used by BulkLoad restore -# - Validation: compare BulkLoad restore vs traditional restore -# 1. Restore with --add-prefix to system keyspace using TRADITIONAL (rangefile) mode -# This creates a "known good" baseline -# 2. Clear normalKeys (original data) -# 3. Restore to normalKeys using BULKLOAD mode (reads SST files) -# 4. Run audit_storage validate_restore to compare: -# - BulkLoad-restored data (in normalKeys) -# - Traditional-restored data (in system key prefix) -# 5. Clean up validation prefix data -# - This validates BulkLoad produces identical results to traditional restore -# -# Configuration aligned with working tests/slow/BulkDumpingS3.toml - -testClass = "Backup" - -[configuration] -storageEngineExcludeTypes = ["ssd-sharded-rocksdb"] # FIXME: remove after allowing bulkloading with fetchKey and shardedrocksdb -disableTss = true # TODO(BulkLoad): support TSS - -# HA (multi-region) configuration - randomly enabled for test coverage -generateFearless = false -simpleConfig = false -minimumRegions = 1 -# singleRegion not set - allows random HA configuration - -# Ensure enough storage servers for non-overlapping BulkLoad teams -extraMachineCountDC = 3 -# -# getTeamForBulkLoad rejects any candidate team sharing a *server* with the source, and source is the union -# of the owners of every shard the task range spans, so a wide enough range has no legal destination. -# Provoking that is the point of this test; escaping it means narrowing the range, which is the path under -# test. Narrowing bottoms out at one manifest per task, and one manifest can still span several destination -# shards, so the cluster must field StorageTeamSize servers on distinct machines owning none of the range. -# -# That headroom has to come from more storage *machines*, not more servers per machine. Both add servers, -# but co-tenants share the machine's simulated disk, and getTeamForBulkLoad also rejects teams whose -# servers are low on disk -- its own comment warns that random low disk space can strand this test. -# Measured at processesPerMachine 2: DiskNearCapacity 1175 vs 149, GetTeamReturnEmpty 534 vs 12, DDExiting -# 58 vs 3, and the restore froze at 35 of 45 tasks for 2200s until the no-progress timeout fired twice. -# Placement itself was fixed -- zero GetTeamFailedToFindValidTeam -- but disk starvation replaced it. So -# pin one server per machine and buy headroom with extraStorageMachineCountPerDC, whose machines are -# storage-class with a disk each. -# -# machineCount is not the lever either: the run this was written for had 22 machines and only 6 holding -# data, because assignClasses starves storage independently of fleet size. -# -# Reproduction: seed 8008 is a local regression pair. At extraStorageMachineCountPerDC 0 it fails the way -# this test used to -- 4 machines holding data, 750 placement failures, 8 SplitDeclined "single manifest", -# restore_error. At 5 the same seed still provokes the condition (100 placement failures, 2 splits) and now -# recovers, completing with the dataset verified. That pair is the check to run when changing anything here: -# the fix must keep the provocation, not remove it. -# -# Provocation is seed-dependent and uncommon -- roughly 1 seed in 8 locally -- so a clean sweep proves -# little on its own. Grep GetTeamFailedToFindValidTeam and DDBulkLoadTaskSplit to tell a passing run that -# exercised the path from one that never reached it. -processesPerMachine = 1 -extraStorageMachineCountPerDC = 5 -# The knobs below still matter for keeping the test honest: manifests are written one per shard -# (FileBackupAgent.cpp), so min_shard_bytes controls how many exist and -# manifest_count_max_per_bulkload_task how many a task groups. Left to their defaults this test degrades -# into a single task holding a single manifest over the whole key space, which is not worth asserting on. - -# Explicit simple config - single replication, single region, explicit process counts -config = "triple usable_regions=1 storage_engine=ssd-2 perpetual_storage_wiggle=0 commit_proxies=3 grv_proxies=3 resolvers=3 logs=3" - -# Disable buggify and fault injection to avoid interference with MockS3/BulkLoad -buggify = false -faultInjection = false - -# Required knobs for BulkLoad functionality (from BulkDumpingS3.toml) -[[knobs]] -manifest_count_max_per_bulkload_task = 10 -min_shard_bytes = 10000 -bulkload_sim_failure_injection = false -shard_encode_location_metadata = true -enable_read_lock_on_range = true -enable_version_vector = false -enable_version_vector_tlog_unicast = false -enable_version_vector_reply_recovery = false -min_byte_sampling_probability = 0.5 -cc_enforce_use_unfit_dd_in_sim = true -disable_audit_storage_final_replica_check_in_sim = true -max_trace_lines = 5000000 -# Allow more time for BulkDump job to complete (Linux runs 4x slower than macOS) -bulkdump_job_timeout = 1200 -bulkload_job_timeout = 1200 - -# Disable buggified delays -[[flow_knobs]] -MAX_BUGGIFIED_DELAY = 0.0 - -# S3/Blobstore settings for stability/determinism -blobstore_max_connection_life = 300 -blobstore_request_timeout_min = 300 -blobstore_request_tries = 5 -blobstore_connect_tries = 5 -blobstore_connect_timeout = 30 -http_send_size = 1024 -http_read_size = 1024 -connection_monitor_loop_time = 0.1 -connection_monitor_timeout = 1.0 -connection_monitor_idle_timeout = 60.0 -dd_team_zero_server_left_log_delay = 0 -dd_rebalance_parallelism = 1 - -[[test]] -testTitle = 'BackupS3BlobBulkLoadRestoreNarrowFleet' -useDB = true -clearAfterTest = false -simBackupAgents = 'BackupToFile' -waitForQuiescence = false -connectionFailuresDisableDuration = 1000000 -runFailureWorkloads = false -timeout = 3600 - - [[test.workload]] - testName = 'Cycle' - nodeCount = 2000 - transactionsPerSecond = 100.0 - testDuration = 30.0 - - [[test.workload]] - testName = 'BackupS3BlobCorrectness' - backupAfter = 10.0 - restoreAfter = 600.0 - abortAndRestartAfter = 0.0 - stopDifferentialAfter = 0.0 - performRestore = true - backupRangesCount = -1 - skipDirtyRestore = false - backupURL = 'blobstore://mocks3:mocksecret:mocktoken@127.0.0.1:8080/backup_container?bucket=backup_bucket®ion=us-east-1&secure_connection=0&cwpf=1&cu=1' - # BulkDump/BulkLoad integration options - snapshotMode = 2 # 2 = BOTH (creates range files AND SST files for comparison) - useRangeFileRestore = false # false = use BulkLoad for restore - # Validation: Compare BulkLoad-restored vs traditional-restored using audit_storage validate_restore - # Compares BulkLoad-restored (normalKeys) vs traditional-restored (prefix) - performValidation = true From 474b0b3de566e08c517e88ab06c9748645d3ea98 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Fri, 18 Sep 2026 11:58:59 -0700 Subject: [PATCH 111/170] Fix structural lint warnings in transport tests and stale-peer workload --- fdbrpc/FlowTransport.cpp | 4 ++-- fdbserver/workloads/StalePeerTest.cpp | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/fdbrpc/FlowTransport.cpp b/fdbrpc/FlowTransport.cpp index 69189b37c29..4b78928baef 100644 --- a/fdbrpc/FlowTransport.cpp +++ b/fdbrpc/FlowTransport.cpp @@ -2587,7 +2587,7 @@ TEST_CASE("noSim/fdbrpc/FlowTransport/PacketLimitOnSend") { TransportData transport(1, WLTOKEN_FIRST_AVAILABLE, nullptr); NetworkAddress address(IPAddress(0x7f000001), 45000, true, false); Endpoint endpoint(NetworkAddressList{ address, {} }, UID(1, 2)); - Reference peer = makeReference(&transport, address); + auto peer = makeReference(&transport, address); sendPacket(&transport, peer, SerializeSource("ok"_sr), endpoint, false); PacketBuffer* const tail = peer->unsent.getWriteBuffer(); uint32_t acceptedLength; @@ -2626,7 +2626,7 @@ TEST_CASE("noSim/fdbrpc/FlowTransport/PacketLimitOnSend") { sendPacket(&transport, peer, SerializeSource("again"_sr), endpoint, false); ASSERT_GT(tail->bytes_written, previousLength); - Reference emptyPeer = makeReference(&transport, address); + auto emptyPeer = makeReference(&transport, address); bool emptyRejected = false; try { sendPacket(&transport, emptyPeer, SerializeSource(StringRef(oversized)), endpoint, true); diff --git a/fdbserver/workloads/StalePeerTest.cpp b/fdbserver/workloads/StalePeerTest.cpp index b8bf5eb2026..8ac767d31f6 100644 --- a/fdbserver/workloads/StalePeerTest.cpp +++ b/fdbserver/workloads/StalePeerTest.cpp @@ -71,7 +71,7 @@ struct StalePeerTestWorkload : TestWorkload { // tracker -- diagnostic Delta dump is skipped). std::string trackedDstRole; - StalePeerTestWorkload(WorkloadContext const& wcx) : TestWorkload(wcx), testPassed(true) { + explicit(false) StalePeerTestWorkload(WorkloadContext const& wcx) : TestWorkload(wcx), testPassed(true) { waitAfterKill = getOption(options, "waitAfterKill"_sr, 60.0); dstKillRole = getOption(options, "dstKillRole"_sr, "clientFacing"_sr).toString(); srcCheckRole = getOption(options, "srcCheckRole"_sr, "any"_sr).toString(); From d583a4b1338671396e36d54fe390cf1becc7f2c8 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Fri, 18 Sep 2026 12:04:14 -0700 Subject: [PATCH 112/170] Make StalePeerTestWorkload construction explicit --- fdbserver/workloads/StalePeerTest.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fdbserver/workloads/StalePeerTest.cpp b/fdbserver/workloads/StalePeerTest.cpp index 8ac767d31f6..ae455504ae1 100644 --- a/fdbserver/workloads/StalePeerTest.cpp +++ b/fdbserver/workloads/StalePeerTest.cpp @@ -71,7 +71,7 @@ struct StalePeerTestWorkload : TestWorkload { // tracker -- diagnostic Delta dump is skipped). std::string trackedDstRole; - explicit(false) StalePeerTestWorkload(WorkloadContext const& wcx) : TestWorkload(wcx), testPassed(true) { + explicit StalePeerTestWorkload(WorkloadContext const& wcx) : TestWorkload(wcx), testPassed(true) { waitAfterKill = getOption(options, "waitAfterKill"_sr, 60.0); dstKillRole = getOption(options, "dstKillRole"_sr, "clientFacing"_sr).toString(); srcCheckRole = getOption(options, "srcCheckRole"_sr, "any"_sr).toString(); From 9da06b10c2869aa3bc7d8cdbf12a98bc6f25fd0b Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Fri, 18 Sep 2026 12:36:43 -0700 Subject: [PATCH 113/170] Retry native CDC server validation after proxy failures --- fdbserver/workloads/NativeCdcEndToEnd.cpp | 68 +++++++++++++++-------- 1 file changed, 45 insertions(+), 23 deletions(-) diff --git a/fdbserver/workloads/NativeCdcEndToEnd.cpp b/fdbserver/workloads/NativeCdcEndToEnd.cpp index 0ec30db4270..8ada52bcf78 100644 --- a/fdbserver/workloads/NativeCdcEndToEnd.cpp +++ b/fdbserver/workloads/NativeCdcEndToEnd.cpp @@ -1238,7 +1238,9 @@ class NativeCdcEndToEndWorkload : public TestWorkload { error = e; } ASSERT(error.present()); - ASSERT_EQ(error.get().code(), error_code_client_invalid_operation); + if (error.get().code() != error_code_client_invalid_operation) { + throw error.get(); + } } Future validateConsumeLeaseAndExclusivity(Database cx, CDCStreamId streamId, CDCProxyInterface* proxy) { @@ -1287,29 +1289,49 @@ class NativeCdcEndToEndWorkload : public TestWorkload { co_await waitForNoActiveConsumes(cx, streamId, proxy); CDCCursor currentCursor = idleConsumer->position(); - // Send both requests without yielding. The first request marks the stream active before its metadata read, so - // the second request deterministically exercises server-side exclusivity even while versions advance. - co_await getCurrentProxyStatus(cx, streamId, proxy); - Future> first = proxy->consume.tryGetReply(CDCConsumeRequest(currentCursor)); - co_await expectConcurrentConsumeRejected(*proxy, currentCursor); - first.cancel(); - co_await waitForNoActiveConsumes(cx, streamId, proxy); - - // The first request may finish before the retry reaches the proxy. A pending request is superseded, while - // an already-completed request retains its reply; either ordering must allow the same consumer to retry. - const UID consumerId = deterministicRandom()->randomUniqueID(); - Future> original = - proxy->consume.tryGetReply(CDCConsumeRequest(currentCursor, consumerId)); - Future> retry = - proxy->consume.tryGetReply(CDCConsumeRequest(currentCursor, consumerId)); - const ErrorOr firstReply = co_await timeoutError(original, operationTimeout); - if (firstReply.isError()) { - ASSERT_EQ(firstReply.getError().code(), error_code_request_maybe_delivered); - } else { - ASSERT_GE(firstReply.get().lastConsumedVersion, currentCursor.lastConsumedVersion); + const double deadline = now() + operationTimeout; + while (true) { + ASSERT_LT(now(), deadline); + try { + // Send both requests without yielding. The first request marks the stream active before its metadata + // read, so the second request deterministically exercises server-side exclusivity even while versions + // advance. + co_await getCurrentProxyStatus(cx, streamId, proxy); + Future> first = proxy->consume.tryGetReply(CDCConsumeRequest(currentCursor)); + co_await timeoutError(expectConcurrentConsumeRejected(*proxy, currentCursor), operationTimeout); + first.cancel(); + co_await waitForNoActiveConsumes(cx, streamId, proxy); + + // The first request may finish before the retry reaches the proxy. A pending request is superseded, + // while an already-completed request retains its reply; either ordering must allow the same consumer to + // retry. + const UID consumerId = deterministicRandom()->randomUniqueID(); + Future> original = + proxy->consume.tryGetReply(CDCConsumeRequest(currentCursor, consumerId)); + Future> retry = + proxy->consume.tryGetReply(CDCConsumeRequest(currentCursor, consumerId)); + const ErrorOr firstReply = co_await timeoutError(original, operationTimeout); + if (firstReply.isError()) { + if (firstReply.getError().code() != error_code_request_maybe_delivered) { + throw firstReply.getError(); + } + } else { + ASSERT_GE(firstReply.get().lastConsumedVersion, currentCursor.lastConsumedVersion); + } + co_await timeoutError(throwErrorOr(retry), operationTimeout); + co_await waitForNoActiveConsumes(cx, streamId, proxy); + co_return; + } catch (Error& e) { + if (e.code() != error_code_wrong_shard_server && e.code() != error_code_broken_promise && + e.code() != error_code_connection_failed && e.code() != error_code_request_maybe_delivered) { + throw; + } + // A status reply cannot prevent a proxy replacement or disconnect before the consume replies. + // Retry both requests so transport failure cannot count as evidence of exclusivity. + CODE_PROBE(true, "Native CDC server consume validation retries after proxy request failure"); + } + co_await waitForNoActiveConsumes(cx, streamId, proxy); } - co_await timeoutError(throwErrorOr(retry), operationTimeout); - co_await waitForNoActiveConsumes(cx, streamId, proxy); } Future requestPopsUntilStopped(Database cx, Reference> stopped) { From da7037d990bc1b07dcf6e3b7e27b780cdec3846a Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Fri, 18 Sep 2026 13:32:51 -0700 Subject: [PATCH 114/170] Enable five clang-tidy correctness checks --- .clang-tidy | 7 + .github/workflows/tidy.yml | 14 +- bindings/c/test/mako/mako.cpp | 67 +++--- bindings/flow/tester/Tester.cpp | 22 +- bindings/java/JavaWorkload.cpp | 3 +- documentation/sphinx/source/clang-tidy.rst | 28 ++- fdbbackup/FileConverter.cpp | 15 +- fdbbackup/FileDecoder.cpp | 7 +- fdbbackup/backup.cpp | 64 +++--- fdbcli/AdvanceVersionCommand.cpp | 7 +- fdbcli/ConsistencyCheckCommand.cpp | 2 + fdbcli/ConsistencyScanCommand.cpp | 2 + fdbcli/FileConfigureCommand.cpp | 2 + fdbcli/ForceRecoveryWithDataLossCommand.cpp | 2 + fdbcli/HotRangeCommand.cpp | 6 +- fdbcli/IdempotencyIdsCommand.cpp | 2 + fdbcli/LockCommand.cpp | 2 + fdbcli/MaintenanceCommand.cpp | 9 +- fdbcli/SnapshotCommand.cpp | 2 + fdbcli/SuspendCommand.cpp | 19 +- fdbcli/VersionEpochCommand.cpp | 6 +- fdbcli/fdbcli.cpp | 8 +- fdbcli/include/fdbcli/fdbcli.h | 2 +- fdbclient/BackupAgentBase.cpp | 38 +++- fdbclient/BackupContainerFileSystem.cpp | 195 ++++++++++++------ fdbclient/BackupContainerLocalDirectory.cpp | 3 +- fdbclient/DatabaseConfiguration.cpp | 14 +- fdbclient/ManagementAPI.cpp | 16 +- fdbclient/S3BlobStore.cpp | 16 +- fdbclient/include/fdbclient/FDBTypes.h | 3 +- fdbclient/include/fdbclient/IBlobStore.h | 2 +- .../include/fdbclient/RandomKeyValueUtils.h | 8 +- fdbmonitor/fdbmonitor.h | 3 + fdbmonitor/fdbmonitor_lib.cpp | 39 +++- fdbmonitor/fdbmonitor_tests.cpp | 35 ++++ fdbrpc/HTTP.cpp | 54 ++++- fdbrpc/JsonWebKeySet.cpp | 10 +- fdbrpc/ReplicationUtils.cpp | 30 +-- fdbrpc/SimExternalConnection.cpp | 6 + fdbserver/SimulatedCluster.cpp | 52 +++-- .../RangePartitionedBackupWorker.cpp | 6 + fdbserver/cdcproxy/CDCProxy.cpp | 22 +- .../clustercontroller/ClusterController.cpp | 74 +++---- .../ClusterHealthIFactor.cpp | 4 +- .../ClusterHealthMonitor.cpp | 4 + .../ClusterHealthMonitorTesting.cpp | 10 + fdbserver/clustercontroller/Status.cpp | 47 ++++- fdbserver/commitproxy/CommitProxyServer.cpp | 3 +- fdbserver/core/MoveKeys.cpp | 86 +++++--- fdbserver/core/QuietDatabase.cpp | 51 ++++- fdbserver/core/WorkloadKeys.cpp | 4 +- fdbserver/datadistributor/DDTxnProcessor.cpp | 2 + .../datadistributor/DataDistribution.cpp | 12 +- fdbserver/fdbserver.cpp | 32 ++- fdbserver/kvstore/VersionedBTree.cpp | 2 + fdbserver/networktest.cpp | 7 +- .../tester/CustomShardConfigWorkload.cpp | 7 +- fdbserver/tester/DatabaseMaintenance.cpp | 6 +- fdbserver/tester/TestSpecParser.cpp | 48 ++++- fdbserver/tester/TesterServer.cpp | 30 +-- fdbserver/tester/WorkloadUtils.cpp | 85 ++++++-- fdbserver/tester/test.cpp | 68 +++--- fdbserver/worker/worker.cpp | 29 ++- fdbserver/workloads/BackgroundSelectors.cpp | 4 +- fdbserver/workloads/BackupToDBAbort.cpp | 4 +- fdbserver/workloads/BulkDumping.cpp | 6 +- fdbserver/workloads/BulkLoading.cpp | 4 +- fdbserver/workloads/CommitBugCheck.cpp | 3 +- fdbserver/workloads/ConfigureDatabase.cpp | 19 +- fdbserver/workloads/ConflictRange.cpp | 4 +- fdbserver/workloads/CpuProfiler.cpp | 8 +- fdbserver/workloads/DDBalance.cpp | 4 +- fdbserver/workloads/DDMetrics.cpp | 3 +- fdbserver/workloads/DDMetricsExclude.cpp | 4 +- .../workloads/DataDistributionMetrics.cpp | 4 +- .../workloads/DifferentClustersSameRV.cpp | 11 +- fdbserver/workloads/FileSystem.cpp | 11 +- fdbserver/workloads/HTTPKeyValueStore.cpp | 10 +- fdbserver/workloads/HealthMetricsApi.cpp | 10 +- .../HighContentionPrefixAllocatorWorkload.cpp | 4 +- fdbserver/workloads/Inventory.cpp | 5 +- fdbserver/workloads/MemoryLifetime.cpp | 8 +- fdbserver/workloads/MetricLogging.cpp | 4 +- fdbserver/workloads/ProtocolVersion.cpp | 4 +- fdbserver/workloads/QueuePush.cpp | 14 +- fdbserver/workloads/RandomMoveKeys.cpp | 4 +- fdbserver/workloads/RandomRangeLock.cpp | 4 +- fdbserver/workloads/RangeLock.cpp | 8 +- fdbserver/workloads/ReadAfterWrite.cpp | 4 +- fdbserver/workloads/ReadHotDetection.cpp | 4 +- fdbserver/workloads/ReadWrite.cpp | 2 + fdbserver/workloads/S3ClientWorkload.cpp | 4 +- fdbserver/workloads/SaveAndKill.cpp | 10 +- fdbserver/workloads/SkewedReadWrite.cpp | 4 +- fdbserver/workloads/SnapTest.cpp | 9 +- .../workloads/SpecialKeySpaceCorrectness.cpp | 4 +- .../workloads/SpecialKeySpaceRobustness.cpp | 4 +- fdbserver/workloads/Storefront.cpp | 7 +- fdbserver/workloads/StreamingRangeRead.cpp | 4 +- fdbserver/workloads/TaskBucketCorrectness.cpp | 15 +- fdbserver/workloads/ThreadSafety.cpp | 4 +- fdbserver/workloads/Throttling.cpp | 4 +- fdbserver/workloads/TimeKeeperCorrectness.cpp | 8 +- fdbserver/workloads/TransactionCost.cpp | 8 + fdbserver/workloads/UnitTests.cpp | 2 + fdbserver/workloads/WatchAndWait.cpp | 4 +- fdbserver/workloads/Watches.cpp | 4 +- fdbserver/workloads/WorkerErrors.cpp | 4 +- fdbserver/workloads/WriteBandwidth.cpp | 8 +- fdbserver/workloads/pubsub.cpp | 5 +- flow/CoroTests.cpp | 16 ++ flow/IThreadPoolTest.cpp | 2 +- flow/IndexedSet.cpp | 4 +- flow/Net2.cpp | 30 ++- flow/Platform.cpp | 16 +- flow/Profiler.cpp | 3 +- flow/Trace.cpp | 141 +++++++++---- flow/UnitTest.cpp | 138 ++++++++++++- flow/UnitTestRunner.cpp | 83 +++++++- flow/bench/BenchAsyncResult.cpp | 4 + flow/flow.cpp | 44 ++-- flow/include/flow/IDispatched.h | 6 +- flow/include/flow/ParseNumber.h | 92 +++++++++ flow/include/flow/UnitTest.h | 18 +- flow/network.cpp | 35 +++- 125 files changed, 1765 insertions(+), 614 deletions(-) create mode 100644 flow/include/flow/ParseNumber.h diff --git a/.clang-tidy b/.clang-tidy index a9dcfba23ab..b4b3693f055 100644 --- a/.clang-tidy +++ b/.clang-tidy @@ -15,6 +15,8 @@ Checks: > bugprone-macro-repeated-side-effects, bugprone-misplaced-widening-cast, bugprone-move-forwarding-reference, + bugprone-nondeterministic-pointer-iteration-order, + bugprone-posix-return, bugprone-redundant-branch-condition, bugprone-return-const-ref-from-parameter, bugprone-shared-ptr-array-mismatch, @@ -34,9 +36,12 @@ Checks: > bugprone-undefined-memory-manipulation, bugprone-unhandled-self-assignment, bugprone-unique-ptr-array-mismatch, + bugprone-unused-return-value, bugprone-use-after-move, bugprone-virtual-near-miss, + cert-err34-c, cppcoreguidelines-avoid-capturing-lambda-coroutines, + cppcoreguidelines-avoid-reference-coroutine-parameters, misc-coroutine-hostile-raii, misc-redundant-expression, modernize-use-auto, @@ -58,6 +63,8 @@ Checks: > CheckOptions: - key: bugprone-dangling-handle.HandleClasses value: 'std::basic_string_view;std::experimental::basic_string_view;std::span;StringRef' + - key: bugprone-unused-return-value.CheckedFunctions + value: '^::std::async$;^::std::launder$;^::std::remove$;^::std::remove_if$;^::std::unique$;^::std::unique_ptr::release$;^::std::basic_string::empty$;^::std::vector::empty$;^::std::back_inserter$;^::std::distance$;^::std::find$;^::std::find_if$;^::std::inserter$;^::std::lower_bound$;^::std::make_pair$;^::std::map::count$;^::std::map::find$;^::std::map::lower_bound$;^::std::multimap::equal_range$;^::std::multimap::upper_bound$;^::std::set::count$;^::std::set::find$;^::std::setfill$;^::std::setprecision$;^::std::setw$;^::std::upper_bound$;^::std::vector::at$;^::bsearch$;^::ferror$;^::feof$;^::isalnum$;^::isalpha$;^::isblank$;^::iscntrl$;^::isdigit$;^::isgraph$;^::islower$;^::isprint$;^::ispunct$;^::isspace$;^::isupper$;^::iswalnum$;^::iswprint$;^::iswspace$;^::isxdigit$;^::memchr$;^::memcmp$;^::strcmp$;^::strcoll$;^::strncmp$;^::strpbrk$;^::strrchr$;^::strspn$;^::strstr$;^::wcscmp$;^::access$;^::bind$;^::connect$;^::difftime$;^::dlsym$;^::fnmatch$;^::getaddrinfo$;^::getopt$;^::htonl$;^::htons$;^::iconv_open$;^::inet_addr$;^::isascii$;^::isatty$;^::mmap$;^::newlocale$;^::openat$;^::pathconf$;^::pthread_equal$;^::pthread_getspecific$;^::pthread_mutex_trylock$;^::readdir$;^::readlink$;^::recvmsg$;^::regexec$;^::scandir$;^::semget$;^::setjmp$;^::shm_open$;^::shmget$;^::sigismember$;^::strcasecmp$;^::strsignal$;^::ttyname$;^::pthread_create$' - key: misc-coroutine-hostile-raii.RAIITypesList value: 'std::lock_guard;std::scoped_lock;MutexHolder;ThreadSpinLockHolder' - key: modernize-use-auto.MinTypeNameLength diff --git a/.github/workflows/tidy.yml b/.github/workflows/tidy.yml index eccacfd0a8b..42befff6df5 100644 --- a/.github/workflows/tidy.yml +++ b/.github/workflows/tidy.yml @@ -55,6 +55,18 @@ jobs: ) ninja -v $PB_HEADERS + - name: Install clang-tidy with pointer-order checks + run: | + # The tool bundled in the build image is older than the pointer-order check. + # Keep its compiler and libc++ for the build; install only the packaged tidy tool. + dnf --disablerepo='*' --enablerepo=baseos,appstream \ + --setopt=baseos.mirrorlist= --setopt=appstream.mirrorlist= \ + --setopt=baseos.baseurl='https://dl.rockylinux.org/pub/rocky/9/BaseOS/$basearch/os/' \ + --setopt=appstream.baseurl='https://dl.rockylinux.org/pub/rocky/9/AppStream/$basearch/os/' \ + --setopt=timeout=15 --setopt=retries=1 install -y clang-tools-extra + /usr/bin/clang-tidy --version + /usr/bin/clang-tidy --list-checks | grep -q bugprone-nondeterministic-pointer-iteration-order + - name: clang-tidy # clang-tidy is roughly as slow as compiling per-file, so check only touched files shell: bash @@ -109,7 +121,7 @@ jobs: if [[ $FILE == fdbserver/*/*RocksDB*.cpp ]]; then ninja -C build_output fdbserver/rocksdb-prefix/src/rocksdb-stamp/rocksdb-download fi - if ! clang-tidy -p build_output --warnings-as-errors='*' "$FILE"; then + if ! /usr/bin/clang-tidy -p build_output --warnings-as-errors='*' "$FILE"; then PASSED=false fi echo diff --git a/bindings/c/test/mako/mako.cpp b/bindings/c/test/mako/mako.cpp index 0d8f50b0914..877f1c652f2 100644 --- a/bindings/c/test/mako/mako.cpp +++ b/bindings/c/test/mako/mako.cpp @@ -21,6 +21,7 @@ #include #include #include +#include #include #include #include @@ -28,6 +29,7 @@ #include #include #include +#include #include #include #include @@ -1154,6 +1156,25 @@ void usage() { "Maximum estimated GRV proxy queue delay in milliseconds. Set as transaction option in run mode."); } +// Numeric options retain their historical prefix parsing and zero default for invalid input. +static int parseIntegerArgument(const char* text) { + char* end = nullptr; + errno = 0; + const long long value = std::strtoll(text, &end, 10); + if (end == text || errno == ERANGE || value < std::numeric_limits::min() || + value > std::numeric_limits::max()) { + return 0; + } + return static_cast(value); +} + +static double parseDoubleArgument(const char* text) { + char* end = nullptr; + errno = 0; + const double value = std::strtod(text, &end); + return end == text || errno == ERANGE ? 0 : value; +} + /* parse benchmark parameters */ int parseArguments(int argc, char* argv[], Arguments& args) { int rc; @@ -1245,7 +1266,7 @@ int parseArguments(int argc, char* argv[], Arguments& args) { usage(); return -1; case 'a': - args.api_version = atoi(optarg); + args.api_version = parseIntegerArgument(optarg); break; case 'c': { const char delim[] = ","; @@ -1257,26 +1278,26 @@ int parseArguments(int argc, char* argv[], Arguments& args) { break; } case 'd': - args.num_databases = atoi(optarg); + args.num_databases = parseIntegerArgument(optarg); break; case 'p': - args.num_processes = atoi(optarg); + args.num_processes = parseIntegerArgument(optarg); break; case 't': - args.num_threads = atoi(optarg); + args.num_threads = parseIntegerArgument(optarg); break; case 'r': - args.rows = atoi(optarg); + args.rows = parseIntegerArgument(optarg); args.row_digits = digits(args.rows); break; case 'l': - args.load_factor = atof(optarg); + args.load_factor = parseDoubleArgument(optarg); break; case 's': - args.seconds = atoi(optarg); + args.seconds = parseIntegerArgument(optarg); break; case 'i': - args.iteration = atoi(optarg); + args.iteration = parseIntegerArgument(optarg); break; case 'x': rc = parseTransaction(args, optarg); @@ -1284,7 +1305,7 @@ int parseArguments(int argc, char* argv[], Arguments& args) { return -1; break; case 'v': - args.verbose = atoi(optarg); + args.verbose = parseIntegerArgument(optarg); break; case 'z': args.zipf = 1; @@ -1311,23 +1332,23 @@ int parseArguments(int argc, char* argv[], Arguments& args) { } break; case ARG_ASYNC: - args.async_xacts = atoi(optarg); + args.async_xacts = parseIntegerArgument(optarg); break; case ARG_KEYLEN: - args.key_length = atoi(optarg); + args.key_length = parseIntegerArgument(optarg); break; case ARG_VALLEN: - args.value_length = atoi(optarg); + args.value_length = parseIntegerArgument(optarg); break; case ARG_TPS: case ARG_TPSMAX: - args.tpsmax = atoi(optarg); + args.tpsmax = parseIntegerArgument(optarg); break; case ARG_TPSMIN: - args.tpsmin = atoi(optarg); + args.tpsmin = parseIntegerArgument(optarg); break; case ARG_TPSINTERVAL: - args.tpsinterval = atoi(optarg); + args.tpsinterval = parseIntegerArgument(optarg); break; case ARG_TPSCHANGE: if (strcmp(optarg, "sin") == 0) @@ -1342,7 +1363,7 @@ int parseArguments(int argc, char* argv[], Arguments& args) { } break; case ARG_SAMPLING: - args.sampling = atoi(optarg); + args.sampling = parseIntegerArgument(optarg); break; case ARG_VERSION: logr.error("Version: {}", FDB_API_VERSION); @@ -1399,11 +1420,11 @@ int parseArguments(int argc, char* argv[], Arguments& args) { } break; case ARG_TXNTRACE: - args.txntrace = atoi(optarg); + args.txntrace = parseIntegerArgument(optarg); break; case ARG_TXNTAGGING: - args.txntagging = atoi(optarg); + args.txntagging = parseIntegerArgument(optarg); if (args.txntagging > 1000) { args.txntagging = 1000; } @@ -1416,7 +1437,7 @@ int parseArguments(int argc, char* argv[], Arguments& args) { memcpy(args.txntagging_prefix, optarg, strlen(optarg)); break; case ARG_CLIENT_THREADS_PER_VERSION: - args.client_threads_per_version = atoi(optarg); + args.client_threads_per_version = parseIntegerArgument(optarg); break; case ARG_DISABLE_CLIENT_BYPASS: args.disable_client_bypass = true; @@ -1477,16 +1498,16 @@ int parseArguments(int argc, char* argv[], Arguments& args) { args.private_key_pem = oss.str(); } break; case ARG_TRANSACTION_TIMEOUT_TX: - args.transaction_timeout_tx = atoi(optarg); + args.transaction_timeout_tx = parseIntegerArgument(optarg); break; case ARG_TRANSACTION_TIMEOUT_DB: - args.transaction_timeout_db = atoi(optarg); + args.transaction_timeout_db = parseIntegerArgument(optarg); break; case ARG_MAX_GRV_QUEUE_DELAY: - args.max_grv_queue_delay_ms = atoi(optarg); + args.max_grv_queue_delay_ms = parseIntegerArgument(optarg); break; case ARG_WARMUP_SECONDS: - args.warmup_seconds = atoi(optarg); + args.warmup_seconds = parseIntegerArgument(optarg); break; } } diff --git a/bindings/flow/tester/Tester.cpp b/bindings/flow/tester/Tester.cpp index 331f4f3fc09..2c86f783d8d 100644 --- a/bindings/flow/tester/Tester.cpp +++ b/bindings/flow/tester/Tester.cpp @@ -28,6 +28,7 @@ #include "bindings/flow/FDBLoanerTypes.h" #include "fdbrpc/fdbrpc.h" #include "flow/DeterministicRandom.h" +#include "flow/ParseNumber.h" #include "flow/TLSConfig.h" // Otherwise we have to type setupNetwork(), FDB::open(), etc. @@ -43,9 +44,7 @@ std::map, Reference> trMap; const int ITERATION_PROGRESSION[] = { 256, 1000, 4096, 6144, 9216, 13824, 20736, 31104, 46656, 69984, 80000 }; const int MAX_ITERATION = sizeof(ITERATION_PROGRESSION) / sizeof(int); -static Future runTest(Reference const& data, - Reference const& db, - StringRef const& prefix); +static Future runTest(Reference data, Reference db, Standalone prefix); THREAD_FUNC networkThread(void* api) { // This is the fdb_flow network we're running on a thread @@ -1516,7 +1515,7 @@ struct AtomicOPFunc : InstructionFunc { Standalone s3 = co_await items[2].value; Standalone value = Tuple::unpack(s3).getString(0); - ASSERT(optionInfo.find(op.toString()) != optionInfo.end()); + ASSERT(optionInfo.contains(op.toString())); FDBMutationType atomicOp = optionInfo[op.toString()]; @@ -1569,7 +1568,7 @@ struct UnitTestsFunc : InstructionFunc { const uint64_t locationCacheSize = 100001; const uint64_t maxWatches = 10001; - const uint64_t timeout = 60 * 1000; + const uint64_t timeout = 60ULL * 1000; const uint64_t noTimeout = 0; const uint64_t retryLimit = 50; const uint64_t noRetryLimit = -1; @@ -1717,9 +1716,7 @@ static Future doInstructions(Reference data) { // printf("Total num instructions:%d\n", data->instructions.size()); } -static Future runTest(Reference const& data, - Reference const& db, - StringRef const& prefix) { +static Future runTest(Reference data, Reference db, Standalone prefix) { ASSERT(data); try { data->db = db; @@ -1860,15 +1857,18 @@ int main(int argc, char** argv) { flushAndExit(FDB_EXIT_SUCCESS);*/ } StringRef prefix((const uint8_t*)argv[1], strlen(argv[1])); - int apiVersion; - sscanf(argv[2], "%d", &apiVersion); + auto apiVersion = parseNumberPrefix(StringRef(static_cast(argv[2]))); + if (!apiVersion.present()) { + fprintf(stderr, "Invalid API version: %s\n", argv[2]); + return 1; + } std::string clusterFilename; if (argc > 3) { clusterFilename = std::string(argv[3]); } // start test - startTest(Uncancellable(), clusterFilename, prefix, apiVersion); + startTest(Uncancellable(), clusterFilename, prefix, apiVersion.get()); // Run the network until someone tells us to stop g_network->run(); diff --git a/bindings/java/JavaWorkload.cpp b/bindings/java/JavaWorkload.cpp index 332bd6d55cc..8d8f61ef8fe 100644 --- a/bindings/java/JavaWorkload.cpp +++ b/bindings/java/JavaWorkload.cpp @@ -423,7 +423,8 @@ struct JVM { auto clazz = getClass("com/apple/foundationdb/testing/Promise"); auto res = env->NewObject(clazz, getMethod(clazz, "", "(J)V"), reinterpret_cast(p.get())); checkException(); - p.release(); + // The Java promise owns the native state until JavaPromise::send deletes it. + p.release(); // NOLINT(bugprone-unused-return-value) return res; } diff --git a/documentation/sphinx/source/clang-tidy.rst b/documentation/sphinx/source/clang-tidy.rst index 66b5963836c..5dc80dd33f1 100644 --- a/documentation/sphinx/source/clang-tidy.rst +++ b/documentation/sphinx/source/clang-tidy.rst @@ -10,12 +10,13 @@ This guide explains how to run ``clang-tidy`` locally so you can fix issues befo What clang-tidy checks ====================== -FoundationDB configures 54 named checks in the ``.clang-tidy`` file at the repository root. The +FoundationDB configures 59 named checks in the ``.clang-tidy`` file at the repository root. The active set depends on the clang-tidy version and can be inspected with ``clang-tidy --list-checks``. The intent is to enable more as we go forward. Here are some example rules: -* **35 Bugprone rules** -- catch potential runtime errors, including unsafe self-assignment, forwarding constructors that hide copy or move constructors, narrow accumulation initializers, mismatched argument comments, obvious infinite loops, chained comparisons, swapped arguments, integer division in floating-point calculations, missed base-class copy construction, repeated macro argument evaluation, near-miss virtual overrides, dangling returned references, and incorrect erase/remove calls -* **1 C++ Core Guidelines rule** -- catch unsafe captures in coroutine lambdas (``cppcoreguidelines-avoid-capturing-lambda-coroutines``) +* **38 Bugprone rules** -- catch potential runtime errors, including unsafe self-assignment, forwarding constructors that hide copy or move constructors, narrow accumulation initializers, mismatched argument comments, obvious infinite loops, chained comparisons, swapped arguments, integer division in floating-point calculations, missed base-class copy construction, repeated macro argument evaluation, near-miss virtual overrides, dangling returned references, incorrect erase/remove calls, ignored return values, incorrect POSIX error checks, and pointer-dependent iteration order +* **1 CERT rule** -- identify numeric conversion APIs that cannot report invalid input (``cert-err34-c``) +* **2 C++ Core Guidelines rules** -- catch unsafe captures in coroutine lambdas and borrowed coroutine parameters (``cppcoreguidelines-avoid-capturing-lambda-coroutines`` and ``cppcoreguidelines-avoid-reference-coroutine-parameters``) * **2 Misc rules** -- catch redundant expressions and RAII objects held across coroutine suspension points * **4 Modernize rules** -- encourage modern C++ practices (e.g., ``modernize-use-auto``, ``modernize-use-override``) * **5 Performance rules** -- avoid unnecessary copies, hidden range-loop conversions, repeated vector growth in simple loops, pointless moves, and move constructors that copy movable members (``performance-for-range-copy``, ``performance-implicit-conversion-in-loop``, ``performance-inefficient-vector-operation``, ``performance-move-const-arg``, ``performance-move-constructor-init``) @@ -31,6 +32,27 @@ with suspicious fields. Reference-counted assignments that acquire the incoming reference before releasing the old one use documented, check-specific ``NOLINTNEXTLINE`` annotations where the checker cannot recognize their safety. +``bugprone-unused-return-value`` retains the standard checked-function list and +also checks ``pthread_create``. It does not diagnose every discarded Flow future: +intentional uncancellable helpers need different treatment from cancellable work. + +``cppcoreguidelines-avoid-reference-coroutine-parameters`` encourages owning +arguments in coroutine frames. A synchronous forwarding wrapper can preserve an +interface that accepts references while its coroutine implementation takes values. +Copying a ``StringRef`` or ``KeyRef`` still does not retain the referenced bytes. + +``bugprone-nondeterministic-pointer-iteration-order`` helps protect simulation +reproducibility. It checks some pointer-keyed unordered-container iterations and +pointer sorting; it is not a complete determinism analysis. This check requires +LLVM 20 or newer. CI installs the packaged ``clang-tidy`` separately from the +build image's compiler so that the warning is active without changing the compiler +or C++ standard library. + +``cert-err34-c`` is the compatible name for the check called +``bugprone-unchecked-string-to-number-conversion`` in newer LLVM versions. +Replacing ``atoi`` with ``strtol`` alone is insufficient: validate the conversion, +range, and the caller's policy for trailing input. + Basic examples of ``clang-tidy`` style and performance improvement changes: .. code-block:: cpp diff --git a/fdbbackup/FileConverter.cpp b/fdbbackup/FileConverter.cpp index 8a710a7cff6..c5d9db4c1fe 100644 --- a/fdbbackup/FileConverter.cpp +++ b/fdbbackup/FileConverter.cpp @@ -30,6 +30,7 @@ #include "fdbclient/BackupContainer.h" #include "fdbclient/MutationList.h" #include "flow/flow.h" +#include "flow/ParseNumber.h" #include "flow/serialize.h" #include "fdbclient/BuildFlags.h" @@ -512,21 +513,27 @@ int parseCommandLine(ConvertParams* param, CSimpleOpt* args) { printConvertUsage(); return FDB_EXIT_ERROR; - case OPT_BEGIN_VERSION: - if (!sscanf(arg, "%" SCNd64, ¶m->begin)) { + case OPT_BEGIN_VERSION: { + auto version = parseNumberPrefix(StringRef(arg)); + if (!version.present()) { std::cerr << "ERROR: could not parse begin version " << arg << "\n"; printConvertUsage(); return FDB_EXIT_ERROR; } + param->begin = version.get(); break; + } - case OPT_END_VERSION: - if (!sscanf(arg, "%" SCNd64, ¶m->end)) { + case OPT_END_VERSION: { + auto version = parseNumberPrefix(StringRef(arg)); + if (!version.present()) { std::cerr << "ERROR: could not parse end version " << arg << "\n"; printConvertUsage(); return FDB_EXIT_ERROR; } + param->end = version.get(); break; + } case OPT_CONTAINER: param->container_url = args->OptionArg(); diff --git a/fdbbackup/FileDecoder.cpp b/fdbbackup/FileDecoder.cpp index 0fa7f94cf20..ef0cf034174 100644 --- a/fdbbackup/FileDecoder.cpp +++ b/fdbbackup/FileDecoder.cpp @@ -52,6 +52,7 @@ #include "flow/Platform.h" #include "flow/Trace.h" #include "flow/flow.h" +#include "flow/ParseNumber.h" #include "flow/serialize.h" #define SevDecodeInfo SevVerbose @@ -346,11 +347,13 @@ int parseDecodeCommandLine(Reference param, CSimpleOpt* args) { break; case OPT_BEGIN_VERSION_FILTER: - param->beginVersionFilter = std::atoll(args->OptionArg()); + param->beginVersionFilter = + parseNumberPrefix(StringRef(static_cast(args->OptionArg()))).orDefault(0); break; case OPT_END_VERSION_FILTER: - param->endVersionFilter = std::atoll(args->OptionArg()); + param->endVersionFilter = + parseNumberPrefix(StringRef(static_cast(args->OptionArg()))).orDefault(0); break; case OPT_CRASHONERROR: diff --git a/fdbbackup/backup.cpp b/fdbbackup/backup.cpp index 3e08d531fc9..0e6e05c57cf 100644 --- a/fdbbackup/backup.cpp +++ b/fdbbackup/backup.cpp @@ -55,6 +55,7 @@ #include "fdbclient/ManagementAPI.h" #include "flow/Platform.h" +#include "flow/ParseNumber.h" #include #include @@ -3037,20 +3038,24 @@ Version parseVersion(const char* str) { StringRef s((const uint8_t*)str, strlen(str)); if (s.endsWith("days"_sr) || s.endsWith("d"_sr)) { - float days; - if (sscanf(str, "%f", &days) != 1) { - fprintf(stderr, "Could not parse version: %s\n", str); - flushAndExit(FDB_EXIT_ERROR); + auto days = parseNumberPrefix(s); + if (days.present()) { + double version = (double)CLIENT_KNOBS->CORE_VERSIONSPERSECOND * 24 * 3600 * -days.get(); + if (version >= (double)std::numeric_limits::min() && + version < -(double)std::numeric_limits::min()) { + return static_cast(version); + } + } + } else { + auto version = parseNumberPrefix(s); + if (version.present()) { + return version.get(); } - return (double)CLIENT_KNOBS->CORE_VERSIONSPERSECOND * 24 * 3600 * -days; } - Version ver; - if (sscanf(str, "%" SCNd64, &ver) != 1) { - fprintf(stderr, "Could not parse version: %s\n", str); - flushAndExit(FDB_EXIT_ERROR); - } - return ver; + fprintf(stderr, "Could not parse version: %s\n", str); + flushAndExit(FDB_EXIT_ERROR); + return invalidVersion; } // Creates a connection to a cluster. Optionally prints an error if the connection fails. @@ -3591,13 +3596,15 @@ int main(int argc, char* argv[]) { case OPT_EXPIRE_MIN_RESTORABLE_DAYS: case OPT_EXPIRE_DELETE_BEFORE_DAYS: { const char* a = args->OptionArg(); - long long ver = 0; - if (!sscanf(a, "%lld", &ver)) { + auto parsedVersion = parseNumberPrefix(StringRef(a)); + if (!parsedVersion.present()) { fprintf(stderr, "ERROR: Could not parse expiration version `%s'\n", a); printHelpTeaser(newArgV[0]); return FDB_EXIT_ERROR; } + Version ver = parsedVersion.get(); + // Interpret the value as days worth of versions relative to now (negative) if (optId == OPT_EXPIRE_MIN_RESTORABLE_DAYS || optId == OPT_EXPIRE_DELETE_BEFORE_DAYS) { ver = -ver * 24 * 60 * 60 * CLIENT_KNOBS->CORE_VERSIONSPERSECOND; @@ -3687,12 +3694,13 @@ int main(int argc, char* argv[]) { case OPT_INITIAL_SNAPSHOT_INTERVAL: case OPT_MOD_ACTIVE_INTERVAL: { const char* a = args->OptionArg(); - int seconds; - if (!sscanf(a, "%d", &seconds)) { + auto parsedSeconds = parseNumberPrefix(StringRef(a)); + if (!parsedSeconds.present()) { fprintf(stderr, "ERROR: Could not parse snapshot interval `%s'\n", a); printHelpTeaser(newArgV[0]); return FDB_EXIT_ERROR; } + int seconds = parsedSeconds.get(); if (optId == OPT_SNAPSHOTINTERVAL) { snapshotIntervalSeconds = seconds; modifyOptions.snapshotIntervalSeconds = seconds; @@ -3780,44 +3788,46 @@ int main(int argc, char* argv[]) { } case OPT_ERRORLIMIT: { const char* a = args->OptionArg(); - if (!sscanf(a, "%d", &maxErrors)) { + auto parsedMaxErrors = parseNumberPrefix(StringRef(a)); + if (!parsedMaxErrors.present()) { fprintf(stderr, "ERROR: Could not parse max number of errors `%s'\n", a); printHelpTeaser(newArgV[0]); return FDB_EXIT_ERROR; } + maxErrors = parsedMaxErrors.get(); break; } case OPT_RESTORE_BEGIN_VERSION: { const char* a = args->OptionArg(); - long long ver = 0; - if (!sscanf(a, "%lld", &ver)) { + auto parsedVersion = parseNumberPrefix(StringRef(a)); + if (!parsedVersion.present()) { fprintf(stderr, "ERROR: Could not parse database beginVersion `%s'\n", a); printHelpTeaser(newArgV[0]); return FDB_EXIT_ERROR; } - beginVersion = ver; + beginVersion = parsedVersion.get(); break; } case OPT_RESTORE_VERSION: { const char* a = args->OptionArg(); - long long ver = 0; - if (!sscanf(a, "%lld", &ver)) { + auto parsedVersion = parseNumberPrefix(StringRef(a)); + if (!parsedVersion.present()) { fprintf(stderr, "ERROR: Could not parse database version `%s'\n", a); printHelpTeaser(newArgV[0]); return FDB_EXIT_ERROR; } - restoreVersion = ver; + restoreVersion = parsedVersion.get(); break; } case OPT_RESTORE_SNAPSHOT_VERSION: { const char* a = args->OptionArg(); - long long ver = 0; - if (!sscanf(a, "%lld", &ver)) { + auto parsedVersion = parseNumberPrefix(StringRef(a)); + if (!parsedVersion.present()) { fprintf(stderr, "ERROR: Could not parse database version `%s'\n", a); printHelpTeaser(newArgV[0]); return FDB_EXIT_ERROR; } - snapshotVersion = ver; + snapshotVersion = parsedVersion.get(); break; } case OPT_RESTORE_USER_DATA: { @@ -3834,8 +3844,8 @@ int main(int argc, char* argv[]) { } #ifdef _WIN32 case OPT_PARENTPID: { - auto pid_str = args->OptionArg(); - int parent_pid = atoi(pid_str); + const char* pid_str = args->OptionArg(); + int parent_pid = parseNumberPrefix(StringRef(pid_str)).orDefault(0); auto pHandle = OpenProcess(SYNCHRONIZE, FALSE, parent_pid); if (!pHandle) { TraceEvent("ParentProcessOpenError").GetLastError(); diff --git a/fdbcli/AdvanceVersionCommand.cpp b/fdbcli/AdvanceVersionCommand.cpp index c6647efd4dc..1bc704bebf6 100644 --- a/fdbcli/AdvanceVersionCommand.cpp +++ b/fdbcli/AdvanceVersionCommand.cpp @@ -18,6 +18,7 @@ * limitations under the License. */ +#include "flow/ParseNumber.h" #include "boost/lexical_cast.hpp" #include "fmt/format.h" #include "fdbcli/fdbcli.h" @@ -37,12 +38,12 @@ Future advanceVersionCommandActor(Reference db, std::vector(tokens[1]); + if (!parsed.present()) { printUsage(tokens[0]); co_return false; } else { + Version v = parsed.get(); Reference tr = db->createTransaction(); while (true) { tr->setOption(FDBTransactionOptions::SPECIAL_KEY_SPACE_ENABLE_WRITES); diff --git a/fdbcli/ConsistencyCheckCommand.cpp b/fdbcli/ConsistencyCheckCommand.cpp index 213bc71a63c..d9e510e2191 100644 --- a/fdbcli/ConsistencyCheckCommand.cpp +++ b/fdbcli/ConsistencyCheckCommand.cpp @@ -30,7 +30,9 @@ namespace fdb_cli { const KeyRef consistencyCheckSpecialKey = "\xff\xff/management/consistency_check_suspended"_sr; +// The CLI retains the tokens and their backing line until this command finishes or is cancelled. Future consistencyCheckCommandActor(Reference tr, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::vector const& tokens, bool intrans) { // Here we do not proceed in a try-catch loop since the transaction is always supposed to succeed. diff --git a/fdbcli/ConsistencyScanCommand.cpp b/fdbcli/ConsistencyScanCommand.cpp index 0d56b7a0c35..4d8a4b9db97 100644 --- a/fdbcli/ConsistencyScanCommand.cpp +++ b/fdbcli/ConsistencyScanCommand.cpp @@ -41,6 +41,8 @@ Future dumpStats(ConsistencyScanState* cs, Reference consistencyScanCommandActor(Database db, std::vector const& tokens) { // Skip the command token so start at begin+1 std::list args(tokens.begin() + 1, tokens.end()); diff --git a/fdbcli/FileConfigureCommand.cpp b/fdbcli/FileConfigureCommand.cpp index 6cf827947b3..9f405980116 100644 --- a/fdbcli/FileConfigureCommand.cpp +++ b/fdbcli/FileConfigureCommand.cpp @@ -33,7 +33,9 @@ #include "flow/ThreadHelper.h" namespace fdb_cli { +// The configuration file is read into owned storage before suspension. Future fileConfigureCommandActor(Reference db, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::string const& filePath, bool isNewDatabase, bool force) { diff --git a/fdbcli/ForceRecoveryWithDataLossCommand.cpp b/fdbcli/ForceRecoveryWithDataLossCommand.cpp index cca286dc795..f74fb6671f6 100644 --- a/fdbcli/ForceRecoveryWithDataLossCommand.cpp +++ b/fdbcli/ForceRecoveryWithDataLossCommand.cpp @@ -27,6 +27,8 @@ #include "flow/ThreadHelper.h" namespace fdb_cli { +// The CLI retains the tokens and their backing line until this command finishes or is cancelled. +// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future forceRecoveryWithDataLossCommandActor(Reference db, std::vector const& tokens) { if (tokens.size() != 2) { printUsage(tokens[0]); diff --git a/fdbcli/HotRangeCommand.cpp b/fdbcli/HotRangeCommand.cpp index ca184bd1386..3ea6e3eac3f 100644 --- a/fdbcli/HotRangeCommand.cpp +++ b/fdbcli/HotRangeCommand.cpp @@ -51,10 +51,12 @@ ReadHotSubRangeRequest::SplitType parseSplitType(const std::string& typeStr) { namespace fdb_cli { +// The CLI retains the tokens and their backing line until this command finishes or is cancelled. Future hotRangeCommandActor(Database localdb, Reference db, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::vector const& tokens, - std::map* const& storage_interface) { + std::map* storage_interface) { if (tokens.size() == 1) { // initialize storage interfaces @@ -72,7 +74,7 @@ Future hotRangeCommandActor(Database localdb, } Key address = tokens[1]; // At present we only support one process(IP:Port) at a time - if (!storage_interface->count(address.toString())) { + if (!storage_interface->contains(address.toString())) { fprintf(stderr, "ERROR: storage process `%s' not recognized.\n", printable(address).c_str()); co_return false; } diff --git a/fdbcli/IdempotencyIdsCommand.cpp b/fdbcli/IdempotencyIdsCommand.cpp index 1984b769cbc..56547997970 100644 --- a/fdbcli/IdempotencyIdsCommand.cpp +++ b/fdbcli/IdempotencyIdsCommand.cpp @@ -36,6 +36,8 @@ Optional parseAgeValue(StringRef token) { namespace fdb_cli { +// The CLI retains the tokens and their backing line until this command finishes or is cancelled. +// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future idempotencyIdsCommandActor(Database db, std::vector const& tokens) { if (tokens.size() < 2 || tokens.size() > 3) { printUsage(tokens[0]); diff --git a/fdbcli/LockCommand.cpp b/fdbcli/LockCommand.cpp index e46950be033..7e937272aae 100644 --- a/fdbcli/LockCommand.cpp +++ b/fdbcli/LockCommand.cpp @@ -60,6 +60,8 @@ namespace fdb_cli { const KeyRef lockSpecialKey = "\xff\xff/management/db_locked"_sr; +// The CLI retains the tokens and their backing line until this command finishes or is cancelled. +// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future lockCommandActor(Reference db, std::vector const& tokens) { if (tokens.size() != 1) { printUsage(tokens[0]); diff --git a/fdbcli/MaintenanceCommand.cpp b/fdbcli/MaintenanceCommand.cpp index b3495ba6e5b..6ec838ec45d 100644 --- a/fdbcli/MaintenanceCommand.cpp +++ b/fdbcli/MaintenanceCommand.cpp @@ -18,6 +18,7 @@ * limitations under the License. */ +#include "flow/ParseNumber.h" #include #include "boost/lexical_cast.hpp" @@ -147,14 +148,12 @@ Future maintenanceCommandActor(Reference db, std::vector(tokens[3]); + if (!seconds.present()) { printUsage(tokens[0]); result = false; } else { - bool setResult = co_await setHealthyZone(db, tokens[2], seconds, true); + bool setResult = co_await setHealthyZone(db, tokens[2], seconds.get(), true); result = setResult; } } else { diff --git a/fdbcli/SnapshotCommand.cpp b/fdbcli/SnapshotCommand.cpp index ed8288a4dd0..afd35973136 100644 --- a/fdbcli/SnapshotCommand.cpp +++ b/fdbcli/SnapshotCommand.cpp @@ -27,6 +27,8 @@ #include "flow/ThreadHelper.h" namespace fdb_cli { +// The CLI retains the tokens and their backing line until this command finishes or is cancelled. +// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future snapshotCommandActor(Reference db, std::vector const& tokens) { bool result = true; if (tokens.size() < 2) { diff --git a/fdbcli/SuspendCommand.cpp b/fdbcli/SuspendCommand.cpp index 3f6c7a6f811..34ae46ae033 100644 --- a/fdbcli/SuspendCommand.cpp +++ b/fdbcli/SuspendCommand.cpp @@ -18,6 +18,10 @@ * limitations under the License. */ +#include "flow/ParseNumber.h" +#include +#include + #include "boost/algorithm/string.hpp" #include "fdbcli/fdbcli.h" @@ -31,8 +35,10 @@ #include "flow/ThreadHelper.h" namespace fdb_cli { +// The CLI retains the tokens and their backing line until this command finishes or is cancelled. Future suspendCommandActor(Reference db, Reference tr, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::vector const& tokens, std::map>* address_interface) { ASSERT(!tokens.empty()); @@ -58,7 +64,7 @@ Future suspendCommandActor(Reference db, result = false; } else { for (int i = 2; i < tokens.size(); i++) { - if (!address_interface->count(tokens[i])) { + if (!address_interface->contains(tokens[i])) { fprintf(stderr, "ERROR: process `%s' not recognized.\n", printable(tokens[i]).c_str()); result = false; break; @@ -66,11 +72,10 @@ Future suspendCommandActor(Reference db, } if (result) { - double seconds{ 0 }; - int n = 0; + auto seconds = parseNumber(tokens[1]); int i{ 0 }; - auto secondsStr = tokens[1].toString(); - if (sscanf(secondsStr.c_str(), "%lf%n", &seconds, &n) != 1 || n != secondsStr.size()) { + if (!seconds.present() || !std::isfinite(seconds.get()) || + seconds.get() < std::numeric_limits::min() || seconds.get() > std::numeric_limits::max()) { printUsage(tokens[0]); result = false; } else { @@ -79,8 +84,8 @@ Future suspendCommandActor(Reference db, addressesVec.push_back(tokens[i].toString()); } addressesStr = boost::algorithm::join(addressesVec, ","); - int64_t suspendRequestSent = - co_await safeThreadFutureToFuture(db->rebootWorker(addressesStr, false, static_cast(seconds))); + int64_t suspendRequestSent = co_await safeThreadFutureToFuture( + db->rebootWorker(addressesStr, false, static_cast(seconds.get()))); if (!suspendRequestSent) { result = false; fprintf( diff --git a/fdbcli/VersionEpochCommand.cpp b/fdbcli/VersionEpochCommand.cpp index 6044b0d71e7..691f2e4d314 100644 --- a/fdbcli/VersionEpochCommand.cpp +++ b/fdbcli/VersionEpochCommand.cpp @@ -18,6 +18,7 @@ * limitations under the License. */ +#include "flow/ParseNumber.h" #include "boost/lexical_cast.hpp" #include "fdbcli/fdbcli.h" @@ -121,11 +122,12 @@ Future versionEpochCommandActor(Reference db, Database cx, std: (tokens.size() == 3 && tokencmp(tokens[1], "set"))) { int64_t v; if (tokens.size() == 3) { - int n = 0; - if (sscanf(tokens[2].toString().c_str(), "%" SCNd64 "%n", &v, &n) != 1 || n != tokens[2].size()) { + auto parsed = parseNumber(tokens[2]); + if (!parsed.present()) { printUsage(tokens[0]); co_return false; } + v = parsed.get(); } else { v = 0; // default version epoch } diff --git a/fdbcli/fdbcli.cpp b/fdbcli/fdbcli.cpp index b8d4b591e97..17ce0f9801c 100644 --- a/fdbcli/fdbcli.cpp +++ b/fdbcli/fdbcli.cpp @@ -18,6 +18,7 @@ * limitations under the License. */ +#include "flow/ParseNumber.h" #include "boost/lexical_cast.hpp" #include "fmt/format.h" #include "fdbclient/ClusterConnectionFile.h" @@ -1311,13 +1312,12 @@ Future cli(CLIOptions opt, LineNoise* plinenoise, Reference(tokens[1]); + if (!v.present()) { printUsage(tokens[0]); is_error = true; } else { - co_await delay(v); + co_await delay(v.get()); } } continue; diff --git a/fdbcli/include/fdbcli/fdbcli.h b/fdbcli/include/fdbcli/fdbcli.h index 513bf8534ff..820fa31bdf4 100644 --- a/fdbcli/include/fdbcli/fdbcli.h +++ b/fdbcli/include/fdbcli/fdbcli.h @@ -301,7 +301,7 @@ Future unlockDatabaseActor(Reference db, UID uid); Future hotRangeCommandActor(Database localDb, Reference db, std::vector const& tokens, - std::map* const& storage_interface); + std::map* storage_interface); // maintenance command Future setHealthyZone(Reference db, StringRef zoneId, double seconds, bool printWarning = false); diff --git a/fdbclient/BackupAgentBase.cpp b/fdbclient/BackupAgentBase.cpp index ef3406d6450..06f06370e4e 100644 --- a/fdbclient/BackupAgentBase.cpp +++ b/fdbclient/BackupAgentBase.cpp @@ -30,6 +30,7 @@ #include "fdbclient/SystemData.h" #include "fdbrpc/simulator.h" #include "flow/ActorCollection.h" +#include "flow/ParseNumber.h" #include "flow/DeterministicRandom.h" #include "flow/network.h" @@ -66,11 +67,19 @@ int64_t BackupAgentBase::parseTime(std::string timestamp) { #endif // Read timezone offset in +/-HHMM format then convert to seconds - int tzHH; - int tzMM; - if (sscanf(timestamp.substr(19, 5).c_str(), "%3d%2d", &tzHH, &tzMM) != 2) { + StringRef timezone(timestamp); + if (timezone.size() < 24) { return -1; } + timezone = timezone.substr(19, 5); + int consumed = 0; + auto hours = parseNumberPrefix(timezone.substr(0, 3), 10, &consumed); + auto minutes = parseNumberPrefix(timezone.substr(consumed, 2)); + if (!hours.present() || !minutes.present()) { + return -1; + } + int tzHH = hours.get(); + int tzMM = minutes.get(); if (tzHH < 0) { tzMM = -tzMM; } @@ -140,13 +149,28 @@ bool copyParameter(Reference source, Reference dest, Key key) { } Version getVersionFromString(std::string const& value) { - Version version = invalidVersion; - int n = 0; - if (sscanf(value.c_str(), "%lld%n", (long long*)&version, &n) != 1 || n != value.size()) { + auto version = parseNumber(StringRef(value)); + if (!version.present()) { TraceEvent(SevWarnAlways, "GetVersionFromString").detail("InvalidVersion", value); throw restore_invalid_version(); } - return version; + return version.get(); +} + +TEST_CASE("/backup/versionparsing") { + ASSERT(getVersionFromString("9223372036854775807") == std::numeric_limits::max()); + ASSERT(getVersionFromString("-9223372036854775808") == std::numeric_limits::min()); + for (const auto& value : { "", " ", "9223372036854775808", "-9223372036854775809", "123suffix", "123 " }) { + bool rejected = false; + try { + getVersionFromString(value); + } catch (Error& e) { + ASSERT(e.code() == error_code_restore_invalid_version); + rejected = true; + } + ASSERT(rejected); + } + return Void(); } // Transaction log data is stored by the FoundationDB core in the diff --git a/fdbclient/BackupContainerFileSystem.cpp b/fdbclient/BackupContainerFileSystem.cpp index 2b9030e8095..4b4ec772c22 100644 --- a/fdbclient/BackupContainerFileSystem.cpp +++ b/fdbclient/BackupContainerFileSystem.cpp @@ -29,9 +29,11 @@ #include "fdbrpc/AsyncFileEncrypted.h" #include "flow/StreamCipher.h" #include "flow/UnitTest.h" +#include "flow/ParseNumber.h" #include #include +#include class BackupContainerFileSystemImpl { public: @@ -170,16 +172,24 @@ class BackupContainerFileSystemImpl { static bool pathToRangeFile(RangeFile& out, const std::string& path, int64_t size) { std::string name = fileNameOnly(path); + StringRef fields(name); + if (name.find('\0') != std::string::npos || fields.eat(","_sr) != "range"_sr) { + return false; + } + auto version = parseNumber(fields.eat(","_sr)); + bool foundSeparator = false; + auto uid = fields.eat(","_sr, &foundSeparator); + auto blockSize = parseNumber(fields); + if (!version.present() || uid.empty() || !foundSeparator || !blockSize.present()) { + return false; + } RangeFile f; f.fileName = path; f.fileSize = size; - int len; - if (sscanf(name.c_str(), "range,%" SCNd64 ",%*[^,],%u%n", &f.version, &f.blockSize, &len) == 2 && - len == name.size()) { - out = f; - return true; - } - return false; + f.version = version.get(); + f.blockSize = blockSize.get(); + out = f; + return true; } static Future writeKeyspaceSnapshotFile(Reference bc, @@ -1384,11 +1394,16 @@ class BackupContainerFileSystemImpl { // Extract the snapshot begin version from a path static Version extractSnapshotBeginVersion(const std::string& path) { - Version snapshotBeginVersion; - if (sscanf(path.c_str(), "kvranges/snapshot.%018" SCNd64, &snapshotBeginVersion) == 1) { - return snapshotBeginVersion; + StringRef input(path); + const StringRef prefix = "kvranges/snapshot."_sr; + if (!input.startsWith(prefix)) { + return invalidVersion; + } + input = input.substr(prefix.size()); + while (!input.empty() && std::isspace(static_cast(input[0]))) { + input = input.substr(1); } - return invalidVersion; + return parseNumberPrefix(input.substr(0, std::min(input.size(), 18))).orDefault(invalidVersion); } // The innermost folder covers 100,000 seconds (1e11 versions) which is 5,000 mutation log files at current @@ -1409,67 +1424,63 @@ class BackupContainerFileSystemImpl { static bool pathToLogFile(LogFile& out, const std::string& path, int64_t size) { std::string name = fileNameOnly(path); + StringRef fields(name); + if (name.find('\0') != std::string::npos || fields.eat(","_sr) != "log"_sr) { + return false; + } + auto beginVersion = parseNumber(fields.eat(","_sr)); + auto endVersion = parseNumber(fields.eat(","_sr)); + bool foundSeparator = false; + auto uid = fields.eat(","_sr, &foundSeparator); + if (!beginVersion.present() || !endVersion.present() || uid.empty() || !foundSeparator) { + return false; + } LogFile f; f.fileName = path; f.fileSize = size; - int len; - if (sscanf(name.c_str(), - "log,%" SCNd64 ",%" SCNd64 ",%*[^,],%u%n", - &f.beginVersion, - &f.endVersion, - &f.blockSize, - &len) == 3 && - len == name.size()) { - out = f; - return true; - } else if (sscanf(name.c_str(), - "log,%" SCNd64 ",%" SCNd64 ",%*[^,],%d-of-%d,%u%n", - &f.beginVersion, - &f.endVersion, - &f.tagId, - &f.totalTags, - &f.blockSize, - &len) == 5 && - len == name.size() && f.tagId >= 0) { - out = f; - return true; + f.beginVersion = beginVersion.get(); + f.endVersion = endVersion.get(); + auto blockSize = parseNumber(fields); + if (!blockSize.present()) { + auto tagId = parseNumber(fields.eat("-of-"_sr, &foundSeparator)); + if (!tagId.present() || tagId.get() < 0 || !foundSeparator) { + return false; + } + auto totalTags = parseNumber(fields.eat(","_sr)); + blockSize = parseNumber(fields); + if (!totalTags.present() || !blockSize.present()) { + return false; + } + f.tagId = tagId.get(); + f.totalTags = totalTags.get(); } - return false; + f.blockSize = blockSize.get(); + out = f; + return true; } static bool pathToKeyspaceSnapshotFile(KeyspaceSnapshotFile& out, const std::string& path) { std::string name = fileNameOnly(path); - KeyspaceSnapshotFile f; - f.fileName = path; - int len; - char typeBuf[64] = {}; - - // Try new format with type suffix: snapshot,beginVersion,endVersion,totalSize,type - if (sscanf(name.c_str(), - "snapshot,%" SCNd64 ",%" SCNd64 ",%" SCNd64 ",%63[^,]%n", - &f.beginVersion, - &f.endVersion, - &f.totalSize, - typeBuf, - &len) == 4 && - len == name.size()) { - f.snapshotType = typeBuf; - out = f; - return true; + StringRef fields(name); + if (name.find('\0') != std::string::npos || fields.eat(","_sr) != "snapshot"_sr) { + return false; } - - // Try original format: snapshot,beginVersion,endVersion,totalSize - if (sscanf(name.c_str(), - "snapshot,%" SCNd64 ",%" SCNd64 ",%" SCNd64 "%n", - &f.beginVersion, - &f.endVersion, - &f.totalSize, - &len) == 3 && - len == name.size()) { - out = f; - return true; + auto beginVersion = parseNumber(fields.eat(","_sr)); + auto endVersion = parseNumber(fields.eat(","_sr)); + bool hasType = false; + auto totalSize = parseNumber(fields.eat(","_sr, &hasType)); + if (!beginVersion.present() || !endVersion.present() || !totalSize.present() || + (hasType && (fields.empty() || fields.size() > 63 || fields.toString().find(',') != std::string::npos))) { + return false; } - return false; + KeyspaceSnapshotFile f; + f.fileName = path; + f.beginVersion = beginVersion.get(); + f.endVersion = endVersion.get(); + f.totalSize = totalSize.get(); + f.snapshotType = fields.toString(); + out = f; + return true; } // fallback for using existing write api if the underlying blob store doesn't support efficient writeEntireFile @@ -1850,10 +1861,9 @@ static Future> readVersionProperty(Referenceread((uint8_t*)s.data(), size, 0); - Version v; - int len; - if (rs == size && sscanf(s.c_str(), "%" SCNd64 "%n", &v, &len) == 1 && len == size) - co_return v; + auto version = parseNumber(StringRef(s)); + if (rs == size && version.present()) + co_return version.get(); TraceEvent(SevWarn, "BackupContainerInvalidProperty").detail("URL", bc->getURL()).detail("Path", path); @@ -2218,6 +2228,59 @@ TEST_CASE("/backup/containers_list") { } } +TEST_CASE("/backup/filenameparsing") { + RangeFile range; + ASSERT(BackupContainerFileSystemImpl::pathToRangeFile(range, "ranges/range,123,uid,4096", 100)); + ASSERT(range.version == 123 && range.blockSize == 4096 && range.fileSize == 100); + for (const auto& name : { "ranges/range,9223372036854775808,uid,4096", + "ranges/range,123,uid,4294967296", + "ranges/range,,uid,4096", + "ranges/range,123,,4096", + "ranges/range,123,uid,4096suffix" }) { + ASSERT(!BackupContainerFileSystemImpl::pathToRangeFile(range, name, 0)); + } + + LogFile log; + ASSERT(BackupContainerFileSystemImpl::pathToLogFile(log, "logs/log,123,456,uid,4096", 100)); + ASSERT(log.beginVersion == 123 && log.endVersion == 456 && log.blockSize == 4096 && log.tagId == -1); + ASSERT(BackupContainerFileSystemImpl::pathToLogFile(log, "plogs/log,123,456,uid,2-of-4,4096", 100)); + ASSERT(log.tagId == 2 && log.totalTags == 4 && log.blockSize == 4096); + for (const auto& name : { "logs/log,123,9223372036854775808,uid,4096", + "plogs/log,123,456,uid,2147483648-of-4,4096", + "plogs/log,123,456,uid,2-of-2147483648,4096", + "plogs/log,123,456,uid,-1-of-4,4096", + "plogs/log,123,456,uid,2-of-4,4294967296", + "plogs/log,123,456,uid,2-of-4,4096suffix" }) { + ASSERT(!BackupContainerFileSystemImpl::pathToLogFile(log, name, 0)); + } + + KeyspaceSnapshotFile snapshot; + ASSERT(BackupContainerFileSystemImpl::pathToKeyspaceSnapshotFile(snapshot, "snapshots/snapshot,123,456,789")); + ASSERT(snapshot.beginVersion == 123 && snapshot.endVersion == 456 && snapshot.totalSize == 789 && + snapshot.snapshotType.empty()); + ASSERT(BackupContainerFileSystemImpl::pathToKeyspaceSnapshotFile(snapshot, "snapshots/snapshot,123,456,789,bulk")); + ASSERT(snapshot.snapshotType == "bulk"); + ASSERT(BackupContainerFileSystemImpl::pathToKeyspaceSnapshotFile( + snapshot, "snapshots/snapshot,123,456,789," + std::string(63, 'x'))); + ASSERT(!BackupContainerFileSystemImpl::pathToKeyspaceSnapshotFile( + snapshot, "snapshots/snapshot,123,456,789," + std::string(64, 'x'))); + for (const auto& name : { "snapshots/snapshot,123,456,9223372036854775808", + "snapshots/snapshot,123,456,", + "snapshots/snapshot,123,456,789,", + "snapshots/snapshot,123,456,789,bulk,extra" }) { + ASSERT(!BackupContainerFileSystemImpl::pathToKeyspaceSnapshotFile(snapshot, name)); + } + + ASSERT(BackupContainerFileSystemImpl::extractSnapshotBeginVersion("kvranges/snapshot.000000000000000123/range") == + 123); + ASSERT(BackupContainerFileSystemImpl::extractSnapshotBeginVersion( + "kvranges/snapshot. \t000000000000000123/range") == 123); + ASSERT(BackupContainerFileSystemImpl::extractSnapshotBeginVersion("kvranges/snapshot. \t") == invalidVersion); + ASSERT(BackupContainerFileSystemImpl::extractSnapshotBeginVersion("kvranges/snapshot.invalid/range") == + invalidVersion); + return Void(); +} + TEST_CASE("/backup/time") { // test formatTime() for (int i = 0; i < 1000; ++i) { diff --git a/fdbclient/BackupContainerLocalDirectory.cpp b/fdbclient/BackupContainerLocalDirectory.cpp index 78f91eb9acf..40088c54de4 100644 --- a/fdbclient/BackupContainerLocalDirectory.cpp +++ b/fdbclient/BackupContainerLocalDirectory.cpp @@ -24,6 +24,7 @@ #include "flow/IAsyncFile.h" #include "flow/FaultInjection.h" #include "flow/Platform.h" +#include "flow/ParseNumber.h" #include "fdbrpc/simulator.h" #include "fdbrpc/SimulatorProcessInfo.h" @@ -271,7 +272,7 @@ Future> BackupContainerLocalDirectory::readFile(const std: // Extract block size from the filename, if present size_t lastComma = path.find_last_of(','); if (lastComma != path.npos) { - blockSize = atoi(path.substr(lastComma + 1).c_str()); + blockSize = parseNumberPrefix(StringRef(path).substr(lastComma + 1)).orDefault(0); } if (blockSize <= 0) { blockSize = deterministicRandom()->randomInt(1e4, 1e6); diff --git a/fdbclient/DatabaseConfiguration.cpp b/fdbclient/DatabaseConfiguration.cpp index f25072085d8..95826a76e34 100644 --- a/fdbclient/DatabaseConfiguration.cpp +++ b/fdbclient/DatabaseConfiguration.cpp @@ -20,6 +20,7 @@ #include #include "fdbclient/DatabaseConfiguration.h" +#include "flow/ParseNumber.h" #include "fdbclient/FDBTypes.h" #include "fdbclient/SystemData.h" #include "flow/ITrace.h" @@ -59,22 +60,19 @@ void DatabaseConfiguration::resetInternal() { } int toInt(ValueRef const& v) { - return atoi(v.toString().c_str()); + return parseNumberPrefix(v).orDefault(0); } void parse(int* i, ValueRef const& v) { - // FIXME: Sanity checking - *i = atoi(v.toString().c_str()); + *i = parseNumberPrefix(v).orDefault(0); } void parse(int64_t* i, ValueRef const& v) { - // FIXME: Sanity checking - *i = atoll(v.toString().c_str()); + *i = parseNumberPrefix(v).orDefault(0); } void parse(double* i, ValueRef const& v) { - // FIXME: Sanity checking - *i = atof(v.toString().c_str()); + *i = parseNumberPrefix(v).orDefault(0); } void parseReplicationPolicy(Reference* policy, ValueRef const& v) { @@ -888,7 +886,7 @@ bool DatabaseConfiguration::isOverridden(std::string key) const { key = configKeysPrefix.toString() + std::move(key); if (mutableConfiguration.present()) { - return mutableConfiguration.get().find(key) != mutableConfiguration.get().end(); + return mutableConfiguration.get().contains(key); } const int keyLen = key.size(); diff --git a/fdbclient/ManagementAPI.cpp b/fdbclient/ManagementAPI.cpp index c83b44233ed..1b58507ee4a 100644 --- a/fdbclient/ManagementAPI.cpp +++ b/fdbclient/ManagementAPI.cpp @@ -48,6 +48,7 @@ #include "fdbrpc/simulator.h" #include "fdbclient/StatusClient.h" #include "flow/Trace.h" +#include "flow/ParseNumber.h" #include "flow/UnitTest.h" #include "fdbrpc/ReplicationPolicy.h" #include "fdbrpc/Replication.h" @@ -103,11 +104,12 @@ std::map configForToken(std::string const& mode) { std::string key = mode.substr(0, pos); std::string value = mode.substr(pos + 1); - if (key == "proxies" && isInteger(value)) { + auto specifiedProxiesCount = key == "proxies" ? parseNumber(StringRef(value)) : Optional(); + if (key == "proxies" && specifiedProxiesCount.present()) { printf("Warning: Proxy role is being split into GRV Proxy and Commit Proxy, now prefer configuring " "'grv_proxies' and 'commit_proxies' separately. Generally we should follow that 'commit_proxies'" " is three times of 'grv_proxies' count and 'grv_proxies' should be not more than 4.\n"); - int proxiesCount = atoi(value.c_str()); + int proxiesCount = specifiedProxiesCount.get(); if (proxiesCount == -1) { proxiesCount = CLIENT_KNOBS->DEFAULT_AUTO_GRV_PROXIES + CLIENT_KNOBS->DEFAULT_AUTO_COMMIT_PROXIES; ASSERT_WE_THINK(proxiesCount >= 2); @@ -132,7 +134,7 @@ std::map configForToken(std::string const& mode) { commitProxyCount); TraceEvent("DatabaseConfigurationProxiesSpecified") - .detail("SpecifiedProxies", atoi(value.c_str())) + .detail("SpecifiedProxies", specifiedProxiesCount.get()) .detail("EffectiveSpecifiedProxies", proxiesCount) .detail("ConvertedGrvProxies", grvProxyCount) .detail("ConvertedCommitProxies", commitProxyCount); @@ -1201,8 +1203,8 @@ struct AutoQuorumChange final : IQuorumChange { Future> fStorageReplicas = tr->get("storage_replicas"_sr.withPrefix(configKeysPrefix)); Future> fLogReplicas = tr->get("log_replicas"_sr.withPrefix(configKeysPrefix)); co_await (success(fStorageReplicas) && success(fLogReplicas)); - int redundancy = std::min(atoi(fStorageReplicas.get().get().toString().c_str()), - atoi(fLogReplicas.get().get().toString().c_str())); + int redundancy = std::min(parseNumberPrefix(fStorageReplicas.get().get()).orDefault(0), + parseNumberPrefix(fLogReplicas.get().get()).orDefault(0)); co_return redundancy; } @@ -1466,7 +1468,7 @@ Future excludeServers(Transaction* tr, std::vector serve std::set exclusions(excl.begin(), excl.end()); bool containNewExclusion = false; for (auto& s : servers) { - if (exclusions.find(s) != exclusions.end()) { + if (exclusions.contains(s)) { continue; } containNewExclusion = true; @@ -1540,7 +1542,7 @@ Future excludeLocalities(Transaction* tr, std::unordered_set std::set exclusion(excl.begin(), excl.end()); bool containNewExclusion = false; for (const auto& l : localities) { - if (exclusion.find(l) != exclusion.end()) { + if (exclusion.contains(l)) { continue; } containNewExclusion = true; diff --git a/fdbclient/S3BlobStore.cpp b/fdbclient/S3BlobStore.cpp index 10dc842d152..0d2255923ad 100644 --- a/fdbclient/S3BlobStore.cpp +++ b/fdbclient/S3BlobStore.cpp @@ -545,11 +545,17 @@ void S3BlobStoreEndpoint::processRequestFailure(Reference S3BlobStoreEndpoint::preRetryCheck(std::string const& verb, - std::string const& resource, - ReusableConnection& rconn, - int requestTimeout, - bool& retryExtended) { +// The strings are consumed before suspension; doRequest_impl retains its connection and retry state while awaiting us. +Future S3BlobStoreEndpoint::preRetryCheck( + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) + std::string const& verb, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) + std::string const& resource, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) + ReusableConnection& rconn, + int requestTimeout, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) + bool& retryExtended) { if (!isWriteRequest(verb) || !CLIENT_KNOBS->BACKUP_ALLOW_DRYRUN) { co_return true; } diff --git a/fdbclient/include/fdbclient/FDBTypes.h b/fdbclient/include/fdbclient/FDBTypes.h index 5ad9fc5c3a9..cb913dadfb6 100644 --- a/fdbclient/include/fdbclient/FDBTypes.h +++ b/fdbclient/include/fdbclient/FDBTypes.h @@ -35,6 +35,7 @@ #include "flow/FastRef.h" #include "flow/ProtocolVersion.h" +#include "flow/ParseNumber.h" #include "flow/flow.h" #include "fdbclient/ProcessClass.h" #include "fdbclient/ProcessData.h" @@ -1514,7 +1515,7 @@ struct EncryptionAtRestModeDeprecated { } // A failed parsing returns 0 (DISABLED) - int num = atoi(val.get().toString().c_str()); + int num = parseNumberPrefix(val.get()).orDefault(0); if (num < 0 || num >= END) { return DISABLED; } diff --git a/fdbclient/include/fdbclient/IBlobStore.h b/fdbclient/include/fdbclient/IBlobStore.h index 949f084efb3..80e731b900e 100644 --- a/fdbclient/include/fdbclient/IBlobStore.h +++ b/fdbclient/include/fdbclient/IBlobStore.h @@ -451,7 +451,7 @@ class IBlobStoreEndpoint { ReusableConnection& rconn, int requestTimeout, bool& retryExtended) { - co_return true; + return true; } // Do an HTTP request to the blob store, read the response. Handles connection, retry, and authentication. diff --git a/fdbclient/include/fdbclient/RandomKeyValueUtils.h b/fdbclient/include/fdbclient/RandomKeyValueUtils.h index e0ad845db62..89b6e9ec6d8 100644 --- a/fdbclient/include/fdbclient/RandomKeyValueUtils.h +++ b/fdbclient/include/fdbclient/RandomKeyValueUtils.h @@ -27,6 +27,7 @@ #include "flow/Arena.h" #include "flow/Error.h" #include "flow/IRandom.h" +#include "flow/ParseNumber.h" #include "fdbclient/FDBTypes.h" template @@ -64,7 +65,7 @@ struct RandomIntGenerator : IGenerator { alpha = true; return (unsigned int)s[0]; } else { - return atol(s.toString().c_str()); + return parseNumberPrefix(s).orDefault(0); } } @@ -220,10 +221,11 @@ struct RandomStringSetGeneratorBase : IKeyGenerator { maxKeyLen = keyGen.getMaxKeyLen(); ASSERT(indexGenerator.max > 0); std::set uniqueKeys; - int inserts = 0; + uint64_t inserts = 0; // for smaller indexGenerator.max, give it more insert try, as it may not find enough unique keys with 3 * max. // It adds roughly log * 100. For example, even for max is 1, it will try at least 100 times. - const uint maxInsertTry = 3 * indexGenerator.max + (((sizeof(uint) * 8) - clz(indexGenerator.max)) * 100); + const uint64_t maxInsertTry = + uint64_t{ 3 } * indexGenerator.max + (((sizeof(uint) * 8) - clz(indexGenerator.max)) * 100); while (uniqueKeys.size() < indexGenerator.max) { auto k = keyGen.next(); uniqueKeys.insert(k); diff --git a/fdbmonitor/fdbmonitor.h b/fdbmonitor/fdbmonitor.h index 0081a6ec99d..956fa2eeedc 100644 --- a/fdbmonitor/fdbmonitor.h +++ b/fdbmonitor/fdbmonitor.h @@ -29,6 +29,7 @@ #include #include #include +#include #include #include #include @@ -88,6 +89,8 @@ void print_usage(const char* name); std::unordered_map> set_watches(std::string path, int ifd); void load_conf(const char* confpath, uid_t& uid, gid_t& gid, sigset_t* mask, fdb_fd_set rfds, int* maxfd); uint64_t getRss(ProcessID id); +// Reads the first two /proc/statm fields and converts resident pages to bytes. Leaves rss unchanged on failure. +bool parseRss(std::string_view statm, long pageSize, uint64_t& rss); void kill_process(ProcessID id, bool wait = true, bool cleanup = true); struct Command; diff --git a/fdbmonitor/fdbmonitor_lib.cpp b/fdbmonitor/fdbmonitor_lib.cpp index 1c6b3a288ea..ac1ea346df6 100644 --- a/fdbmonitor/fdbmonitor_lib.cpp +++ b/fdbmonitor/fdbmonitor_lib.cpp @@ -19,6 +19,8 @@ */ #include +#include +#include #include #include #ifndef _WIN32 @@ -417,6 +419,32 @@ std::unordered_map> id_command; std::unordered_map pid_id; std::unordered_map id_pid; +bool parseRss(std::string_view statm, long pageSize, uint64_t& rss) { + auto parsePages = [&statm](uint64_t& pages) { + while (!statm.empty() && std::isspace(static_cast(statm.front()))) { + statm.remove_prefix(1); + } + if (statm.empty()) { + return false; + } + const char* end = statm.data() + statm.size(); + auto result = std::from_chars(statm.data(), statm.data() + statm.size(), pages); + if (result.ec != std::errc() || (result.ptr != end && !std::isspace(static_cast(*result.ptr)))) { + return false; + } + statm.remove_prefix(result.ptr - statm.data()); + return true; + }; + uint64_t sizePages; + uint64_t residentPages; + if (pageSize <= 0 || !parsePages(sizePages) || !parsePages(residentPages) || + residentPages > std::numeric_limits::max() / static_cast(pageSize)) { + return false; + } + rss = residentPages * static_cast(pageSize); + return true; +} + // Return resident memory in bytes for the given process, or 0 if error. uint64_t getRss(ProcessID id) { #ifndef __linux__ @@ -431,14 +459,15 @@ uint64_t getRss(ProcessID id) { log_msg(SevWarn, "Unable to open stat file for %s\n", id.c_str()); return 0; } - long rss = 0; - int ret = fscanf(stat_file, "%*s%ld", &rss); - if (ret == 0) { + char stat_buf[256]; + bool read = fgets(stat_buf, sizeof(stat_buf), stat_file) != nullptr; + fclose(stat_file); + uint64_t rss; + if (!read || !parseRss(stat_buf, sysconf(_SC_PAGESIZE), rss)) { log_msg(SevWarn, "Unable to parse rss size for %s\n", id.c_str()); return 0; } - fclose(stat_file); - return static_cast(rss) * sysconf(_SC_PAGESIZE); + return rss; #endif } diff --git a/fdbmonitor/fdbmonitor_tests.cpp b/fdbmonitor/fdbmonitor_tests.cpp index 01d22a52227..61b606fb09a 100644 --- a/fdbmonitor/fdbmonitor_tests.cpp +++ b/fdbmonitor/fdbmonitor_tests.cpp @@ -3,6 +3,7 @@ #include #include #include +#include namespace fdbmonitor { namespace tests { @@ -154,6 +155,39 @@ void testEnvVarUtils() { assert_msg(!EnvVarUtils::keyValueValid("=BAZ", "FOO=BAR =BAZ"), "Key must be non-empty"); } +void testRssParsing() { + uint64_t rss = 123; + assert_msg(parseRss("100 20 3 4 5 6 7\n", 4096, rss) && rss == 81920, "Resident pages must convert to bytes"); + assert_msg(parseRss(" \t100 0\n", 4096, rss) && rss == 0, "Zero resident pages must be valid"); + const char bounded[] = { '1', '0', '0', ' ', '2', '0' }; + assert_msg(parseRss(std::string_view(bounded, 5), 4096, rss) && rss == 8192, "Parsing must honor the input length"); + const uint64_t maxBytes = std::numeric_limits::max(); + assert_msg(parseRss("100 " + std::to_string(maxBytes), 1, rss) && rss == maxBytes, + "The largest byte count must be valid"); + const uint64_t maxPages = maxBytes / 4096; + assert_msg(parseRss("100 " + std::to_string(maxPages), 4096, rss) && rss == maxPages * 4096, + "The largest whole-page byte count must be valid"); + for (const char* invalid : { "", + " \t", + "100", + "100 ", + "x 20", + "100 x", + "100 2x", + "-1 20", + "100 -1", + "18446744073709551616 20", + "100 18446744073709551616" }) { + rss = 123; + assert_msg(!parseRss(invalid, 4096, rss) && rss == 123, "Invalid statm must leave RSS unchanged"); + } + rss = 123; + assert_msg(!parseRss("100 " + std::to_string(maxPages + 1), 4096, rss) && rss == 123, + "Resident byte-count overflow must be rejected"); + assert_msg(!parseRss("100 20", 0, rss) && rss == 123, "Zero page size must be rejected"); + assert_msg(!parseRss("100 20", -1, rss) && rss == 123, "Failed page-size lookup must be rejected"); +} + } // namespace tests } // namespace fdbmonitor @@ -162,4 +196,5 @@ int main(int argc, char** argv) { testPathOps(); testEnvVarUtils(); + testRssParsing(); } diff --git a/fdbrpc/HTTP.cpp b/fdbrpc/HTTP.cpp index 7de0e52c23d..fad5985eec4 100644 --- a/fdbrpc/HTTP.cpp +++ b/fdbrpc/HTTP.cpp @@ -26,6 +26,8 @@ #include "flow/Trace.h" #include "flow/Knobs.h" #include "flow/CodeProbe.h" +#include "flow/ParseNumber.h" +#include "flow/UnitTest.h" #include "md5/md5.h" #include "libb64/encode.h" #include @@ -476,6 +478,33 @@ Future readHTTPData(HTTPData* r, } } +static Optional parseHTTPVersion(StringRef text, int* consumed = nullptr) { + if (!text.startsWith("HTTP/"_sr)) { + return {}; + } + int versionLength = 0; + auto version = parseNumberPrefix(text.substr(5), 10, &versionLength); + if (version.present() && consumed != nullptr) { + *consumed = 5 + versionLength; + } + return version; +} + +static bool parseHTTPResponseLine(StringRef text, float& version, int& code) { + int versionLength = 0; + auto parsedVersion = parseHTTPVersion(text, &versionLength); + if (!parsedVersion.present()) { + return false; + } + auto parsedCode = parseNumberPrefix(text.substr(versionLength)); + if (!parsedCode.present()) { + return false; + } + version = parsedVersion.get(); + code = parsedCode.get(); + return true; +} + // Reads an HTTP request from a network connection // If the connection fails while being read the exception will emitted // If the response is not parsable or complete in some way, http_bad_response will be thrown @@ -530,9 +559,8 @@ Future read_http_request(Reference r, Reference read_http_response(Reference r, Referenceversion, &r->code, &reachedEnd) < 2 || reachedEnd < 0) { + if (!parseHTTPResponseLine(StringRef(buf).substr(pos, lineLen), r->version, r->code)) { TraceEvent(SevWarn, "HTTPResponseParseFailure") .detail("Buffer", buf.substr(pos, std::min(lineLen, (size_t)100))) .detail("Pos", pos) @@ -588,6 +615,23 @@ Future read_http_response(Reference r, Referencedata, conn, &buf, &pos, header_only, skipCheckMD5); } +TEST_CASE("/fdbrpc/HTTP/NumericFields") { + ASSERT(!parseHTTPVersion("not-http"_sr).present()); + ASSERT(!parseHTTPVersion("HTTP/"_sr).present()); + ASSERT(!parseHTTPVersion("HTTP/1e1000"_sr).present()); + ASSERT(parseHTTPVersion("HTTP/1.1"_sr).get() == 1.1f); + + float version = 0; + int code = 0; + ASSERT(parseHTTPResponseLine("HTTP/1.1 200 OK"_sr, version, code)); + ASSERT(version == 1.1f && code == 200); + ASSERT(parseHTTPResponseLine("HTTP/1.0\t404 Not Found"_sr, version, code)); + ASSERT(version == 1.0f && code == 404); + ASSERT(!parseHTTPResponseLine("HTTP/1.1 "_sr, version, code)); + ASSERT(!parseHTTPResponseLine("HTTP/1.1 2147483648"_sr, version, code)); + return Void(); +} + Future HTTP::IncomingResponse::read(Reference conn, bool header_only) { return read_http_response(Reference::addRef(this), conn, header_only); } diff --git a/fdbrpc/JsonWebKeySet.cpp b/fdbrpc/JsonWebKeySet.cpp index 62e5c3f5653..8b61d4e5ba6 100644 --- a/fdbrpc/JsonWebKeySet.cpp +++ b/fdbrpc/JsonWebKeySet.cpp @@ -356,24 +356,32 @@ Optional parseRsaKey(StringRef b64n, JWK_PARSE_ERROR_OSSL("RSA_set0_key()"); return {}; } - // set0 == ownership taken by rsa, no need to free + // Successful RSA_set0_key transfers ownership to rsa. + // NOLINTBEGIN(bugprone-unused-return-value) n.release(); e.release(); d.release(); + // NOLINTEND(bugprone-unused-return-value) if (!isPublic) { if (1 != ::RSA_set0_factors(rsa, p, q)) { JWK_PARSE_ERROR_OSSL("RSA_set0_factors()"); return {}; } + // Successful RSA_set0_factors transfers ownership to rsa. + // NOLINTBEGIN(bugprone-unused-return-value) p.release(); q.release(); + // NOLINTEND(bugprone-unused-return-value) if (1 != ::RSA_set0_crt_params(rsa, dp, dq, qi)) { JWK_PARSE_ERROR_OSSL("RSA_set0_crt_params()"); return {}; } + // Successful RSA_set0_crt_params transfers ownership to rsa. + // NOLINTBEGIN(bugprone-unused-return-value) dp.release(); dq.release(); qi.release(); + // NOLINTEND(bugprone-unused-return-value) } auto pkey = AutoCPointer(::EVP_PKEY_new(), &::EVP_PKEY_free); if (!pkey) { diff --git a/fdbrpc/ReplicationUtils.cpp b/fdbrpc/ReplicationUtils.cpp index 38566e3c5e0..3d9ba022476 100644 --- a/fdbrpc/ReplicationUtils.cpp +++ b/fdbrpc/ReplicationUtils.cpp @@ -22,6 +22,7 @@ #include "flow/Hash3.h" #include "flow/UnitTest.h" #include "flow/Platform.h" +#include "flow/ParseNumber.h" #include "fdbrpc/ReplicationPolicy.h" #include "fdbrpc/Replication.h" @@ -761,6 +762,9 @@ Reference randomAcrossPolicy(LocalitySet const& serverSet) { } int testReplication() { + auto parseEnvInt = [](const char* value, int defaultValue) { + return value ? parseNumberPrefix(StringRef(value)).orDefault(0) : defaultValue; + }; const char* testTotalEnv = getenv("REPLICATION_TESTTOTAL"); const char* debugLevelEnv = getenv("REPLICATION_DEBUGLEVEL"); const char* policyTotalEnv = getenv("REPLICATION_POLICYTOTAL"); @@ -773,16 +777,16 @@ int testReplication() { const char* rateSampleEnv = getenv("REPLICATION_RATESAMPLE"); const char* policySampleEnv = getenv("REPLICATION_POLICYSAMPLE"); const char* policyMinEnv = getenv("REPLICATION_POLICYEXTRA"); - int totalTests = testTotalEnv ? atoi(testTotalEnv) : 10000; - int skipTotal = skipTotalEnv ? atoi(skipTotalEnv) : 0; - int findBest = findBestEnv ? atoi(findBestEnv) : 0; - int policyIndexStatic = policyIndexEnv ? atoi(policyIndexEnv) : -1; - int policyTotal = policyTotalEnv ? atoi(policyTotalEnv) : 100; - bool stopOnError = stopOnErrorEnv ? (atoi(stopOnErrorEnv) > 0) : false; - bool validate = validateEnv ? (atoi(validateEnv) > 0) : true; - int rateSample = rateSampleEnv ? atoi(rateSampleEnv) : 1000; - int policySample = policySampleEnv ? atoi(policySampleEnv) : 100; - int policyMin = policyMinEnv ? atoi(policyMinEnv) : 2; + int totalTests = parseEnvInt(testTotalEnv, 10000); + int skipTotal = parseEnvInt(skipTotalEnv, 0); + int findBest = parseEnvInt(findBestEnv, 0); + int policyIndexStatic = parseEnvInt(policyIndexEnv, -1); + int policyTotal = parseEnvInt(policyTotalEnv, 100); + bool stopOnError = parseEnvInt(stopOnErrorEnv, 0) > 0; + bool validate = parseEnvInt(validateEnv, 1) > 0; + int rateSample = parseEnvInt(rateSampleEnv, 1000); + int policySample = parseEnvInt(policySampleEnv, 100); + int policyMin = parseEnvInt(policyMinEnv, 2); int policyIndex, testCounter, alsoSize, debugBackup, maxAlsoSize; std::vector serverIndexes; Reference testServers; @@ -791,7 +795,7 @@ int testReplication() { int totalErrors = 0; if (debugLevelEnv) - g_replicationdebug = atoi(debugLevelEnv); + g_replicationdebug = parseEnvInt(debugLevelEnv, 0); debugBackup = g_replicationdebug; testServers = createTestLocalityMap(serverIndexes, @@ -869,7 +873,7 @@ int testReplication() { } if (g_replicationdebug >= 0) printf("Succeeded in completing %d of %d policies\n", testCounter - totalErrors, totalTests); - if ((g_replicationdebug > 0) || ((reportCacheEnv) && (atoi(reportCacheEnv) > 0))) { + if ((g_replicationdebug > 0) || parseEnvInt(reportCacheEnv, 0) > 0) { testServers->cacheReport(); } @@ -881,7 +885,7 @@ void filterLocalityDataForPolicy(const std::set& keys, LocalityData for (auto iter = ld->_data.begin(); iter != ld->_data.end();) { auto prev = iter; iter++; - if (keys.find(prev->first.toString()) == keys.end()) { + if (!keys.contains(prev->first.toString())) { ld->_data.erase(prev); } } diff --git a/fdbrpc/SimExternalConnection.cpp b/fdbrpc/SimExternalConnection.cpp index 4594350a5ea..66233496d9f 100644 --- a/fdbrpc/SimExternalConnection.cpp +++ b/fdbrpc/SimExternalConnection.cpp @@ -51,6 +51,8 @@ class SimExternalConnectionImpl { const bool wasNonBlocking = self->socket.non_blocking(); boost::system::error_code err; if (!wasNonBlocking) { + // The error is checked through the output parameter. + // NOLINTNEXTLINE(bugprone-unused-return-value) self->socket.non_blocking(true, err); if (err) { throw connection_failed(); @@ -62,6 +64,8 @@ class SimExternalConnectionImpl { boost::system::error_code restoreErr; if (!wasNonBlocking) { + // The error is checked through the output parameter. + // NOLINTNEXTLINE(bugprone-unused-return-value) self->socket.non_blocking(false, restoreErr); } if (restoreErr || (err && err != boost::asio::error::would_block && err != boost::asio::error::try_again)) { @@ -94,6 +98,8 @@ class SimExternalConnectionImpl { address = boost::asio::ip::address_v4(ip.toV4()); } boost::system::error_code err; + // The error is checked through the output parameter. + // NOLINTNEXTLINE(bugprone-unused-return-value) socket.connect(ip::tcp::endpoint(address, toAddr.port), err); if (err) { co_return Reference(); diff --git a/fdbserver/SimulatedCluster.cpp b/fdbserver/SimulatedCluster.cpp index 090b41777bb..1ddefa72aa8 100644 --- a/fdbserver/SimulatedCluster.cpp +++ b/fdbserver/SimulatedCluster.cpp @@ -28,6 +28,7 @@ #include +#include "flow/ParseNumber.h" #include "fdbclient/DatabaseConfiguration.h" #include "fdbclient/FDBTypes.h" #include "fdbrpc/Locality.h" @@ -382,20 +383,19 @@ class TestConfig : public BasicTestConfig { } if (attrib == "extraDatabaseCount") { - sscanf(value.c_str(), "%d", &extraDatabaseCount); + extraDatabaseCount = parseNumberPrefix(StringRef(value)).orDefault(extraDatabaseCount); } if (attrib == "minimumReplication") { - sscanf(value.c_str(), "%d", &minimumReplication); + minimumReplication = parseNumberPrefix(StringRef(value)).orDefault(minimumReplication); } if (attrib == "minimumRegions") { - sscanf(value.c_str(), "%d", &minimumRegions); + minimumRegions = parseNumberPrefix(StringRef(value)).orDefault(minimumRegions); } if (attrib == "configureLocked") { - int configureLockedInt; - sscanf(value.c_str(), "%d", &configureLockedInt); + int configureLockedInt = parseNumberPrefix(StringRef(value)).orDefault(0); configureLocked = (configureLockedInt != 0); } @@ -404,7 +404,7 @@ class TestConfig : public BasicTestConfig { } if (attrib == "logAntiQuorum") { - sscanf(value.c_str(), "%d", &logAntiQuorum); + logAntiQuorum = parseNumberPrefix(StringRef(value)).orDefault(logAntiQuorum); } if (attrib == "storageEngineExcludeTypes") { @@ -418,7 +418,7 @@ class TestConfig : public BasicTestConfig { } } if (attrib == "maxTLogVersion") { - sscanf(value.c_str(), "%d", &maxTLogVersion); + maxTLogVersion = parseNumberPrefix(StringRef(value)).orDefault(maxTLogVersion); } if (attrib == "disableTss") { disableTss = strcmp(value.c_str(), "true") == 0; @@ -442,10 +442,12 @@ class TestConfig : public BasicTestConfig { longRunningTest = strcmp(value.c_str(), "true") == 0; } if (attrib == "simulationNormalRunTestsTimeoutSeconds") { - sscanf(value.c_str(), "%d", &simulationNormalRunTestsTimeoutSeconds); + simulationNormalRunTestsTimeoutSeconds = + parseNumberPrefix(StringRef(value)).orDefault(simulationNormalRunTestsTimeoutSeconds); } if (attrib == "simulationBuggifyRunTestsTimeoutSeconds") { - sscanf(value.c_str(), "%d", &simulationBuggifyRunTestsTimeoutSeconds); + simulationBuggifyRunTestsTimeoutSeconds = + parseNumberPrefix(StringRef(value)).orDefault(simulationBuggifyRunTestsTimeoutSeconds); } } @@ -665,8 +667,8 @@ Future runDr(Reference connRecord) { .detail("ConnectionString", connRecord->getConnectionString().toString()) .detail("ExtraString", fdbSimulationPolicyState().extraDatabases[0]); - DatabaseBackupAgent dbAgent = DatabaseBackupAgent(cx); - DatabaseBackupAgent extraAgent = DatabaseBackupAgent(drDatabase); + DatabaseBackupAgent dbAgent(cx); + DatabaseBackupAgent extraAgent(drDatabase); auto drPollDelay = 1.0 / CLIENT_KNOBS->BACKUP_AGGREGATE_POLL_RATE; @@ -1360,19 +1362,27 @@ Future restartSimulatedSystem(std::vector>* systemActors, // allows multiple ipAddr entries ini.SetMultiKey(); + auto parseRestartInteger = [](const char* text) { + Optional value = text == nullptr ? Optional() : parseNumberPrefix(StringRef(text)); + if (!value.present()) { + throw test_specification_invalid(); + } + return value.get(); + }; + try { - int machineCount = atoi(ini.GetValue("META", "machineCount")); - int processesPerMachine = atoi(ini.GetValue("META", "processesPerMachine")); + int machineCount = parseRestartInteger(ini.GetValue("META", "machineCount")); + int processesPerMachine = parseRestartInteger(ini.GetValue("META", "processesPerMachine")); int listenersPerProcess = 1; auto listenersPerProcessStr = ini.GetValue("META", "listenersPerProcess"); if (listenersPerProcessStr != nullptr) { - listenersPerProcess = atoi(listenersPerProcessStr); + listenersPerProcess = parseRestartInteger(listenersPerProcessStr); } - int desiredCoordinators = atoi(ini.GetValue("META", "desiredCoordinators")); - int testerCount = atoi(ini.GetValue("META", "testerCount")); + int desiredCoordinators = parseRestartInteger(ini.GetValue("META", "desiredCoordinators")); + int testerCount = parseRestartInteger(ini.GetValue("META", "testerCount")); auto tssModeStr = ini.GetValue("META", "tssMode"); if (tssModeStr != nullptr) { - fdbSimulationPolicyState().tssMode = static_cast(atoi(tssModeStr)); + fdbSimulationPolicyState().tssMode = static_cast(parseRestartInteger(tssModeStr)); } ClusterConnectionString conn(ini.GetValue("META", "connectionString")); if (testConfig->extraDatabaseMode == FDBExtraDatabaseMode::Local) { @@ -1412,7 +1422,8 @@ Future restartSimulatedSystem(std::vector>* systemActors, zoneId = Standalone(zoneIdStr); } - auto cType = static_cast(atoi(ini.GetValue(machineIdString.c_str(), "mClass"))); + auto cType = static_cast( + parseRestartInteger(ini.GetValue(machineIdString.c_str(), "mClass"))); // using specialized class types can lead to nondeterministic recruitment if (cType == ProcessClass::MasterClass || cType == ProcessClass::ResolutionClass) { cType = ProcessClass::StatelessClass; @@ -1424,7 +1435,7 @@ Future restartSimulatedSystem(std::vector>* systemActors, } std::vector ipAddrs; - int processes = atoi(ini.GetValue(machineIdString.c_str(), "processes")); + int processes = parseRestartInteger(ini.GetValue(machineIdString.c_str(), "processes")); auto ip = ini.GetValue(machineIdString.c_str(), "ipAddr"); @@ -1731,8 +1742,7 @@ SimulationStorageEngine chooseSimulationStorageEngine(const TestConfig& testConf } } else if (SERVER_KNOBS->ENFORCE_SHARDED_ROCKSDB_SIM_IF_AVALIABLE && - testConfig.storageEngineExcludeTypes.find(SimulationStorageEngine::SHARDED_ROCKSDB) == - testConfig.storageEngineExcludeTypes.end()) { + !testConfig.storageEngineExcludeTypes.contains(SimulationStorageEngine::SHARDED_ROCKSDB)) { reason = "ENFORCE_SHARDED_ROCKSDB_SIM_IF_AVALIABLE is enabled"_sr; result = SimulationStorageEngine::SHARDED_ROCKSDB; diff --git a/fdbserver/backupworker/RangePartitionedBackupWorker.cpp b/fdbserver/backupworker/RangePartitionedBackupWorker.cpp index e28c14b86ce..991ed4e79bf 100644 --- a/fdbserver/backupworker/RangePartitionedBackupWorker.cpp +++ b/fdbserver/backupworker/RangePartitionedBackupWorker.cpp @@ -391,8 +391,10 @@ Future pullPartitionMapFromTLog(RangePartitionedBackupData* self, Parti // Persist the (epoch, version) -> PartitionMap row to SS so older epoch backup workers can read it during // recovery. Multiple workers may call this concurrently for the same (epoch, version) but only one succeed in writing // to SS. +// The partition map is serialized into owned storage before suspension. Future persistPartitionMapToSS(RangePartitionedBackupData* self, Version partitionMapVersion, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) PartitionMap const& partitionMap) { auto tr = makeReference(self->cx); Key key = backupPartitionMapHistoryKeyFor(self->backupEpoch, partitionMapVersion); @@ -526,16 +528,20 @@ Future uploadPartitionList(RangePartitionedBackupData* self, PartitionMap // Persists partitionMap to SS history (so that catch-up backup workers can find it during recovery) and writes the // partitionId_keyRange_Map file for every active backup container. +// The partition-map owner awaits both persistence and upload before releasing it. Future persistAndUploadPartitionMap(RangePartitionedBackupData* self, Version pmVersion, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) PartitionMap const& partitionMap) { co_await persistPartitionMapToSS(self, pmVersion, partitionMap); co_await uploadPartitionList(self, partitionMap); } // Updates local routing state to use the new partition map. +// The partition map is read only before suspension. Future setActivePartitionMap(RangePartitionedBackupData* self, Version pmVersion, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) PartitionMap const& partitionMap) { self->logFolderBaseVersion = pmVersion; ASSERT(partitionMap.contains(self->tag)); diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index 92cf4ec3531..b2b75123498 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -1289,15 +1289,19 @@ Future CDCProxy::rotateContendedPeek() { co_await delay(SERVER_KNOBS->BLOCKING_PEEK_TIMEOUT); } -Future CDCProxy::materializeBufferSelection(Reference tag, - Reference cursor, - Version throughVersion, - CDCBufferSelection const& selection, - int64_t rawPeekReservation, - FlowLock::Releaser& reservation, - int64_t bufferLimit, - Prefetch prefetch, - Future invalidated) { +// The awaited buffer pass retains its selection and mutable permit reservation. +Future CDCProxy::materializeBufferSelection( + Reference tag, + Reference cursor, + Version throughVersion, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) + CDCBufferSelection const& selection, + int64_t rawPeekReservation, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) + FlowLock::Releaser& reservation, + int64_t bufferLimit, + Prefetch prefetch, + Future invalidated) { const int64_t materializationReservation = reservation.remaining - rawPeekReservation; ASSERT_GE(materializationReservation, 0); if (selection.selectedBytes <= materializationReservation) { diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index 732365836e0..bfd2be72bb5 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -579,6 +579,8 @@ Future monitorAndRecruitLogRouters(ClusterControllerData* self) { } } +// Proxy endpoints are copied into owned failure futures before suspension. +// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future> monitorCDCProxies(std::vector const& cdcProxies) { std::vector> failures; failures.reserve(cdcProxies.size()); @@ -614,9 +616,12 @@ bool containsCDCProxy(std::vector const& proxies, UID proxyId proxies.begin(), proxies.end(), [proxyId](CDCProxyInterface const& proxy) { return proxy.id() == proxyId; }); } +// The recruitment loop retains both snapshots until this awaited replacement pass finishes. Future recruitFailedCDCProxies(ClusterControllerData* self, uint64_t recoveryCount, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::vector const& monitoredProxies, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::vector const& failedIndexes) { if (!self->db.recoveryData.isValid() || self->db.recoveryData->cstate.myDBState.recoveryCount != recoveryCount) { co_return; @@ -3250,10 +3255,8 @@ Future workerHealthMonitor(ClusterControllerData* self) { // recovered. bool hasRecoveredServer = false; for (auto it = self->excludedDegradedServers.begin(); it != self->excludedDegradedServers.end();) { - if (self->degradationInfo.degradedServers.find(it->first) == - self->degradationInfo.degradedServers.end() && - self->degradationInfo.disconnectedServers.find(it->first) == - self->degradationInfo.disconnectedServers.end()) { + if (!self->degradationInfo.degradedServers.contains(it->first) && + !self->degradationInfo.disconnectedServers.contains(it->first)) { self->excludedDegradedServers.erase(it++); hasRecoveredServer = true; } else { @@ -4450,17 +4453,17 @@ TEST_CASE("/fdbserver/clustercontroller/updateWorkerHealth") { req.disconnectedPeers.push_back(badPeer1); req.disconnectedPeers.push_back(badPeer2); data.updateWorkerHealth(req); - ASSERT(data.workerHealth.find(workerAddress) != data.workerHealth.end()); + ASSERT(data.workerHealth.contains(workerAddress)); auto& health = data.workerHealth[workerAddress]; ASSERT_EQ(health.degradedPeers.size(), 2); - ASSERT(health.degradedPeers.find(badPeer1) != health.degradedPeers.end()); + ASSERT(health.degradedPeers.contains(badPeer1)); ASSERT_EQ(health.degradedPeers[badPeer1].startTime, health.degradedPeers[badPeer1].lastRefreshTime); - ASSERT(health.degradedPeers.find(badPeer2) != health.degradedPeers.end()); + ASSERT(health.degradedPeers.contains(badPeer2)); ASSERT_EQ(health.degradedPeers[badPeer2].startTime, health.degradedPeers[badPeer2].lastRefreshTime); ASSERT_EQ(health.disconnectedPeers.size(), 2); - ASSERT(health.disconnectedPeers.find(badPeer1) != health.disconnectedPeers.end()); + ASSERT(health.disconnectedPeers.contains(badPeer1)); ASSERT_EQ(health.disconnectedPeers[badPeer1].startTime, health.disconnectedPeers[badPeer1].lastRefreshTime); - ASSERT(health.disconnectedPeers.find(badPeer2) != health.disconnectedPeers.end()); + ASSERT(health.disconnectedPeers.contains(badPeer2)); ASSERT_EQ(health.disconnectedPeers[badPeer2].startTime, health.disconnectedPeers[badPeer2].lastRefreshTime); } @@ -4479,23 +4482,23 @@ TEST_CASE("/fdbserver/clustercontroller/updateWorkerHealth") { req.disconnectedPeers.push_back(badPeer1); req.disconnectedPeers.push_back(badPeer3); data.updateWorkerHealth(req); - ASSERT(data.workerHealth.find(workerAddress) != data.workerHealth.end()); + ASSERT(data.workerHealth.contains(workerAddress)); auto& health = data.workerHealth[workerAddress]; ASSERT_EQ(health.degradedPeers.size(), 3); - ASSERT(health.degradedPeers.find(badPeer1) != health.degradedPeers.end()); + ASSERT(health.degradedPeers.contains(badPeer1)); ASSERT_LT(health.degradedPeers[badPeer1].startTime, health.degradedPeers[badPeer1].lastRefreshTime); - ASSERT(health.degradedPeers.find(badPeer2) != health.degradedPeers.end()); + ASSERT(health.degradedPeers.contains(badPeer2)); ASSERT_EQ(health.degradedPeers[badPeer2].startTime, health.degradedPeers[badPeer2].lastRefreshTime); ASSERT_EQ(health.degradedPeers[badPeer2].startTime, health.degradedPeers[badPeer1].startTime); - ASSERT(health.degradedPeers.find(badPeer3) != health.degradedPeers.end()); + ASSERT(health.degradedPeers.contains(badPeer3)); ASSERT_EQ(health.degradedPeers[badPeer3].startTime, health.degradedPeers[badPeer3].lastRefreshTime); ASSERT_EQ(health.disconnectedPeers.size(), 3); - ASSERT(health.disconnectedPeers.find(badPeer1) != health.disconnectedPeers.end()); + ASSERT(health.disconnectedPeers.contains(badPeer1)); ASSERT_LT(health.disconnectedPeers[badPeer1].startTime, health.disconnectedPeers[badPeer1].lastRefreshTime); - ASSERT(health.disconnectedPeers.find(badPeer2) != health.disconnectedPeers.end()); + ASSERT(health.disconnectedPeers.contains(badPeer2)); ASSERT_EQ(health.disconnectedPeers[badPeer2].startTime, health.disconnectedPeers[badPeer2].lastRefreshTime); ASSERT_EQ(health.disconnectedPeers[badPeer2].startTime, health.disconnectedPeers[badPeer1].startTime); - ASSERT(health.disconnectedPeers.find(badPeer3) != health.disconnectedPeers.end()); + ASSERT(health.disconnectedPeers.contains(badPeer3)); ASSERT_EQ(health.disconnectedPeers[badPeer3].startTime, health.disconnectedPeers[badPeer3].lastRefreshTime); previousStartTime = health.degradedPeers[badPeer3].startTime; @@ -4509,14 +4512,14 @@ TEST_CASE("/fdbserver/clustercontroller/updateWorkerHealth") { UpdateWorkerHealthRequest req; req.address = workerAddress; data.updateWorkerHealth(req); - ASSERT(data.workerHealth.find(workerAddress) != data.workerHealth.end()); + ASSERT(data.workerHealth.contains(workerAddress)); auto& health = data.workerHealth[workerAddress]; ASSERT_EQ(health.degradedPeers.size(), 3); - ASSERT(health.degradedPeers.find(badPeer3) != health.degradedPeers.end()); + ASSERT(health.degradedPeers.contains(badPeer3)); ASSERT_EQ(health.degradedPeers[badPeer3].startTime, previousStartTime); ASSERT_EQ(health.degradedPeers[badPeer3].lastRefreshTime, previousRefreshTime); ASSERT_EQ(health.disconnectedPeers.size(), 3); - ASSERT(health.disconnectedPeers.find(badPeer3) != health.disconnectedPeers.end()); + ASSERT(health.disconnectedPeers.contains(badPeer3)); ASSERT_EQ(health.disconnectedPeers[badPeer3].startTime, previousStartTime); ASSERT_EQ(health.disconnectedPeers[badPeer3].lastRefreshTime, previousRefreshTime); } @@ -4529,8 +4532,8 @@ TEST_CASE("/fdbserver/clustercontroller/updateWorkerHealth") { req.recoveredPeers.push_back(badPeer1); data.updateWorkerHealth(req); auto& health = data.workerHealth[workerAddress]; - ASSERT(health.degradedPeers.find(badPeer1) == health.degradedPeers.end()); - ASSERT(health.disconnectedPeers.find(badPeer1) == health.disconnectedPeers.end()); + ASSERT(!health.degradedPeers.contains(badPeer1)); + ASSERT(!health.disconnectedPeers.contains(badPeer1)); } } @@ -4574,12 +4577,11 @@ TEST_CASE("/fdbserver/clustercontroller/updateRecoveredWorkers") { data.updateRecoveredWorkers(); ASSERT_EQ(data.workerHealth.size(), 1); - ASSERT(data.workerHealth.find(worker1) != data.workerHealth.end()); - ASSERT(data.workerHealth[worker1].degradedPeers.find(badPeer1) != data.workerHealth[worker1].degradedPeers.end()); - ASSERT(data.workerHealth[worker1].degradedPeers.find(badPeer2) == data.workerHealth[worker1].degradedPeers.end()); - ASSERT(data.workerHealth[worker1].degradedPeers.find(disconnectedPeer3) != - data.workerHealth[worker1].degradedPeers.end()); - ASSERT(data.workerHealth.find(worker2) == data.workerHealth.end()); + ASSERT(data.workerHealth.contains(worker1)); + ASSERT(data.workerHealth[worker1].degradedPeers.contains(badPeer1)); + ASSERT(!data.workerHealth[worker1].degradedPeers.contains(badPeer2)); + ASSERT(data.workerHealth[worker1].degradedPeers.contains(disconnectedPeer3)); + ASSERT(!data.workerHealth.contains(worker2)); return Void(); } @@ -4617,7 +4619,7 @@ TEST_CASE("/fdbserver/clustercontroller/getDegradationInfo") { now() }; auto degradationInfo = data.getDegradationInfo(); ASSERT(degradationInfo.degradedServers.size() == 1); - ASSERT(degradationInfo.degradedServers.find(badPeer1) != degradationInfo.degradedServers.end()); + ASSERT(degradationInfo.degradedServers.contains(badPeer1)); ASSERT(degradationInfo.disconnectedServers.empty()); data.workerHealth.clear(); } @@ -4629,7 +4631,7 @@ TEST_CASE("/fdbserver/clustercontroller/getDegradationInfo") { now() }; auto degradationInfo = data.getDegradationInfo(); ASSERT(degradationInfo.disconnectedServers.size() == 1); - ASSERT(degradationInfo.disconnectedServers.find(badPeer1) != degradationInfo.disconnectedServers.end()); + ASSERT(degradationInfo.disconnectedServers.contains(badPeer1)); ASSERT(degradationInfo.degradedServers.empty()); data.workerHealth.clear(); } @@ -4647,11 +4649,10 @@ TEST_CASE("/fdbserver/clustercontroller/getDegradationInfo") { now() }; auto degradationInfo = data.getDegradationInfo(); ASSERT(degradationInfo.degradedServers.size() == 1); - ASSERT(degradationInfo.degradedServers.find(worker) != degradationInfo.degradedServers.end() || - degradationInfo.degradedServers.find(badPeer1) != degradationInfo.degradedServers.end()); + ASSERT(degradationInfo.degradedServers.contains(worker) || degradationInfo.degradedServers.contains(badPeer1)); ASSERT(degradationInfo.disconnectedServers.size() == 1); - ASSERT(degradationInfo.disconnectedServers.find(worker) != degradationInfo.disconnectedServers.end() || - degradationInfo.disconnectedServers.find(badPeer2) != degradationInfo.disconnectedServers.end()); + ASSERT(degradationInfo.disconnectedServers.contains(worker) || + degradationInfo.disconnectedServers.contains(badPeer2)); data.workerHealth.clear(); } @@ -4678,9 +4679,9 @@ TEST_CASE("/fdbserver/clustercontroller/getDegradationInfo") { now() }; auto degradationInfo = data.getDegradationInfo(); ASSERT(degradationInfo.degradedServers.size() == 1); - ASSERT(degradationInfo.degradedServers.find(worker) != degradationInfo.degradedServers.end()); + ASSERT(degradationInfo.degradedServers.contains(worker)); ASSERT(degradationInfo.disconnectedServers.size() == 1); - ASSERT(degradationInfo.disconnectedServers.find(worker) != degradationInfo.disconnectedServers.end()); + ASSERT(degradationInfo.disconnectedServers.contains(worker)); data.workerHealth.clear(); } @@ -4711,8 +4712,7 @@ TEST_CASE("/fdbserver/clustercontroller/getDegradationInfo") { data.workerHealth[badPeer4].disconnectedPeers[worker] = { now() - SERVER_KNOBS->CC_MIN_DEGRADATION_INTERVAL - 1, now() }; ASSERT(data.getDegradationInfo().disconnectedServers.size() == 1); - ASSERT(data.getDegradationInfo().disconnectedServers.find(worker) != - data.getDegradationInfo().disconnectedServers.end()); + ASSERT(data.getDegradationInfo().disconnectedServers.contains(worker)); data.workerHealth.clear(); } diff --git a/fdbserver/clustercontroller/ClusterHealthIFactor.cpp b/fdbserver/clustercontroller/ClusterHealthIFactor.cpp index de593105438..a4279eb5467 100644 --- a/fdbserver/clustercontroller/ClusterHealthIFactor.cpp +++ b/fdbserver/clustercontroller/ClusterHealthIFactor.cpp @@ -42,8 +42,8 @@ WorkerEvents filterEmptyEvents(WorkerEvents const& events) { } Future fetchSpaceLevel(LatestWorkerEvents eventsAndErrors, - std::string const& availableBytesField, - std::string const& totalBytesField, + std::string availableBytesField, + std::string totalBytesField, double interventionThreshold, double criticalInterventionThreshold, char const* failureTraceEventName) { diff --git a/fdbserver/clustercontroller/ClusterHealthMonitor.cpp b/fdbserver/clustercontroller/ClusterHealthMonitor.cpp index c3e8fa7cd8d..57baddf2a84 100644 --- a/fdbserver/clustercontroller/ClusterHealthMonitor.cpp +++ b/fdbserver/clustercontroller/ClusterHealthMonitor.cpp @@ -227,6 +227,8 @@ AsyncResult WorkerEventProvider::getLatestEvents(std::string return latestEventOnWorkers(workers, eventName); } +// The event name is copied into the request helper before suspension. +// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) AsyncResult WorkerEventProvider::getLatestRatekeeperEvents(std::string const& eventName) const { if (!ratekeeperWorker.present()) { co_return LatestWorkerEvents(); @@ -234,7 +236,9 @@ AsyncResult WorkerEventProvider::getLatestRatekeeperEvents(s co_return co_await latestEventOnWorker(ratekeeperWorker.get(), eventName); } +// The event name is copied into the request helper before suspension. AsyncResult WorkerEventProvider::getLatestDataDistributorEvents( + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::string const& eventName) const { if (!dataDistributorWorker.present()) { co_return LatestWorkerEvents(); diff --git a/fdbserver/clustercontroller/ClusterHealthMonitorTesting.cpp b/fdbserver/clustercontroller/ClusterHealthMonitorTesting.cpp index 296de920996..10c8d0c1c55 100644 --- a/fdbserver/clustercontroller/ClusterHealthMonitorTesting.cpp +++ b/fdbserver/clustercontroller/ClusterHealthMonitorTesting.cpp @@ -74,6 +74,8 @@ class FakeWorkerEventProvider final : public IWorkerEventProvider, public Refere latestTLogEventsByName[std::move(eventName)] = std::move(latestEvents); } + // The fake provider reads the name and produces its result without suspending. + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) AsyncResult getLatestEvents(std::string const& eventName) const override { auto it = latestEventsByName.find(eventName); if (it == latestEventsByName.end()) { @@ -88,6 +90,8 @@ class FakeWorkerEventProvider final : public IWorkerEventProvider, public Refere AsyncResult> areAllCoordinatorsReachable() const override { co_return allCoordinatorsReachable; } + // The fake provider reads the name and produces its result without suspending. + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) AsyncResult getLatestRatekeeperEvents(std::string const& eventName) const override { auto it = latestRatekeeperEventsByName.find(eventName); if (it != latestRatekeeperEventsByName.end()) { @@ -96,6 +100,8 @@ class FakeWorkerEventProvider final : public IWorkerEventProvider, public Refere co_return co_await getLatestEvents(eventName); } + // The fake provider reads the name and produces its result without suspending. + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) AsyncResult getLatestDataDistributorEvents(std::string const& eventName) const override { auto it = latestDataDistributorEventsByName.find(eventName); if (it != latestDataDistributorEventsByName.end()) { @@ -104,6 +110,8 @@ class FakeWorkerEventProvider final : public IWorkerEventProvider, public Refere co_return co_await getLatestEvents(eventName); } + // The fake provider reads the name and produces its result without suspending. + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) AsyncResult getLatestStorageServerEvents(std::string const& eventName) const override { auto it = latestStorageServerEventsByName.find(eventName); if (it == latestStorageServerEventsByName.end()) { @@ -112,6 +120,8 @@ class FakeWorkerEventProvider final : public IWorkerEventProvider, public Refere co_return it->second; } + // The fake provider reads the name and produces its result without suspending. + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) AsyncResult getLatestTLogEvents(std::string const& eventName) const override { auto it = latestTLogEventsByName.find(eventName); if (it == latestTLogEventsByName.end()) { diff --git a/fdbserver/clustercontroller/Status.cpp b/fdbserver/clustercontroller/Status.cpp index cd91c939c5c..bdfdad9938c 100644 --- a/fdbserver/clustercontroller/Status.cpp +++ b/fdbserver/clustercontroller/Status.cpp @@ -18,6 +18,7 @@ * limitations under the License. */ +#include "flow/ParseNumber.h" #include #include "fdbclient/json_spirit/json_spirit_value.h" #include "flow/genericactors.h" @@ -181,7 +182,20 @@ class StatusCounter { explicit(false) StatusCounter(const std::string& parsableText) { parseText(parsableText); } StatusCounter& parseText(const std::string& parsableText) { - sscanf(parsableText.c_str(), "%lf %lf %" SCNd64 "", &hz, &roughness, &counter); + StringRef remaining(parsableText); + int consumed = 0; + auto parsedHz = parseNumberPrefix(remaining, 10, &consumed); + remaining = remaining.substr(consumed); + consumed = 0; + auto parsedRoughness = parseNumberPrefix(remaining, 10, &consumed); + remaining = remaining.substr(consumed); + auto parsedCounter = parseNumberPrefix(remaining); + if (!parsedHz.present() || !parsedRoughness.present() || !parsedCounter.present()) { + throw attribute_not_found(); + } + hz = parsedHz.get(); + roughness = parsedRoughness.get(); + counter = parsedCounter.get(); return *this; } @@ -217,11 +231,32 @@ class StatusCounter { int64_t counter; }; +TEST_CASE("/status/counterParsing") { + StatusCounter counter("1.25 2.5 42"); + ASSERT_EQ(counter.getHz(), 1.25); + ASSERT_EQ(counter.getRoughness(), 2.5); + ASSERT_EQ(counter.getCounter(), 42); + for (const char* text : { "", "1.25 2.5", "1.25 x 42", "1.25 2.5 9223372036854775808" }) { + bool rejected = false; + try { + counter.parseText(text); + } catch (Error& e) { + ASSERT_EQ(e.code(), error_code_attribute_not_found); + rejected = true; + } + ASSERT(rejected); + ASSERT_EQ(counter.getHz(), 1.25); + ASSERT_EQ(counter.getRoughness(), 2.5); + ASSERT_EQ(counter.getCounter(), 42); + } + return Void(); +} + static JsonBuilderObject getError(const TraceEventFields& errorFields) { JsonBuilderObject statusObj; try { if (errorFields.size()) { - double time = atof(errorFields.getValue("Time").c_str()); + double time = errorFields.getDouble("Time"); statusObj["time"] = time; statusObj["raw_log_message"] = errorFields.toString(); @@ -1214,10 +1249,10 @@ static AsyncResult recoveryStateStatusFetcher(Database cx, // Add additional metadata for certain statuses if (mStatusCode == RecoveryStatus::recruiting_transaction_servers) { - int requiredLogs = atoi(md.getValue("RequiredTLogs").c_str()); - int requiredCommitProxies = atoi(md.getValue("RequiredCommitProxies").c_str()); - int requiredGrvProxies = atoi(md.getValue("RequiredGrvProxies").c_str()); - int requiredResolvers = atoi(md.getValue("RequiredResolvers").c_str()); + int requiredLogs = md.getInt("RequiredTLogs"); + int requiredCommitProxies = md.getInt("RequiredCommitProxies"); + int requiredGrvProxies = md.getInt("RequiredGrvProxies"); + int requiredResolvers = md.getInt("RequiredResolvers"); // int requiredProcesses = std::max(requiredLogs, std::max(requiredResolvers, requiredCommitProxies)); // int requiredMachines = std::max(requiredLogs, 1); diff --git a/fdbserver/commitproxy/CommitProxyServer.cpp b/fdbserver/commitproxy/CommitProxyServer.cpp index 5b2ecc821eb..1ae3e695eae 100644 --- a/fdbserver/commitproxy/CommitProxyServer.cpp +++ b/fdbserver/commitproxy/CommitProxyServer.cpp @@ -18,6 +18,7 @@ * limitations under the License. */ +#include "flow/ParseNumber.h" #include #include #include @@ -2495,7 +2496,7 @@ Future proxySnapCreate(ProxySnapRequest snapReq, ProxyCommitData* commitDa auto result = commitData->txnStateStore->readValue("log_anti_quorum"_sr.withPrefix(configKeysPrefix)).get(); int logAntiQuorum = 0; if (result.present()) { - logAntiQuorum = atoi(result.get().toString().c_str()); + logAntiQuorum = parseNumberPrefix(result.get()).orDefault(0); } // FIXME: logAntiQuorum not supported, remove it later, // In version2, we probably don't need this limitation, but this needs to be tested. diff --git a/fdbserver/core/MoveKeys.cpp b/fdbserver/core/MoveKeys.cpp index 029d0bf44a4..71746314e20 100644 --- a/fdbserver/core/MoveKeys.cpp +++ b/fdbserver/core/MoveKeys.cpp @@ -257,7 +257,7 @@ Future deleteCheckpoints(Transaction* tr, std::set checkpointIds, UID continue; } CheckpointMetaData checkpoint = decodeCheckpointValue(value.get()); - ASSERT(checkpointIds.find(checkpoint.checkpointID) != checkpointIds.end()); + ASSERT(checkpointIds.contains(checkpoint.checkpointID)); const Key key = checkpointKeyFor(checkpoint.checkpointID); checkpoint.setState(CheckpointMetaData::Deleting); tr->set(key, checkpointValue(checkpoint)); @@ -1610,9 +1610,12 @@ static Optional decodeKeyServersState(RangeResult const& // owns the FlowLock slot). Returns the interfaces plus the read version at // which they were fetched — the read version is what waitForShardReady() // needs (see finishMoveKeys where we save it before dropping the txn). +// Both server lists are consumed into owned containers before suspension. static Future, Version>> buildKeysDestServerInterfaces( Transaction* tr, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::vector const& dest, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::vector const& completeSrc, bool hasRemote) { std::set completeSrcSet(completeSrc.begin(), completeSrc.end()); @@ -1648,10 +1651,14 @@ static Future, Version>> buildKeys // populated) when only TSS is slow, so subsequent iterations can skip it. // Returns `destSize - (SSes not ready)`; the caller retries if not equal to // `destSize`. `keys` is only used for tracing. +// The pending move retains these snapshots until this awaited phase finishes. +// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) static Future waitForKeysDestServers(std::vector const& storageServerInterfaces, int destSize, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) KeyRange const& keys, Version readVersion, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::map const& tssMapping, int* waitForTSSCounter, std::unordered_set* tssToIgnore, @@ -1746,13 +1753,17 @@ static Future waitForKeysDestServers(std::vector co // change during the wait; the caller retries via retryAfterPostWaitChange(). // `currentKeys` and `endKey` are in/out because a KRM boundary re-truncation // can shorten them. +// The pending move retains these snapshots until this awaited phase finishes. static Future reverifyKeysDestAndCommit(Transaction* tr, MoveKeysLock lock, const DDEnabledState* ddEnabledState, KeyRange* currentKeys, Key* endKey, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::vector const& dest, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::set const& allServers, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) KeyRange const& keys, UID relocationIntervalId, FinishMoveRetryBudget* retryBudget, @@ -2504,17 +2515,23 @@ struct DecodedShardsKeyServers { // running the per-sub-range AUDIT_DATAMOVE_PRE_CHECK when enabled. On a // stamp mismatch, sets *cancelDataMove=true and throws retry() so the outer // loop enters the cancel path. `dataMove` is only used for tracing. -static Future decodeAndPreCheckShards(Database occ, - Transaction* tr, - RangeResult const& UIDtoTagMap, - RangeResult const& keyServers, - std::vector const& destServers, - UID dataMoveId, - bool runPreCheck, - DataMoveMetaData const& dataMove, - UID relocationIntervalId, - Severity sevDm, - bool* cancelDataMove) { +// The pending move retains its metadata and read arenas throughout this awaited phase. +static Future decodeAndPreCheckShards( + Database occ, + Transaction* tr, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) + RangeResult const& UIDtoTagMap, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) + RangeResult const& keyServers, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) + std::vector const& destServers, + UID dataMoveId, + bool runPreCheck, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) + DataMoveMetaData const& dataMove, + UID relocationIntervalId, + Severity sevDm, + bool* cancelDataMove) { std::vector completeSrc; std::unordered_set allServers; @@ -2576,9 +2593,12 @@ static Future decodeAndPreCheckShards(Database occ, // finishMoveShards analog of buildKeysDestServerInterfaces. Only difference: // a missing serverList entry throws retry() rather than asserting — shards // tolerates the SS-removed race by re-reading dataMove and starting over. +// Both server lists are consumed into owned containers before suspension. static Future, Version>> buildShardsDestServerInterfaces( Transaction* tr, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::vector const& destServers, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::vector const& completeSrc, bool hasRemote) { std::set completeSrcSet(completeSrc.begin(), completeSrc.end()); @@ -2622,10 +2642,15 @@ static Future, Version>> buildShar // count for the caller's ready-versus-target comparison; also fills // `readyServers_out` and `tssCount_out` for the caller's post-wait // tracing. +// The pending move retains these snapshots until this awaited phase finishes. +// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) static Future waitForShardsDestServers(std::vector const& storageServerInterfaces, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::vector const& newDestinationIds, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) KeyRange const& range, Version readVersion, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::map const& tssMapping, bool* skipTss, double* ssReadyTime, @@ -2634,6 +2659,7 @@ static Future waitForShardsDestServers(std::vector UID dataMoveId, UID relocationIntervalId, Severity sevDm, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) DataMoveMetaData const& dataMove) { std::vector> serverReady; // only for count below std::vector> tssReady; // for waiting in parallel with tss @@ -2713,22 +2739,26 @@ enum class ReverifyShardsResult { RetryLoop, PartialCommitted, FullyCommitted }; // the outer loop skips the per-sub-range AUDIT precheck on the next attempt. // `cancelDataMove` is set on bulk-load-outdated so the outer catch enters // the cancel path. -static Future reverifyShardsAndCommit(Transaction* tr, - Database occ, - MoveKeysLock lock, - const DDEnabledState* ddEnabledState, - UID dataMoveId, - std::vector const& destServers, - KeyRange* range, - std::unordered_set const& allServers, - Optional bulkLoadTaskState, - DataMoveMetaData* postWaitDataMove_out, - UID relocationIntervalId, - Severity sevDm, - FinishMoveRetryBudget* retryBudget, - bool* runPreCheck, - bool* cancelDataMove, - TxnCounters* counters) { +// The pending move retains these snapshots until this awaited phase finishes. +static Future reverifyShardsAndCommit( + Transaction* tr, + Database occ, + MoveKeysLock lock, + const DDEnabledState* ddEnabledState, + UID dataMoveId, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) + std::vector const& destServers, + KeyRange* range, + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) + std::unordered_set const& allServers, + Optional bulkLoadTaskState, + DataMoveMetaData* postWaitDataMove_out, + UID relocationIntervalId, + Severity sevDm, + FinishMoveRetryBudget* retryBudget, + bool* runPreCheck, + bool* cancelDataMove, + TxnCounters* counters) { tr->trState->taskID = TaskPriority::MoveKeys; tr->setOption(FDBTransactionOptions::PRIORITY_SYSTEM_IMMEDIATE); tr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); diff --git a/fdbserver/core/QuietDatabase.cpp b/fdbserver/core/QuietDatabase.cpp index 878ae842d40..e41e88adb15 100644 --- a/fdbserver/core/QuietDatabase.cpp +++ b/fdbserver/core/QuietDatabase.cpp @@ -18,6 +18,8 @@ * limitations under the License. */ +#include "flow/UnitTest.h" +#include "flow/ParseNumber.h" #include #include #include @@ -143,14 +145,43 @@ Future getDataInFlight(Database cx, Reference co // Computes the queue size for storage servers and tlogs using the bytesInput and bytesDurable attributes int64_t getQueueSize(const TraceEventFields& md) { - double inputRate, durableRate; - double inputRoughness, durableRoughness; - int64_t inputBytes, durableBytes; - - sscanf(md.getValue("BytesInput").c_str(), "%lf %lf %" SCNd64, &inputRate, &inputRoughness, &inputBytes); - sscanf(md.getValue("BytesDurable").c_str(), "%lf %lf %" SCNd64, &durableRate, &durableRoughness, &durableBytes); + auto parseBytes = [](const std::string& text) { + StringRef remaining(text); + for (int i = 0; i < 2; ++i) { + int consumed = 0; + if (!parseNumberPrefix(remaining, 10, &consumed).present()) { + throw attribute_not_found(); + } + remaining = remaining.substr(consumed); + } + auto bytes = parseNumberPrefix(remaining); + if (!bytes.present()) { + throw attribute_not_found(); + } + return bytes.get(); + }; + return parseBytes(md.getValue("BytesInput")) - parseBytes(md.getValue("BytesDurable")); +} - return inputBytes - durableBytes; +TEST_CASE("/fdbserver/QuietDatabase/queueCounterParsing") { + TraceEventFields fields; + fields.addField("BytesInput", "1.25 2.5 100"); + fields.addField("BytesDurable", "1.0 2.0 40"); + ASSERT_EQ(getQueueSize(fields), 60); + for (const char* text : { "", "1.0 2.0", "1.0 2.0 9223372036854775808" }) { + TraceEventFields invalid; + invalid.addField("BytesInput", text); + invalid.addField("BytesDurable", "1.0 2.0 40"); + bool rejected = false; + try { + getQueueSize(invalid); + } catch (Error& e) { + ASSERT_EQ(e.code(), error_code_attribute_not_found); + rejected = true; + } + ASSERT(rejected); + } + return Void(); } int64_t getDurableVersion(const TraceEventFields& md) { @@ -189,8 +220,8 @@ Future> getCoordWorkers(Database cx, Reference secondary = worker.interf.tLog.getEndpoint().addresses.secondaryAddress; - if (coordinatorsAddrSet.find(primary) != coordinatorsAddrSet.end() || - (secondary.present() && (coordinatorsAddrSet.find(secondary.get()) != coordinatorsAddrSet.end()))) { + if (coordinatorsAddrSet.contains(primary) || + (secondary.present() && coordinatorsAddrSet.contains(secondary.get()))) { result.push_back(worker.interf); } } @@ -285,7 +316,7 @@ Future, int>> getStorageWorkers(Database }); int usableRegions = 1; if (regionsValue.present()) { - usableRegions = atoi(regionsValue.get().toString().c_str()); + usableRegions = parseNumberPrefix(regionsValue.get()).orDefault(0); } auto masterDcId = dbInfo->get().master.locality.dcId(); diff --git a/fdbserver/core/WorkloadKeys.cpp b/fdbserver/core/WorkloadKeys.cpp index 3a165908186..b6994b8d379 100644 --- a/fdbserver/core/WorkloadKeys.cpp +++ b/fdbserver/core/WorkloadKeys.cpp @@ -1,3 +1,4 @@ +#include "flow/ParseNumber.h" #include #include @@ -9,8 +10,7 @@ Key doubleToTestKey(double p) { } double testKeyToDouble(const KeyRef& p) { - uint64_t x = 0; - sscanf(p.toString().c_str(), "%" SCNx64, &x); + uint64_t x = parseNumberPrefix(p, 16).orDefault(0); return *(double*)&x; } diff --git a/fdbserver/datadistributor/DDTxnProcessor.cpp b/fdbserver/datadistributor/DDTxnProcessor.cpp index 2fba953ddb9..3d3852ab567 100644 --- a/fdbserver/datadistributor/DDTxnProcessor.cpp +++ b/fdbserver/datadistributor/DDTxnProcessor.cpp @@ -327,6 +327,8 @@ class DDTxnProcessorImpl { // // serverKeys entries are left in place — they drain naturally as DD // moves shards using the old path. + // The initialization loop retains this transaction while awaiting metadata rewrites. + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) static Future rewriteShardEncodedMetadata(Transaction& tr, UID distributorId) { TraceEvent(SevInfo, "DDInitShardEncodeOff", distributorId) .detail("KnobValue", SERVER_KNOBS->SHARD_ENCODE_LOCATION_METADATA); diff --git a/fdbserver/datadistributor/DataDistribution.cpp b/fdbserver/datadistributor/DataDistribution.cpp index 9365b10bbfa..5c7d97ab462 100644 --- a/fdbserver/datadistributor/DataDistribution.cpp +++ b/fdbserver/datadistributor/DataDistribution.cpp @@ -18,6 +18,7 @@ * limitations under the License. */ +#include "flow/ParseNumber.h" #include #include #include @@ -2242,8 +2243,7 @@ Future scheduleBulkLoadJob(Reference self, Promise // No matter whether the task range is aligned with the manifest entry range, the task // begin key must be in the manifestEntryMap. See manifestEntryMap definition for more // details. - ASSERT(self->bulkLoadJobManager.get().manifestEntryMap->find(task.getRange().begin) != - self->bulkLoadJobManager.get().manifestEntryMap->end()); + ASSERT(self->bulkLoadJobManager.get().manifestEntryMap->contains(task.getRange().begin)); if (task.onAnyPhase( { BulkLoadPhase::Complete, BulkLoadPhase::Acknowledged, BulkLoadPhase::Error })) { ASSERT(task.getRange().end == res[i + 1].key); @@ -3455,7 +3455,7 @@ Future>> getSta Optional regionsValue = co_await tr.get("usable_regions"_sr.withPrefix(configKeysPrefix)); int usableRegions = 1; if (regionsValue.present()) { - usableRegions = atoi(regionsValue.get().toString().c_str()); + usableRegions = parseNumberPrefix(regionsValue.get()).orDefault(0); } auto masterDcId = dbInfo->get().master.locality.dcId(); int storageFailures = 0; @@ -3493,7 +3493,7 @@ Future>> getSta for (const auto& tlog : *tlogs) { TraceEvent(SevDebug, "GetStatefulWorkersTLog").detail("Addr", tlog.address()); - if (workersMap.find(tlog.address()) == workersMap.end()) { + if (!workersMap.contains(tlog.address())) { TraceEvent(SevWarn, "MissingTLogWorkerInterface").detail("TlogAddress", tlog.address()); throw snap_tlog_failed(); } @@ -3518,8 +3518,8 @@ Future>> getSta // as we use primary addresses from storage and tlog interfaces above NetworkAddress primary = worker.interf.address(); Optional secondary = worker.interf.tLog.getEndpoint().addresses.secondaryAddress; - if (coordinatorsAddrSet.find(primary) != coordinatorsAddrSet.end() || - (secondary.present() && (coordinatorsAddrSet.find(secondary.get()) != coordinatorsAddrSet.end()))) { + if (coordinatorsAddrSet.contains(primary) || + (secondary.present() && coordinatorsAddrSet.contains(secondary.get()))) { if (result.contains(primary)) { ASSERT(workersMap[primary].id() == result[primary].first.id()); result[primary].second.append(",coord"); diff --git a/fdbserver/fdbserver.cpp b/fdbserver/fdbserver.cpp index 8e4f8f37de8..7afbe781456 100644 --- a/fdbserver/fdbserver.cpp +++ b/fdbserver/fdbserver.cpp @@ -22,6 +22,7 @@ // a macro that makes boost interprocess break on Windows. #define BOOST_DATE_TIME_NO_LIB +#include "flow/ParseNumber.h" #include #include #include @@ -1340,11 +1341,13 @@ struct CLIOptions { } case OPT_NUMTESTERS: { const char* a = args.OptionArg(); - if (!sscanf(a, "%d", &minTesterCount)) { + auto parsed = parseNumberPrefix(StringRef(a)); + if (!parsed.present()) { fprintf(stderr, "ERROR: Could not parse numtesters `%s'\n", a); printHelpTeaser(argv[0]); flushAndExit(FDB_EXIT_ERROR); } + minTesterCount = parsed.get(); break; } case OPT_ROLLSIZE: { @@ -1385,8 +1388,13 @@ struct CLIOptions { } #ifdef _WIN32 case OPT_PARENTPID: { - auto pid_str = args.OptionArg(); - int parent_pid = atoi(pid_str); + const char* pid_str = args.OptionArg(); + auto parsedPid = parseNumberPrefix(StringRef(pid_str)); + if (!parsedPid.present() || parsedPid.get() <= 0) { + fprintf(stderr, "ERROR: Invalid parent process id `%s'\n", pid_str); + flushAndExit(FDB_EXIT_ERROR); + } + int parent_pid = parsedPid.get(); auto pHandle = OpenProcess(SYNCHRONIZE, FALSE, parent_pid); if (!pHandle) { TraceEvent("ParentProcessOpenError").GetLastError(); @@ -1408,9 +1416,13 @@ struct CLIOptions { break; #else case OPT_PARENTPID: { - auto pid_str = args.OptionArg(); - int* parent_pid = new (int); - *parent_pid = atoi(pid_str); + const char* pid_str = args.OptionArg(); + auto parsedPid = parseNumberPrefix(StringRef(pid_str)); + if (!parsedPid.present() || parsedPid.get() <= 0) { + fprintf(stderr, "ERROR: Invalid parent process id `%s'\n", pid_str); + flushAndExit(FDB_EXIT_ERROR); + } + int* parent_pid = new int(parsedPid.get()); startThread(&parentWatcher, parent_pid, 0, "fdb-parentwatch"); break; } @@ -1564,11 +1576,13 @@ struct CLIOptions { break; case OPT_IO_TRUST_SECONDS: { const char* a = args.OptionArg(); - if (!sscanf(a, "%lf", &fileIoTimeout)) { + auto parsed = parseNumberPrefix(StringRef(a)); + if (!parsed.present()) { fprintf(stderr, "ERROR: Could not parse io_trust_seconds `%s'\n", a); printHelpTeaser(argv[0]); flushAndExit(FDB_EXIT_ERROR); } + fileIoTimeout = parsed.get(); break; } case OPT_IO_TRUST_WARN_ONLY: @@ -2179,10 +2193,10 @@ int main(int argc, char* argv[]) { int backupFailed = true; const char* isRestoringStr = ini.GetValue("RESTORE", "isRestoring", nullptr); if (isRestoringStr) { - isRestoring = atoi(isRestoringStr); + isRestoring = parseNumberPrefix(StringRef(isRestoringStr)).orDefault(0); const char* backupFailedStr = ini.GetValue("RESTORE", "BackupFailed", nullptr); if (isRestoring && backupFailedStr) { - backupFailed = atoi(backupFailedStr); + backupFailed = parseNumberPrefix(StringRef(backupFailedStr)).orDefault(0); } } if (isRestoring && !backupFailed) { diff --git a/fdbserver/kvstore/VersionedBTree.cpp b/fdbserver/kvstore/VersionedBTree.cpp index 1a819a732d3..178035270af 100644 --- a/fdbserver/kvstore/VersionedBTree.cpp +++ b/fdbserver/kvstore/VersionedBTree.cpp @@ -11201,6 +11201,8 @@ struct KVSource { for (auto& p : prefixes) { prefixesSorted.push_back(&p); } + // The comparator orders by prefix bytes, independent of pointer values. + // NOLINTNEXTLINE(bugprone-nondeterministic-pointer-iteration-order) std::sort(prefixesSorted.begin(), prefixesSorted.end(), [](const Prefix* a, const Prefix* b) { return KeyRef((uint8_t*)a->begin(), a->size()) < KeyRef((uint8_t*)b->begin(), b->size()); }); diff --git a/fdbserver/networktest.cpp b/fdbserver/networktest.cpp index d8f6473cbc6..98d1932bd10 100644 --- a/fdbserver/networktest.cpp +++ b/fdbserver/networktest.cpp @@ -18,6 +18,7 @@ * limitations under the License. */ +#include "flow/ParseNumber.h" #include "fmt/format.h" #include "fdbserver/NetworkTest.h" #include "flow/ActorCollection.h" @@ -219,6 +220,8 @@ static void networkTestnanosleep() { return; } +// The server list is parsed into owned addresses before suspension. +// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future networkTestClient(std::string const& testServers) { if (testServers == "nanosleep") { networkTestnanosleep(); @@ -260,8 +263,8 @@ struct RandomIntRange { if (high.empty()) { high = low; } - min = low.empty() ? 0 : atol(low.toString().c_str()); - max = high.empty() ? 0 : atol(high.toString().c_str()); + min = low.empty() ? 0 : parseNumberPrefix(low).orDefault(0); + max = high.empty() ? 0 : parseNumberPrefix(high).orDefault(0); if (min > max) { std::swap(min, max); } diff --git a/fdbserver/tester/CustomShardConfigWorkload.cpp b/fdbserver/tester/CustomShardConfigWorkload.cpp index ff0feb112d6..fba7171835f 100644 --- a/fdbserver/tester/CustomShardConfigWorkload.cpp +++ b/fdbserver/tester/CustomShardConfigWorkload.cpp @@ -25,8 +25,7 @@ #include "fdbclient/DataDistributionConfig.h" #include "fdbserver/tester/tester.h" -Future customShardConfigWorkload(Database const& cxUnsafe) { - auto cx = cxUnsafe; +static Future customShardConfigWorkloadImpl(Database cx) { ReadYourWritesTransaction tr(cx); bool verbose = (KEYBACKEDTYPES_DEBUG != 0); @@ -132,3 +131,7 @@ Future customShardConfigWorkload(Database const& cxUnsafe) { co_await tr.onError(err); } } + +Future customShardConfigWorkload(Database const& cxUnsafe) { + return customShardConfigWorkloadImpl(cxUnsafe); +} diff --git a/fdbserver/tester/DatabaseMaintenance.cpp b/fdbserver/tester/DatabaseMaintenance.cpp index fe957260870..dd65b99626b 100644 --- a/fdbserver/tester/DatabaseMaintenance.cpp +++ b/fdbserver/tester/DatabaseMaintenance.cpp @@ -153,7 +153,7 @@ std::string toHTML(const StringRef& binaryString) { } // namespace -Future dumpDatabase(Database const& cx, std::string const& outputFilename, KeyRange const& range) { +static Future dumpDatabaseImpl(Database cx, std::string outputFilename, KeyRange range) { try { Transaction tr(cx); while (true) { @@ -196,6 +196,10 @@ Future dumpDatabase(Database const& cx, std::string const& outputFilename, } } +Future dumpDatabase(Database const& cx, std::string const& outputFilename, KeyRange const& range) { + return dumpDatabaseImpl(cx, outputFilename, range); +} + std::vector aggregateMetrics(std::vector> metrics) { std::map> metricMap; for (int i = 0; i < metrics.size(); i++) { diff --git a/fdbserver/tester/TestSpecParser.cpp b/fdbserver/tester/TestSpecParser.cpp index 86799505310..31528fc67a0 100644 --- a/fdbserver/tester/TestSpecParser.cpp +++ b/fdbserver/tester/TestSpecParser.cpp @@ -28,6 +28,7 @@ #include #include "flow/Platform.h" +#include "flow/ParseNumber.h" #include "flow/Trace.h" #include "flow/UnitTest.h" #include "fdbclient/NativeAPI.h" @@ -36,6 +37,15 @@ namespace { +template +T parseTestNumber(const std::string& value) { + auto parsed = parseNumberPrefix(value); + if (!parsed.present()) { + throw test_specification_invalid(); + } + return parsed.get(); +} + std::map> testSpecGlobalKeys = { // These are read by SimulatedCluster and used before testers exist. Thus, they must // be recognized and accepted, but there's no point in placing them into a testSpec. @@ -88,14 +98,13 @@ std::maptimeout)); + spec->timeout = parseTestNumber(value); ASSERT(spec->timeout > 0); TraceEvent("TestParserTest").detail("ParsedTimeout", spec->timeout); } }, { "databasePingDelay", [](const std::string& value, TestSpec* spec) { - double databasePingDelay; - sscanf(value.c_str(), "%lf", &databasePingDelay); + double databasePingDelay = parseTestNumber(value); ASSERT(databasePingDelay >= 0); if (!spec->useDB && databasePingDelay > 0) { TraceEvent(SevError, "TestParserError") @@ -133,7 +142,7 @@ std::mapstartDelay); + spec->startDelay = parseTestNumber(value); TraceEvent("TestParserTest").detail("ParsedStartDelay", spec->startDelay); } }, { "runConsistencyCheck", @@ -148,7 +157,7 @@ std::mapmaxDDRunTime)); + spec->maxDDRunTime = parseTestNumber(value); ASSERT(spec->maxDDRunTime >= 0); TraceEvent("TestParserTest").detail("ParsedMaxDDRunTime", spec->maxDDRunTime); } }, @@ -178,8 +187,7 @@ std::map(value); ASSERT(connectionFailuresDisableDuration >= 0); spec->simConnectionFailuresDisableDuration = connectionFailuresDisableDuration; TraceEvent("TestParserTest") @@ -253,6 +261,26 @@ std::string toml_to_string(const T& value) { } } +TEST_CASE("/fdbserver/tester/TestSpecParser/NumericOptions") { + TestSpec spec; + testSpecTestKeys.at("timeout")("200.0", &spec); + ASSERT_EQ(spec.timeout, 200); + testSpecTestKeys.at("startDelay")(" \t+1.25suffix", &spec); + ASSERT_EQ(spec.startDelay, 1.25); + for (const auto& [option, value] : std::vector>{ + { "timeout", "99999999999999999999" }, { "startDelay", "1e9999" }, { "databasePingDelay", "invalid" } }) { + try { + testSpecTestKeys.at(option)(value, &spec); + ASSERT(false); + } catch (Error& e) { + ASSERT_EQ(e.code(), error_code_test_specification_invalid); + } + } + ASSERT_EQ(spec.timeout, 200); + ASSERT_EQ(spec.startDelay, 1.25); + return Void(); +} + TEST_CASE("/fdbserver/tester/TestSpecParser/TOMLArrayToString") { std::istringstream input(R"( strings = ['a', 'b'] @@ -304,11 +332,11 @@ std::vector readTests(std::ifstream& ifs) { } testSpecTestKeys[attrib](value, &spec); - } else if (testSpecTestKeys.find(attrib) != testSpecTestKeys.end()) { + } else if (testSpecTestKeys.contains(attrib)) { if (parsingWorkloads) TraceEvent(SevError, "TestSpecTestParamInWorkload").detail("Attrib", attrib).detail("Value", value); testSpecTestKeys[attrib](value, &spec); - } else if (testSpecGlobalKeys.find(attrib) != testSpecGlobalKeys.end()) { + } else if (testSpecGlobalKeys.contains(attrib)) { if (!beforeFirstTest) TraceEvent(SevError, "TestSpecGlobalParamInTest").detail("Attrib", attrib).detail("Value", value); testSpecGlobalKeys[attrib](value); @@ -393,7 +421,7 @@ TestSet readTOMLTests_(std::string fileName) { if (k == "workload" || k == "knobs") { continue; } - if (testSpecTestKeys.find(k) != testSpecTestKeys.end()) { + if (testSpecTestKeys.contains(k)) { testSpecTestKeys[k](toml_to_string(v), &spec); } else { TraceEvent(SevError, "TestSpecUnrecognizedTestParam") diff --git a/fdbserver/tester/TesterServer.cpp b/fdbserver/tester/TesterServer.cpp index 86f812a67f7..52ea491c6c0 100644 --- a/fdbserver/tester/TesterServer.cpp +++ b/fdbserver/tester/TesterServer.cpp @@ -141,6 +141,8 @@ void printSimulatedTopology() { return; } auto processes = g_simulator->getAllProcesses(); + // The comparator orders by locality and network address, independent of pointer values. + // NOLINTNEXTLINE(bugprone-nondeterministic-pointer-iteration-order) std::sort(processes.begin(), processes.end(), [](ISimulator::ProcessInfo* lhs, ISimulator::ProcessInfo* rhs) { auto l = lhs->locality; auto r = rhs->locality; @@ -527,21 +529,11 @@ Future testerServerWorkload(WorkloadRequest work, } // namespace -Future testerServerCore(TesterInterface const& interf, - Reference const& ccr, - Reference const> const& dbInfo, - LocalityData const& locality, - Optional const& expectedWorkLoad) { - // C++20 coroutine safety: const& parameters only store the reference in the coroutine frame, - // not the object. The referred-to object may be destroyed after the coroutine suspends - // (e.g. local variables in a caller's if-block, or temporaries from default arguments). - // Copy all const& parameters to ensure they survive across suspend points. - TesterInterface interfCopy = interf; - Reference ccrCopy = ccr; - Reference const> dbInfoCopy = dbInfo; - LocalityData localityCopy = locality; - Optional expectedWorkLoadCopy = expectedWorkLoad; - +static Future testerServerCoreImpl(TesterInterface interfCopy, + Reference ccrCopy, + Reference const> dbInfoCopy, + LocalityData localityCopy, + Optional expectedWorkLoadCopy) { PromiseStream> addWorkload; Future workerFatalError = actorCollection(addWorkload.getFuture()); @@ -626,3 +618,11 @@ Future testerServerCore(TesterInterface const& interf, } co_return; } + +Future testerServerCore(TesterInterface const& interf, + Reference const& ccr, + Reference const> const& dbInfo, + LocalityData const& locality, + Optional const& expectedWorkLoad) { + return testerServerCoreImpl(interf, ccr, dbInfo, locality, expectedWorkLoad); +} diff --git a/fdbserver/tester/WorkloadUtils.cpp b/fdbserver/tester/WorkloadUtils.cpp index d84414b8287..c68eaef2868 100644 --- a/fdbserver/tester/WorkloadUtils.cpp +++ b/fdbserver/tester/WorkloadUtils.cpp @@ -21,19 +21,28 @@ #include #include #include +#include #include #include #include "flow/CoroUtils.h" #include "flow/DeterministicRandom.h" +#include "flow/ParseNumber.h" #include "flow/Trace.h" +#include "flow/UnitTest.h" #include "flow/genericactors.h" #include "fdbserver/core/ServerDBInfo.h" #include "fdbserver/tester/workloads.h" namespace { +template +Optional parseNumericOption(StringRef value) { + // Legacy options accept numeric prefixes, including "100000.0" for integers. + return parseNumberPrefix(value); +} + constexpr char HEX_CHAR_LOOKUP[16] = { '0', '1', '2', '3', '4', '5', '6', '7', '8', '9', 'a', 'b', 'c', 'd', 'e', 'f' }; Future> getMetricsCompoundWorkload(CompoundWorkload* self) { @@ -150,10 +159,10 @@ Value getOption(VectorRef options, Key key, Value defaultValue) { int getOption(VectorRef options, Key key, int defaultValue) { for (int i = 0; i < options.size(); i++) { if (options[i].key == key) { - int r; - if (sscanf(options[i].value.toString().c_str(), "%d", &r)) { + auto r = parseNumericOption(options[i].value); + if (r.present()) { options[i].value = ""_sr; - return r; + return r.get(); } else { TraceEvent(SevError, "InvalidTestOption").detail("OptionName", key); throw test_specification_invalid(); @@ -167,10 +176,10 @@ int getOption(VectorRef options, Key key, int defaultValue) { uint64_t getOption(VectorRef options, Key key, uint64_t defaultValue) { for (int i = 0; i < options.size(); i++) { if (options[i].key == key) { - uint64_t r; - if (sscanf(options[i].value.toString().c_str(), "%" SCNd64, &r)) { + auto r = parseNumericOption(options[i].value); + if (r.present()) { options[i].value = ""_sr; - return r; + return r.get(); } else { TraceEvent(SevError, "InvalidTestOption").detail("OptionName", key); throw test_specification_invalid(); @@ -184,10 +193,10 @@ uint64_t getOption(VectorRef options, Key key, uint64_t defaultValu int64_t getOption(VectorRef options, Key key, int64_t defaultValue) { for (int i = 0; i < options.size(); i++) { if (options[i].key == key) { - int64_t r; - if (sscanf(options[i].value.toString().c_str(), "%" SCNd64, &r)) { + auto r = parseNumericOption(options[i].value); + if (r.present()) { options[i].value = ""_sr; - return r; + return r.get(); } else { TraceEvent(SevError, "InvalidTestOption").detail("OptionName", key); throw test_specification_invalid(); @@ -201,10 +210,11 @@ int64_t getOption(VectorRef options, Key key, int64_t defaultValue) double getOption(VectorRef options, Key key, double defaultValue) { for (int i = 0; i < options.size(); i++) { if (options[i].key == key) { - float r; - if (sscanf(options[i].value.toString().c_str(), "%f", &r)) { + // Preserve the float rounding used by existing simulation configurations. + auto r = parseNumericOption(options[i].value); + if (r.present()) { options[i].value = ""_sr; - return r; + return r.get(); } } } @@ -245,14 +255,22 @@ std::vector getOption(VectorRef options, Key key, std::vector< for (int i = 0; i < options.size(); i++) { if (options[i].key == key) { std::vector v; + auto appendValue = [&](StringRef value) { + auto parsed = parseNumericOption(value); + if (!parsed.present()) { + TraceEvent(SevError, "InvalidTestOption").detail("OptionName", key); + throw test_specification_invalid(); + } + v.push_back(parsed.get()); + }; int begin = 0; for (int c = 0; c < options[i].value.size(); c++) { if (options[i].value[c] == ',') { - v.push_back(atoi((char*)options[i].value.begin() + begin)); + appendValue(options[i].value.substr(begin, c - begin)); begin = c + 1; } } - v.push_back(atoi((char*)options[i].value.begin() + begin)); + appendValue(options[i].value.substr(begin)); options[i].value = ""_sr; return v; } @@ -260,6 +278,45 @@ std::vector getOption(VectorRef options, Key key, std::vector< return defaultValue; } +TEST_CASE("/fdbserver/WorkloadUtils/numericOptions") { + ASSERT_EQ(parseNumericOption(" \t+0012 \n"_sr).get(), 12); + ASSERT_EQ(parseNumericOption("100000.0"_sr).get(), 100000); + ASSERT_EQ(parseNumericOption("12suffix"_sr).get(), 12); + ASSERT_EQ(parseNumericOption("12\0suffix"_sr).get(), 12); + ASSERT_EQ(parseNumericOption("-9223372036854775808"_sr).get(), std::numeric_limits::min()); + ASSERT_EQ(parseNumericOption("9223372036854775807"_sr).get(), std::numeric_limits::max()); + ASSERT_EQ(parseNumericOption("18446744073709551615"_sr).get(), std::numeric_limits::max()); + ASSERT_EQ(parseNumericOption("-1"_sr).get(), std::numeric_limits::max()); + ASSERT_EQ(parseNumericOption(" \t+1.25e2 \n"_sr).get(), 125.0f); + ASSERT_EQ(parseNumericOption("1.25suffix"_sr).get(), 1.25f); + ASSERT(!parseNumericOption("9223372036854775808"_sr).present()); + ASSERT(!parseNumericOption("-9223372036854775809"_sr).present()); + ASSERT(!parseNumericOption("18446744073709551616"_sr).present()); + ASSERT( + !parseNumericOption(std::to_string(static_cast(std::numeric_limits::max()) + 1)).present()); + ASSERT(!parseNumericOption("1e9999"_sr).present()); + ASSERT(!parseNumericOption("1e-9999"_sr).present()); + for (StringRef text : { ""_sr, " \t"_sr, "+"_sr, "suffix12"_sr }) { + ASSERT(!parseNumericOption(text).present()); + ASSERT(!parseNumericOption(text).present()); + } + + const uint8_t backing[] = { '1', ',', '2', '3', 0 }; + Standalone> options; + options.push_back(options.arena(), KeyValueRef("integers"_sr, StringRef(backing, 3))); + const std::vector parsed = getOption(options, "integers"_sr, std::vector{}); + ASSERT(parsed == std::vector({ 1, 2 })); + ASSERT(options[0].value.empty()); + + options.push_back(options.arena(), KeyValueRef("double"_sr, "0.1"_sr)); + ASSERT_EQ(getOption(options, "double"_sr, 0.0), static_cast(0.1f)); + options.push_back(options.arena(), KeyValueRef("invalidDouble"_sr, "invalid"_sr)); + ASSERT_EQ(getOption(options, "invalidDouble"_sr, 3.5), 3.5); + ASSERT(options[2].value == "invalid"_sr); + + return Void(); +} + bool hasOption(VectorRef options, Key key) { for (const auto& option : options) { if (option.key == key) { diff --git a/fdbserver/tester/test.cpp b/fdbserver/tester/test.cpp index 1bf71946483..8753d42dc6b 100644 --- a/fdbserver/tester/test.cpp +++ b/fdbserver/tester/test.cpp @@ -65,13 +65,9 @@ void throwIfError(const std::vector>>& futures, std::string er } } -Future runWorkload(Database const& cx, - std::vector const& testers, - TestSpec const& spec) { - // C++20 coroutine safety: copy const& params to survive across suspend points - Database cxCopy = cx; - std::vector testersCopy = testers; - TestSpec specCopy = spec; +static Future runWorkloadImpl(Database cxCopy, + std::vector testersCopy, + TestSpec specCopy) { std::string name = printable(specCopy.title); TraceEvent("TestRunning") @@ -187,6 +183,12 @@ Future runWorkload(Database const& cx, co_return DistributedTestResults(aggregateMetrics(metricsResults), success, failure); } +Future runWorkload(Database const& cx, + std::vector const& testers, + TestSpec const& spec) { + return runWorkloadImpl(cx, testers, spec); +} + // Sets the database configuration by running the ChangeConfig workload Future changeConfiguration(Database cx, std::vector testers, StringRef configMode) { TestSpec spec; @@ -875,29 +877,15 @@ Future runTests8(Reference runTests(Reference const& connRecordUnsafe, - test_type_t const& whatToRunUnsafe, - test_location_t const& atUnsafe, - int const& minTestersExpectedUnsafe, - std::string const& fileNameUnsafe, - StringRef const& startingConfigurationUnsafe, - LocalityData const& localityUnsafe, - UnitTestParameters const& testOptionsUnsafe, - bool const& restartingTestUnsafe) { - // C++20 coroutine safety: copy parameters that might bind to temporaries (default args). - // const& parameters only store the reference in the coroutine frame; temporaries are - // destroyed after the first suspend point, leaving dangling references. - // Just do this for all parameters, including ones that could be passed by value. - Reference connRecord = connRecordUnsafe; - test_type_t whatToRun = whatToRunUnsafe; - test_location_t at = atUnsafe; - int minTestersExpected = minTestersExpectedUnsafe; - std::string fileName = fileNameUnsafe; - StringRef startingConfiguration = startingConfigurationUnsafe; - LocalityData locality = localityUnsafe; - UnitTestParameters testOptions = testOptionsUnsafe; - bool restartingTest = restartingTestUnsafe; - +static Future runTestsImpl(Reference connRecord, + test_type_t whatToRun, + test_location_t at, + int minTestersExpected, + std::string fileName, + Standalone startingConfiguration, + LocalityData locality, + UnitTestParameters testOptions, + bool restartingTest) { TestSet testSet; std::unique_ptr knobProtectiveGroup(nullptr); auto cc = makeReference>>(); @@ -1015,6 +1003,26 @@ Future runTests(Reference const& connRecordUnsaf .run(); } +Future runTests(Reference const& connRecordUnsafe, + test_type_t const& whatToRunUnsafe, + test_location_t const& atUnsafe, + int const& minTestersExpectedUnsafe, + std::string const& fileNameUnsafe, + StringRef const& startingConfigurationUnsafe, + LocalityData const& localityUnsafe, + UnitTestParameters const& testOptionsUnsafe, + bool const& restartingTestUnsafe) { + return runTestsImpl(connRecordUnsafe, + whatToRunUnsafe, + atUnsafe, + minTestersExpectedUnsafe, + fileNameUnsafe, + Standalone(startingConfigurationUnsafe), + localityUnsafe, + testOptionsUnsafe, + restartingTestUnsafe); +} + namespace { Future testExpectedErrorImpl(Future test, const char* testDescr, diff --git a/fdbserver/worker/worker.cpp b/fdbserver/worker/worker.cpp index 5dd0bf244a1..1c543cd51f5 100644 --- a/fdbserver/worker/worker.cpp +++ b/fdbserver/worker/worker.cpp @@ -18,6 +18,7 @@ * limitations under the License. */ +#include "flow/ParseNumber.h" #include #include #include @@ -1237,7 +1238,7 @@ UpdateWorkerHealthRequest doPeerHealthCheck(const WorkerInterface& interf, // Note that we don't need to calculate recovered peer in this case since all the recently closed peers are // considered permanently closed peers. for (const auto& address : FlowTransport::transport().getRecentClosedPeers()) { - if (allPeers.find(address) != allPeers.end()) { + if (allPeers.contains(address)) { // We have checked this peer in the above for loop. continue; } @@ -1491,7 +1492,13 @@ Future runProfiler(ProfilerRequest req) { bool checkHighMemory(int64_t threshold, bool* error) { #if defined(__linux__) && defined(USE_GPERFTOOLS) && !defined(VALGRIND) *error = false; - uint64_t page_size = sysconf(_SC_PAGESIZE); + const long pageSizeResult = sysconf(_SC_PAGESIZE); + if (pageSizeResult <= 0) { + TraceEvent("GetPageSizeFailure").log(); + *error = true; + return false; + } + const uint64_t page_size = static_cast(pageSizeResult); int fd = open("/proc/self/statm", O_RDONLY | O_CLOEXEC); if (fd < 0) { TraceEvent("OpenStatmFileFailure").log(); @@ -1502,15 +1509,23 @@ bool checkHighMemory(int64_t threshold, bool* error) { const int buf_sz = 256; char stat_buf[buf_sz]; ssize_t stat_nread = read(fd, stat_buf, buf_sz); - if (stat_nread < 0) { + close(fd); + if (stat_nread <= 0) { TraceEvent("ReadStatmFileFailure").log(); *error = true; return false; } - uint64_t vmsize, rss; - sscanf(stat_buf, "%lu %lu", &vmsize, &rss); - rss *= page_size; + StringRef statText(reinterpret_cast(stat_buf), stat_nread); + int consumed = 0; + auto vmsize = parseNumberPrefix(statText, 10, &consumed); + auto rssPages = parseNumberPrefix(statText.substr(consumed)); + if (!vmsize.present() || !rssPages.present() || rssPages.get() > std::numeric_limits::max() / page_size) { + TraceEvent("ParseStatmFileFailure").log(); + *error = true; + return false; + } + uint64_t rss = rssPages.get() * page_size; if (rss >= threshold) { return true; } @@ -2890,7 +2905,7 @@ class WorkerServerCore { lastSnapReq(lastSnapReq), snapReqMap(snapReqMap), snapReqResultMap(snapReqResultMap), lastSnapTime(lastSnapTime) {} - Future run(Future const& handleErrors) { + Future run(Future handleErrors) { auto res = co_await race(interf.clientInterface.reboot.getFuture(), serveServerDBInfoUpdates(), serveFailureInjectionRequests(), diff --git a/fdbserver/workloads/BackgroundSelectors.cpp b/fdbserver/workloads/BackgroundSelectors.cpp index c6992f97cb7..7399616590f 100644 --- a/fdbserver/workloads/BackgroundSelectors.cpp +++ b/fdbserver/workloads/BackgroundSelectors.cpp @@ -49,7 +49,9 @@ struct BackgroundSelectorWorkload : TestWorkload { Future setup(Database const& cx) override { return Void(); } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { for (int c = 0; c < actorsPerClient; c++) clients.push_back(timeout(backgroundSelectorWorker(cx, this), testDuration, Void())); co_await waitForAll(clients); diff --git a/fdbserver/workloads/BackupToDBAbort.cpp b/fdbserver/workloads/BackupToDBAbort.cpp index 2293ac51388..e5a14ace3d1 100644 --- a/fdbserver/workloads/BackupToDBAbort.cpp +++ b/fdbserver/workloads/BackupToDBAbort.cpp @@ -92,7 +92,9 @@ struct BackupToDBAbort : TestWorkload { } } - Future check(const Database& cx) override { + Future check(const Database& cx) override { return checkImpl(cx); } + + Future checkImpl(Database cx) { TraceEvent("BDBA_UnlockPrimary").log(); // Too much of the tester framework expects the primary database to be unlocked, so we unlock it // once all of the workloads have finished. diff --git a/fdbserver/workloads/BulkDumping.cpp b/fdbserver/workloads/BulkDumping.cpp index c6f2ae64290..57ffe588e15 100644 --- a/fdbserver/workloads/BulkDumping.cpp +++ b/fdbserver/workloads/BulkDumping.cpp @@ -436,7 +436,7 @@ struct BulkDumping : TestWorkload { } for (const auto& [key, value] : newKvs) { // newKvs should not contain keys outside the bulkDumpJobRange - ASSERT(keyOutsideDumpData.find(key) == keyOutsideDumpData.end() && bulkDumpJobRange.contains(key)); + ASSERT(!keyOutsideDumpData.contains(key) && bulkDumpJobRange.contains(key)); if (self->keyContainedInRanges(key, ignoreRanges)) { continue; } @@ -465,7 +465,9 @@ struct BulkDumping : TestWorkload { // (9) Validate the loaded data in DB is same as the data in DB before dumping within the bulkdump job range and // bulkload job range. Note that the bulkload job can be unretriable error. In this case, we ignore the error range; // (10) Validate the bulk load job history. - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { if (clientId != 0) { co_return; } diff --git a/fdbserver/workloads/BulkLoading.cpp b/fdbserver/workloads/BulkLoading.cpp index 3dd3d7fd28f..c9ed2359f88 100644 --- a/fdbserver/workloads/BulkLoading.cpp +++ b/fdbserver/workloads/BulkLoading.cpp @@ -742,7 +742,9 @@ struct BulkLoading : TestWorkload { TraceEvent("BulkLoadingWorkLoadComplexTestComplete"); } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { if (clientId != 0) { co_return; } diff --git a/fdbserver/workloads/CommitBugCheck.cpp b/fdbserver/workloads/CommitBugCheck.cpp index fc4c7549b09..0eac6af5e66 100644 --- a/fdbserver/workloads/CommitBugCheck.cpp +++ b/fdbserver/workloads/CommitBugCheck.cpp @@ -18,6 +18,7 @@ * limitations under the License. */ +#include "flow/ParseNumber.h" #include "fdbserver/core/TesterInterface.h" #include "fdbserver/tester/workloads.h" @@ -115,7 +116,7 @@ struct CommitBugWorkload : TestWorkload { Optional val = co_await tr.get(key); int num = 0; if (val.present()) { - num = atoi(val.get().toString().c_str()); + num = parseNumberPrefix(val.get()).orDefault(0); if (num != i) { TraceEvent(SevError, "CommitBug2Failed").detail("Value", num).detail("Expected", i); self->success = false; diff --git a/fdbserver/workloads/ConfigureDatabase.cpp b/fdbserver/workloads/ConfigureDatabase.cpp index 3e82f43919c..8f76c54918b 100644 --- a/fdbserver/workloads/ConfigureDatabase.cpp +++ b/fdbserver/workloads/ConfigureDatabase.cpp @@ -18,6 +18,7 @@ * limitations under the License. */ +#include "flow/ParseNumber.h" #include #include "fdbclient/FDBTypes.h" @@ -256,11 +257,7 @@ struct ConfigureDatabaseWorkload : TestWorkload { void getMetrics(std::vector& m) override { m.push_back(retries.getMetric()); } - static inline uint64_t valueToUInt64(const StringRef& v) { - long long unsigned int x = 0; - sscanf(v.toString().c_str(), "%llx", &x); - return x; - } + static inline uint64_t valueToUInt64(const StringRef& v) { return parseNumberPrefix(v, 16).orDefault(0); } inline Standalone getDatabaseName(int dbIndex) { return StringRef(format("DestroyDB%d", dbIndex)); } @@ -269,11 +266,15 @@ struct ConfigureDatabaseWorkload : TestWorkload { return ManagementAPI::changeConfig(cx.getReference(), config, force); } - Future setup(Database const& cx) override { + Future setup(Database const& cx) override { return setupImpl(cx); } + + Future setupImpl(Database cx) { co_await ManagementAPI::changeConfig(cx.getReference(), "single storage_migration_type=aggressive", true); } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { DatabaseConfiguration config = co_await getDatabaseConfiguration(cx); TraceEvent("ConfigureDatabase_Config").detail("Config", config.toString()); if (!SERVER_KNOBS->SHARD_ENCODE_LOCATION_METADATA) { @@ -314,7 +315,9 @@ struct ConfigureDatabaseWorkload : TestWorkload { co_return false; } - Future check(Database const& cx) override { + Future check(Database const& cx) override { return checkImpl(cx); } + + Future checkImpl(Database cx) { co_await delay(30.0); // only storage_migration_type=gradual && perpetual_storage_wiggle=1 need this check because in QuietDatabase // perpetual wiggle will be forced to close For other cases, later ConsistencyCheck will check KV store type diff --git a/fdbserver/workloads/ConflictRange.cpp b/fdbserver/workloads/ConflictRange.cpp index e835d8a3a0b..22dd9b2d142 100644 --- a/fdbserver/workloads/ConflictRange.cpp +++ b/fdbserver/workloads/ConflictRange.cpp @@ -65,7 +65,9 @@ struct ConflictRangeWorkload : TestWorkload { m.push_back(retries.getMetric()); } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { if (clientId == 0) co_await timeout(conflictRangeClient(cx, this), testDuration, Void()); } diff --git a/fdbserver/workloads/CpuProfiler.cpp b/fdbserver/workloads/CpuProfiler.cpp index 9c7b2ec8bb7..e1aeeda453d 100644 --- a/fdbserver/workloads/CpuProfiler.cpp +++ b/fdbserver/workloads/CpuProfiler.cpp @@ -96,7 +96,9 @@ struct CpuProfilerWorkload : TestWorkload { } } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { co_await delay(initialDelay); if (clientId == 0) TraceEvent("SignalProfilerOn").log(); @@ -111,7 +113,9 @@ struct CpuProfilerWorkload : TestWorkload { } } - Future check(Database const& cx) override { + Future check(Database const& cx) override { return checkImpl(cx); } + + Future checkImpl(Database cx) { // If no duration was given, then shut the profiler off now if (duration <= 0) { if (clientId == 0) diff --git a/fdbserver/workloads/DDBalance.cpp b/fdbserver/workloads/DDBalance.cpp index b889e947f18..9a4e3da24dd 100644 --- a/fdbserver/workloads/DDBalance.cpp +++ b/fdbserver/workloads/DDBalance.cpp @@ -56,7 +56,9 @@ struct DDBalanceWorkload : TestWorkload { Future setup(Database const& cx) override { return ddbalanceSetup(cx, this); } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { for (int c = 0; c < moversPerClient; c++) clients.push_back(timeout(ddBalanceMover(cx, this, c), testDuration, Void())); co_await waitForAll(clients); diff --git a/fdbserver/workloads/DDMetrics.cpp b/fdbserver/workloads/DDMetrics.cpp index 6094184695c..6415958fbfe 100644 --- a/fdbserver/workloads/DDMetrics.cpp +++ b/fdbserver/workloads/DDMetrics.cpp @@ -38,8 +38,7 @@ struct DDMetricsWorkload : TestWorkload { TraceEvent("GetHighPriorityReliocationsInFlight").detail("Stage", "ContactingMaster"); TraceEventFields md = co_await timeoutError(masterWorker.eventLogRequest.getReply(EventLogRequest("MovingData"_sr)), 1.0); - int relocations; - sscanf(md.getValue("UnhealthyRelocations").c_str(), "%d", &relocations); + int relocations = md.getInt("UnhealthyRelocations"); co_return relocations; } diff --git a/fdbserver/workloads/DDMetricsExclude.cpp b/fdbserver/workloads/DDMetricsExclude.cpp index 00b3854c952..7f60d23da2b 100644 --- a/fdbserver/workloads/DDMetricsExclude.cpp +++ b/fdbserver/workloads/DDMetricsExclude.cpp @@ -69,7 +69,9 @@ struct DDMetricsExcludeWorkload : TestWorkload { co_return -1.0; } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { try { std::vector excluded; excluded.push_back(AddressExclusion(IPAddress::parse(excludeIp.toString()).get(), excludePort)); diff --git a/fdbserver/workloads/DataDistributionMetrics.cpp b/fdbserver/workloads/DataDistributionMetrics.cpp index 4a45c87f96c..a56fecb481d 100644 --- a/fdbserver/workloads/DataDistributionMetrics.cpp +++ b/fdbserver/workloads/DataDistributionMetrics.cpp @@ -196,7 +196,9 @@ struct DataDistributionMetricsWorkload : KVWorkload { co_return true; } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { std::vector> clients; clients.push_back(resultConsistencyCheckClient(cx, this)); for (int i = 0; i < actorCount; ++i) diff --git a/fdbserver/workloads/DifferentClustersSameRV.cpp b/fdbserver/workloads/DifferentClustersSameRV.cpp index 6c59f6c5951..d9d23701d96 100644 --- a/fdbserver/workloads/DifferentClustersSameRV.cpp +++ b/fdbserver/workloads/DifferentClustersSameRV.cpp @@ -107,7 +107,7 @@ struct DifferentClustersSameRVWorkload : TestWorkload { void getMetrics(std::vector& m) override {} - static Future>> doRead(Database cx, Value const& keyToRead) { + static Future>> doRead(Database cx, Value keyToRead) { Transaction tr(cx); while (true) { tr.setOption(FDBTransactionOptions::READ_LOCK_AWARE); @@ -233,7 +233,7 @@ struct DifferentClustersSameRVWorkload : TestWorkload { co_await unlockDatabase(originalDB, lockUid); // So quietDatabase can finish } - static Future writerClient(Database cx, Value const& keyToRead) { + static Future writerClient(Database cx, Value keyToRead) { Transaction tr(cx); while (true) { Error err; @@ -259,10 +259,7 @@ struct DifferentClustersSameRVWorkload : TestWorkload { } } - static Future> readAtVersion(Value const& keyToRead, - const char* name, - Transaction* tr, - Version version) { + static Future> readAtVersion(Value keyToRead, const char* name, Transaction* tr, Version version) { Optional res; try { tr->reset(); @@ -275,7 +272,7 @@ struct DifferentClustersSameRVWorkload : TestWorkload { } } - static Future readerClientSeparateDBs(Database cx, Database extraDB, Value const& keyToRead) { + static Future readerClientSeparateDBs(Database cx, Database extraDB, Value keyToRead) { Transaction tr1(cx); Transaction tr2(extraDB); Version rv1{ 0 }; diff --git a/fdbserver/workloads/FileSystem.cpp b/fdbserver/workloads/FileSystem.cpp index 371f28f345e..76725ee6e19 100644 --- a/fdbserver/workloads/FileSystem.cpp +++ b/fdbserver/workloads/FileSystem.cpp @@ -18,6 +18,7 @@ * limitations under the License. */ +#include "flow/ParseNumber.h" #include "fdbrpc/DDSketch.h" #include "fdbclient/NativeAPI.h" #include "fdbserver/core/TesterInterface.h" @@ -161,7 +162,9 @@ struct FileSystemWorkload : TestWorkload { .detail("FilesToSetUp", nodesToSetUp); } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { FileSystemOp* operation; if (operationName == "deletionQuery") operation = new ServerDeletionCountQuery(); @@ -218,11 +221,7 @@ struct FileSystemWorkload : TestWorkload { } } - static int testKeyToInt(const KeyRef& p) { - int x = 0; - sscanf(p.toString().c_str(), "%d", &x); - return x; - } + static int testKeyToInt(const KeyRef& p) { return parseNumberPrefix(p).orDefault(0); } Future writeClient(Database cx, FileSystemWorkload* self) { double clientBegin = now(); diff --git a/fdbserver/workloads/HTTPKeyValueStore.cpp b/fdbserver/workloads/HTTPKeyValueStore.cpp index eff3c1ff0a8..1b79b7e1e1e 100644 --- a/fdbserver/workloads/HTTPKeyValueStore.cpp +++ b/fdbserver/workloads/HTTPKeyValueStore.cpp @@ -18,6 +18,7 @@ * limitations under the License. */ +#include "flow/ParseNumber.h" #include "flow/Arena.h" #include "flow/IRandom.h" #include "flow/Trace.h" @@ -112,8 +113,13 @@ Future httpKVRequestCallback(Reference kvStore, ASSERT(req->data.headers.contains("UID")); ASSERT(req->data.headers.contains("SeqNo")); - int clientId = atoi(req->data.headers["ClientID"].c_str()); - int seqNo = atoi(req->data.headers["SeqNo"].c_str()); + auto clientIdValue = parseNumberPrefix(StringRef(req->data.headers["ClientID"])); + auto seqNoValue = parseNumberPrefix(StringRef(req->data.headers["SeqNo"])); + if (!clientIdValue.present() || !seqNoValue.present()) { + throw http_request_failed(); + } + int clientId = clientIdValue.get(); + int seqNo = seqNoValue.get(); ASSERT(req->data.headers.contains("Content-Length")); ASSERT_EQ(req->data.headers["Content-Length"], std::to_string(req->data.content.size())); diff --git a/fdbserver/workloads/HealthMetricsApi.cpp b/fdbserver/workloads/HealthMetricsApi.cpp index ad6e7c068e4..3e8e251dac5 100644 --- a/fdbserver/workloads/HealthMetricsApi.cpp +++ b/fdbserver/workloads/HealthMetricsApi.cpp @@ -63,7 +63,9 @@ struct HealthMetricsApiWorkload : TestWorkload { maxAllowedStaleness = getOption(options, "maxAllowedStaleness"_sr, 60.0); } - Future setup(Database const& cx) override { + Future setup(Database const& cx) override { return setupImpl(cx); } + + Future setupImpl(Database cx) { if (!sendDetailedHealthMetrics) { // Internally cached health metrics time out after this knob. Wait // an extra second to avoid any off-by-1 ">" vs ">=" type issues. @@ -72,9 +74,9 @@ struct HealthMetricsApiWorkload : TestWorkload { cx->healthMetrics.tLogQueue.clear(); } } - Future start(Database const& cx) override { - co_await timeout(healthMetricsChecker(cx), testDuration, Void()); - } + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { co_await timeout(healthMetricsChecker(cx), testDuration, Void()); } Future check(Database const& cx) override { if (!gotMetrics) { diff --git a/fdbserver/workloads/HighContentionPrefixAllocatorWorkload.cpp b/fdbserver/workloads/HighContentionPrefixAllocatorWorkload.cpp index fbbb3ed3f44..733cd05328a 100644 --- a/fdbserver/workloads/HighContentionPrefixAllocatorWorkload.cpp +++ b/fdbserver/workloads/HighContentionPrefixAllocatorWorkload.cpp @@ -117,7 +117,9 @@ struct HighContentionPrefixAllocatorWorkload : TestWorkload { Future start(Database const& cx) override { return runTest(cx); } - Future check(Database const& cx) override { + Future check(Database const& cx) override { return checkImpl(cx); } + + Future checkImpl(Database cx) { if (expectedPrefixes != allocatedPrefixes.size()) { TraceEvent(SevError, "HighContentionAllocationWorkloadFailure") .detail("Reason", "Incorrect Number of Prefixes Allocated") diff --git a/fdbserver/workloads/Inventory.cpp b/fdbserver/workloads/Inventory.cpp index 81ca552a2be..f1460502304 100644 --- a/fdbserver/workloads/Inventory.cpp +++ b/fdbserver/workloads/Inventory.cpp @@ -18,6 +18,7 @@ * limitations under the License. */ +#include "flow/ParseNumber.h" #include "fdbclient/NativeAPI.h" #include "fdbserver/core/TesterInterface.h" #include "fdbserver/tester/workloads.h" @@ -121,7 +122,7 @@ struct InventoryTestWorkload : TestWorkload { std::map actualResults; for (int i = 0; i < data.size(); i++) - actualResults[data[i].key] = atoi(data[i].value.toString().c_str()); + actualResults[data[i].key] = parseNumberPrefix(data[i].value).orDefault(0); for (auto i = self->minExpectedResults.begin(); i != self->minExpectedResults.end(); ++i) actualResults[i->first]; bool error = false; @@ -155,7 +156,7 @@ struct InventoryTestWorkload : TestWorkload { Future inventoryTestWrite(Transaction* tr, Key key) { Optional val = co_await tr->get(key); - int count = !val.present() ? 0 : atoi(val.get().toString().c_str()); + int count = !val.present() ? 0 : parseNumberPrefix(val.get()).orDefault(0); ASSERT(count >= 0 && count < 1000000); tr->set(key, format("%d", count + 1)); } diff --git a/fdbserver/workloads/MemoryLifetime.cpp b/fdbserver/workloads/MemoryLifetime.cpp index 86256708f16..8f785221e08 100644 --- a/fdbserver/workloads/MemoryLifetime.cpp +++ b/fdbserver/workloads/MemoryLifetime.cpp @@ -45,12 +45,16 @@ struct MemoryLifetime : KVWorkload { void getMetrics(std::vector& m) override {} - Future setup(Database const& cx) override { + Future setup(Database const& cx) override { return setupImpl(cx); } + + Future setupImpl(Database cx) { Promise loadTime; co_await bulkSetup(cx, this, nodeCount, loadTime); } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { double startTime = now(); ReadYourWritesTransaction tr(cx); Reverse reverse = Reverse::False; diff --git a/fdbserver/workloads/MetricLogging.cpp b/fdbserver/workloads/MetricLogging.cpp index 94ffd7007bc..b93202df8ab 100644 --- a/fdbserver/workloads/MetricLogging.cpp +++ b/fdbserver/workloads/MetricLogging.cpp @@ -49,7 +49,9 @@ struct MetricLoggingWorkload : TestWorkload { } } - Future setup(Database const& cx) override { + Future setup(Database const& cx) override { return setupImpl(); } + + Future setupImpl() { co_await delay(2.0); for (int i = 0; i < metricCount; i++) { if (testBool) { diff --git a/fdbserver/workloads/ProtocolVersion.cpp b/fdbserver/workloads/ProtocolVersion.cpp index 0096b2da8b8..4f4f577b264 100644 --- a/fdbserver/workloads/ProtocolVersion.cpp +++ b/fdbserver/workloads/ProtocolVersion.cpp @@ -27,7 +27,9 @@ struct ProtocolVersionWorkload : TestWorkload { explicit ProtocolVersionWorkload(WorkloadContext const& wcx) : TestWorkload(wcx) {} - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(); } + + Future startImpl() { std::vector allProcesses = g_simulator->getAllProcesses(); auto diffVersionProcess = find_if(allProcesses.begin(), allProcesses.end(), [](const ISimulator::ProcessInfo* p) { diff --git a/fdbserver/workloads/QueuePush.cpp b/fdbserver/workloads/QueuePush.cpp index 9fbe630030b..9439f0122fa 100644 --- a/fdbserver/workloads/QueuePush.cpp +++ b/fdbserver/workloads/QueuePush.cpp @@ -17,6 +17,7 @@ * See the License for the specific language governing permissions and * limitations under the License. */ +#include "flow/ParseNumber.h" #include #include "fdbclient/FDBTypes.h" @@ -80,19 +81,20 @@ struct QueuePushWorkload : TestWorkload { static Key keyForIndex(int base, int offset) { return StringRef(format("%08x%08x", base, offset)); } static std::pair valuesForKey(KeyRef value) { - int base, offset; ASSERT(value.size() == 16); - - if (sscanf(value.substr(0, 8).toString().c_str(), "%x", &base) && - sscanf(value.substr(8, 8).toString().c_str(), "%x", &offset)) { - return std::make_pair(base, offset); + auto base = parseNumberPrefix(value.substr(0, 8), 16); + auto offset = parseNumberPrefix(value.substr(8, 8), 16); + if (base.present() && offset.present()) { + return std::make_pair(static_cast(base.get()), static_cast(offset.get())); } else { // SOMEDAY: what should this really be? Should we rely on exceptions for control flow here? throw client_invalid_operation(); } } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { for (int i = 0; i < actorCount; i++) { clients.push_back(writeClient(cx, this)); } diff --git a/fdbserver/workloads/RandomMoveKeys.cpp b/fdbserver/workloads/RandomMoveKeys.cpp index d815721da86..4ec68f31aca 100644 --- a/fdbserver/workloads/RandomMoveKeys.cpp +++ b/fdbserver/workloads/RandomMoveKeys.cpp @@ -57,7 +57,9 @@ struct MoveKeysWorkload : FailureInjectionWorkload { return alreadyAdded < 1 && work.useDatabase && 0.1 / (1 + alreadyAdded) > random.random01(); } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { if (enabled) { // Get the database configuration so as to use proper team size Transaction tr(cx); diff --git a/fdbserver/workloads/RandomRangeLock.cpp b/fdbserver/workloads/RandomRangeLock.cpp index b060a93cc47..d14f8c5b70c 100644 --- a/fdbserver/workloads/RandomRangeLock.cpp +++ b/fdbserver/workloads/RandomRangeLock.cpp @@ -164,7 +164,9 @@ struct RandomRangeLockWorkload : FailureInjectionWorkload { } } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { if (enabled) { // Run lockActorCount number of actor concurrently. // Each actor conducts (1) locking a range for a while and (2) unlocking the range. diff --git a/fdbserver/workloads/RangeLock.cpp b/fdbserver/workloads/RangeLock.cpp index b9d9abd2bba..e27af4827ca 100644 --- a/fdbserver/workloads/RangeLock.cpp +++ b/fdbserver/workloads/RangeLock.cpp @@ -578,7 +578,7 @@ struct RangeLocking : TestWorkload { std::map currentKvsInDB; currentKvsInDB = co_await self->getKVSFromDB(self, cx); for (const auto& [key, value] : currentKvsInDB) { - if (self->kvs.find(key) == self->kvs.end()) { + if (!self->kvs.contains(key)) { TraceEvent(SevError, "RangeLockWorkLoadHistory") .detail("Ops", "CheckDBUniqueKey") .detail("Key", key) @@ -596,7 +596,7 @@ struct RangeLocking : TestWorkload { } } for (const auto& [key, value] : self->kvs) { - if (currentKvsInDB.find(key) == currentKvsInDB.end()) { + if (!currentKvsInDB.contains(key)) { TraceEvent(SevError, "RangeLockWorkLoadHistory") .detail("Ops", "CheckMemoryUniqueKey") .detail("Key", key) @@ -827,7 +827,9 @@ struct RangeLocking : TestWorkload { ASSERT(remainingLocks.empty()); } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { if (clientId != 0) { co_return; } diff --git a/fdbserver/workloads/ReadAfterWrite.cpp b/fdbserver/workloads/ReadAfterWrite.cpp index 194fdf0782b..6955c3233a6 100644 --- a/fdbserver/workloads/ReadAfterWrite.cpp +++ b/fdbserver/workloads/ReadAfterWrite.cpp @@ -105,7 +105,9 @@ struct ReadAfterWriteWorkload : KVWorkload { } } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { Future lifetime = benchmark(cx); co_await delay(testDuration); } diff --git a/fdbserver/workloads/ReadHotDetection.cpp b/fdbserver/workloads/ReadHotDetection.cpp index 4eb07e8c5de..716ae13142d 100644 --- a/fdbserver/workloads/ReadHotDetection.cpp +++ b/fdbserver/workloads/ReadHotDetection.cpp @@ -45,7 +45,9 @@ struct ReadHotDetectionWorkload : TestWorkload { readKey = StringRef(format("testkey%08x", deterministicRandom()->randomInt(0, keyCount))); } - Future setup(Database const& cx) override { + Future setup(Database const& cx) override { return setupImpl(cx); } + + Future setupImpl(Database cx) { Standalone largeValue; Standalone smallValue; largeValue = randomString(largeValue.arena(), 100000); diff --git a/fdbserver/workloads/ReadWrite.cpp b/fdbserver/workloads/ReadWrite.cpp index 19e726e75ce..5e452d091c8 100644 --- a/fdbserver/workloads/ReadWrite.cpp +++ b/fdbserver/workloads/ReadWrite.cpp @@ -104,6 +104,8 @@ struct ReadWriteCommonImpl { } } + // The workload owns setup's mutable metrics and outlives its returned future. + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) static Future setup(Database cx, ReadWriteCommon& self) { if (!self.doSetup) co_return; diff --git a/fdbserver/workloads/S3ClientWorkload.cpp b/fdbserver/workloads/S3ClientWorkload.cpp index d68af07c2cd..ad969058aaf 100644 --- a/fdbserver/workloads/S3ClientWorkload.cpp +++ b/fdbserver/workloads/S3ClientWorkload.cpp @@ -153,7 +153,9 @@ struct S3ClientWorkload : TestWorkload { } } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(); } + + Future startImpl() { if (clientId != 0) { // Our simulation test can trigger multiple same workloads at the same time // Only run one time workload in the simulation diff --git a/fdbserver/workloads/SaveAndKill.cpp b/fdbserver/workloads/SaveAndKill.cpp index b52b9e97efb..f1148344f93 100644 --- a/fdbserver/workloads/SaveAndKill.cpp +++ b/fdbserver/workloads/SaveAndKill.cpp @@ -55,7 +55,9 @@ struct SaveAndKillWorkload : TestWorkload { g_simulator->disableSwapsToAll(); return Void(); } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { int i{ 0 }; co_await delay(deterministicRandom()->random01() * testDuration); DatabaseConfiguration config = co_await getDatabaseConfiguration(cx); @@ -82,12 +84,12 @@ struct SaveAndKillWorkload : TestWorkload { g_simulator->currentlyRebootingProcesses; std::map allProcessesMap; for (const auto& [_, process] : rebootingProcesses) { - if (allProcessesMap.find(process->dataFolder) == allProcessesMap.end() && !process->isSpawnedKVProcess()) { + if (!allProcessesMap.contains(process->dataFolder) && !process->isSpawnedKVProcess()) { allProcessesMap[process->dataFolder] = process; } } for (const auto& process : processes) { - if (allProcessesMap.find(process->dataFolder) == allProcessesMap.end() && !process->isSpawnedKVProcess()) { + if (!allProcessesMap.contains(process->dataFolder) && !process->isSpawnedKVProcess()) { allProcessesMap[process->dataFolder] = process; } } @@ -99,7 +101,7 @@ struct SaveAndKillWorkload : TestWorkload { std::string machineId = printable(process->locality.machineId()); const char* machineIdString = machineId.c_str(); if (!process->excludeFromRestarts) { - if (machines.find(machineId) == machines.end()) { + if (!machines.contains(machineId)) { machines.insert(std::pair(machineId, 1)); ini.SetValue("META", format("%d", j).c_str(), machineIdString); ini.SetValue( diff --git a/fdbserver/workloads/SkewedReadWrite.cpp b/fdbserver/workloads/SkewedReadWrite.cpp index 074898f2920..2a02fe0c8ba 100644 --- a/fdbserver/workloads/SkewedReadWrite.cpp +++ b/fdbserver/workloads/SkewedReadWrite.cpp @@ -204,7 +204,9 @@ struct SkewedReadWriteWorkload : ReadWriteCommon { } } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { std::vector> clients; if (enableReadLatencyLogging) clients.push_back(tracePeriodically()); diff --git a/fdbserver/workloads/SnapTest.cpp b/fdbserver/workloads/SnapTest.cpp index 1dde81408a5..b7c30a2cf8f 100644 --- a/fdbserver/workloads/SnapTest.cpp +++ b/fdbserver/workloads/SnapTest.cpp @@ -18,6 +18,7 @@ * limitations under the License. */ +#include "flow/ParseNumber.h" #include #include "fdbclient/ManagementAPI.h" #include "fdbclient/NativeAPI.h" @@ -218,7 +219,13 @@ struct SnapTestWorkload : TestWorkload { CSimpleIni ini; ini.SetUnicode(); ini.LoadFile(restartInfoLocation.c_str()); - bool backupFailed = atoi(ini.GetValue("RESTORE", "BackupFailed")); + const char* backupFailedText = ini.GetValue("RESTORE", "BackupFailed"); + auto backupFailedValue = + backupFailedText == nullptr ? Optional() : parseNumberPrefix(StringRef(backupFailedText)); + if (!backupFailedValue.present()) { + throw test_specification_invalid(); + } + bool backupFailed = backupFailedValue.get(); if (backupFailed) { // since backup failed, skip the restore checking TraceEvent(SevWarnAlways, "BackupFailedSkippingRestoreCheck").log(); diff --git a/fdbserver/workloads/SpecialKeySpaceCorrectness.cpp b/fdbserver/workloads/SpecialKeySpaceCorrectness.cpp index 2bb9022d542..c460c850368 100644 --- a/fdbserver/workloads/SpecialKeySpaceCorrectness.cpp +++ b/fdbserver/workloads/SpecialKeySpaceCorrectness.cpp @@ -116,7 +116,9 @@ struct SpecialKeySpaceCorrectnessWorkload : TestWorkload { return Void(); } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { testRywLifetime(cx); co_await timeout(testSpecialKeySpaceErrors(cx, this) && getRangeCallActor(cx, this) && testConflictRanges(cx, /*read*/ true) && testConflictRanges(cx, /*read*/ false) && diff --git a/fdbserver/workloads/SpecialKeySpaceRobustness.cpp b/fdbserver/workloads/SpecialKeySpaceRobustness.cpp index 958c64253a1..1282e2f100e 100644 --- a/fdbserver/workloads/SpecialKeySpaceRobustness.cpp +++ b/fdbserver/workloads/SpecialKeySpaceRobustness.cpp @@ -39,7 +39,9 @@ struct SpecialKeySpaceRobustnessWorkload : TestWorkload { Future _setup(Database cx, SpecialKeySpaceRobustnessWorkload* self) { return Void(); } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { // Only use one client to avoid potential conflicts on changing cluster configuration if (clientId == 0) co_await managementApiCorrectnessActor(cx, this); diff --git a/fdbserver/workloads/Storefront.cpp b/fdbserver/workloads/Storefront.cpp index 9c7c8589615..1e14042821a 100644 --- a/fdbserver/workloads/Storefront.cpp +++ b/fdbserver/workloads/Storefront.cpp @@ -18,6 +18,7 @@ * limitations under the License. */ +#include "flow/ParseNumber.h" #include "fdbclient/NativeAPI.h" #include "fdbserver/core/TesterInterface.h" #include "fdbserver/tester/workloads.h" @@ -87,11 +88,7 @@ struct StorefrontWorkload : TestWorkload { return x; }*/ - static inline int valueToInt(const StringRef& v) { - int x = 0; - sscanf(v.toString().c_str(), "%d", &x); - return x; - } + static inline int valueToInt(const StringRef& v) { return parseNumberPrefix(v).orDefault(0); } Key keyForIndex(int n) { return itemKey(n); } Key itemKey(int item) { return StringRef(format("/items/%016d", item)); } diff --git a/fdbserver/workloads/StreamingRangeRead.cpp b/fdbserver/workloads/StreamingRangeRead.cpp index 34dd5362d89..9347b996858 100644 --- a/fdbserver/workloads/StreamingRangeRead.cpp +++ b/fdbserver/workloads/StreamingRangeRead.cpp @@ -92,7 +92,9 @@ struct StreamingRangeReadWorkload : KVWorkload { return delay(testDuration); } - Future check(Database const& cx) override { + Future check(Database const& cx) override { return checkImpl(cx); } + + Future checkImpl(Database cx) { client = Void(); co_await checkSelectorBoundaries(cx->clone()); co_return true; diff --git a/fdbserver/workloads/TaskBucketCorrectness.cpp b/fdbserver/workloads/TaskBucketCorrectness.cpp index 50b0539fb62..3bd6e1abb48 100644 --- a/fdbserver/workloads/TaskBucketCorrectness.cpp +++ b/fdbserver/workloads/TaskBucketCorrectness.cpp @@ -18,6 +18,7 @@ * limitations under the License. */ +#include "flow/ParseNumber.h" #include "flow/UnitTest.h" #include "flow/Error.h" #include "fdbclient/Tuple.h" @@ -76,8 +77,8 @@ struct SayHelloTaskFunc : TaskFuncBase { if (!task->params["chained"_sr].compare("false"_sr)) { co_await done->set(tr, taskBucket); } else { - int subtaskCount = atoi(task->params["subtaskCount"_sr].toString().c_str()); - int currTaskNumber = atoi(value.removePrefix("task_"_sr).toString().c_str()); + int subtaskCount = parseNumberPrefix(task->params["subtaskCount"_sr]).orDefault(0); + int currTaskNumber = parseNumberPrefix(value.removePrefix("task_"_sr)).orDefault(0); TraceEvent("TaskBucketCorrectnessSayHello") .detail("SubtaskCount", subtaskCount) .detail("CurrTaskNumber", currTaskNumber); @@ -134,7 +135,7 @@ struct SayHelloToEveryoneTaskFunc : TaskFuncBase { int subtaskCount = 1; if (!task->params["chained"_sr].compare("false"_sr)) { - subtaskCount = atoi(task->params["subtaskCount"_sr].toString().c_str()); + subtaskCount = parseNumberPrefix(task->params["subtaskCount"_sr]).orDefault(0); } for (int i = 0; i < subtaskCount; ++i) { auto new_task = makeReference( @@ -253,7 +254,9 @@ struct TaskBucketCorrectnessWorkload : TestWorkload { co_await allDone->onSetAddTask(tr, taskBucket, taskDone); } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { auto tr = makeReference(cx); Subspace taskSubspace("backup-agent"_sr); auto taskBucket = makeReference(taskSubspace.get("tasks"_sr)); @@ -325,7 +328,9 @@ struct TaskBucketCorrectnessWorkload : TestWorkload { } } - Future check(Database const& cx) override { + Future check(Database const& cx) override { return checkImpl(cx); } + + Future checkImpl(Database cx) { bool ret = co_await runRYWTransaction( cx, [=](Reference tr) { return checkSayHello(tr, subtaskCount); }); co_return ret; diff --git a/fdbserver/workloads/ThreadSafety.cpp b/fdbserver/workloads/ThreadSafety.cpp index 03e74536a3d..ca57a1964f6 100644 --- a/fdbserver/workloads/ThreadSafety.cpp +++ b/fdbserver/workloads/ThreadSafety.cpp @@ -137,7 +137,9 @@ struct ThreadSafetyWorkload : TestWorkload { Future setup(Database const& cx) override { return Void(); } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { std::vector threadInfo; Reference dbRef = diff --git a/fdbserver/workloads/Throttling.cpp b/fdbserver/workloads/Throttling.cpp index 549a4fc066c..1504d434f80 100644 --- a/fdbserver/workloads/Throttling.cpp +++ b/fdbserver/workloads/Throttling.cpp @@ -182,7 +182,9 @@ struct ThrottlingWorkload : KVWorkload { } } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { std::vector> clientActors; clientActors.reserve(static_cast(actorsPerClient > 0 ? actorsPerClient : 0) + 2); for (int actorId = 0; actorId < actorsPerClient; ++actorId) { diff --git a/fdbserver/workloads/TimeKeeperCorrectness.cpp b/fdbserver/workloads/TimeKeeperCorrectness.cpp index 4c907f7892a..637a63ea531 100644 --- a/fdbserver/workloads/TimeKeeperCorrectness.cpp +++ b/fdbserver/workloads/TimeKeeperCorrectness.cpp @@ -37,7 +37,9 @@ struct TimeKeeperCorrectnessWorkload : TestWorkload { void getMetrics(std::vector& m) override {} - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { TraceEvent(SevInfo, "TKCorrectness_Start").log(); double start = now(); @@ -64,7 +66,9 @@ struct TimeKeeperCorrectnessWorkload : TestWorkload { TraceEvent(SevInfo, "TKCorrectness_Completed").log(); } - Future check(Database const& cx) override { + Future check(Database const& cx) override { return checkImpl(cx); } + + Future checkImpl(Database cx) { KeyBackedMap dbTimeKeeper = KeyBackedMap(timeKeeperPrefixRange.begin); auto tr = makeReference(cx); diff --git a/fdbserver/workloads/TransactionCost.cpp b/fdbserver/workloads/TransactionCost.cpp index 53a90e59ca8..c3762ee70f2 100644 --- a/fdbserver/workloads/TransactionCost.cpp +++ b/fdbserver/workloads/TransactionCost.cpp @@ -62,6 +62,8 @@ class TransactionCostWorkload : public TestWorkload { }; class ReadLargeValueTest : public ITest { + // The enclosing workload retains this test and outlives its setup future. + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future setupImpl(TransactionCostWorkload const& workload, Database cx) { Transaction tr(cx); while (true) { @@ -148,6 +150,8 @@ class TransactionCostWorkload : public TestWorkload { }; class ReadRangeTest : public ITest { + // The enclosing workload retains this test and outlives its setup future. + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future setupImpl(TransactionCostWorkload const& workload, Database cx) { Transaction tr(cx); while (true) { @@ -181,6 +185,8 @@ class TransactionCostWorkload : public TestWorkload { }; class ReadMultipleValuesTest : public ITest { + // The enclosing workload retains this test and outlives its setup future. + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future setupImpl(TransactionCostWorkload const& workload, Database cx) { Transaction tr(cx); while (true) { @@ -218,6 +224,8 @@ class TransactionCostWorkload : public TestWorkload { }; class LargeReadRangeTest : public ITest { + // The enclosing workload retains this test and outlives its setup future. + // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future setupImpl(TransactionCostWorkload const& workload, Database cx) { Transaction tr(cx); while (true) { diff --git a/fdbserver/workloads/UnitTests.cpp b/fdbserver/workloads/UnitTests.cpp index 1b3a90ee4fa..5d5ebb7bdb9 100644 --- a/fdbserver/workloads/UnitTests.cpp +++ b/fdbserver/workloads/UnitTests.cpp @@ -188,6 +188,8 @@ struct UnitTestWorkload : TestWorkload { } } + // The comparator orders by test names, independent of pointer values. + // NOLINTNEXTLINE(bugprone-nondeterministic-pointer-iteration-order) std::sort(tests.begin(), tests.end(), [](auto lhs, auto rhs) { return std::string_view(lhs->name) < std::string_view(rhs->name); }); diff --git a/fdbserver/workloads/WatchAndWait.cpp b/fdbserver/workloads/WatchAndWait.cpp index 1ed822edaa2..da6d89d9d8d 100644 --- a/fdbserver/workloads/WatchAndWait.cpp +++ b/fdbserver/workloads/WatchAndWait.cpp @@ -84,7 +84,9 @@ struct WatchAndWaitWorkload : TestWorkload { m.push_back(retries.getMetric()); } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { std::vector> watches; uint64_t endNode = (nodeCount * (clientId + 1)) / clientCount; uint64_t startNode = (nodeCount * clientId) / clientCount; diff --git a/fdbserver/workloads/Watches.cpp b/fdbserver/workloads/Watches.cpp index 81f591df66a..e73430f26a8 100644 --- a/fdbserver/workloads/Watches.cpp +++ b/fdbserver/workloads/Watches.cpp @@ -55,7 +55,9 @@ struct WatchesWorkload : TestWorkload { out.insert("RandomRangeLock"); } - Future setup(Database const& cx) override { + Future setup(Database const& cx) override { return setupImpl(cx); } + + Future setupImpl(Database cx) { // return _setup(cx, this); std::vector> setupActors; for (int i = 0; i < nodes; i++) { diff --git a/fdbserver/workloads/WorkerErrors.cpp b/fdbserver/workloads/WorkerErrors.cpp index e1cf5216898..ccc8b504ab8 100644 --- a/fdbserver/workloads/WorkerErrors.cpp +++ b/fdbserver/workloads/WorkerErrors.cpp @@ -52,7 +52,9 @@ struct WorkerErrorsWorkload : TestWorkload { co_return results; } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(); } + + Future startImpl() { std::vector workers = co_await getWorkers(dbInfo); std::vector errors = co_await latestEventOnWorkers(workers); for (const auto& e : errors) { diff --git a/fdbserver/workloads/WriteBandwidth.cpp b/fdbserver/workloads/WriteBandwidth.cpp index af58ca464fa..4525a158456 100644 --- a/fdbserver/workloads/WriteBandwidth.cpp +++ b/fdbserver/workloads/WriteBandwidth.cpp @@ -76,7 +76,9 @@ struct WriteBandwidthWorkload : KVWorkload { Standalone operator()(uint64_t n) { return KeyValueRef(keyForIndex(n, false), randomValue()); } - Future setup(Database const& cx) override { + Future setup(Database const& cx) override { return setupImpl(cx); } + + Future setupImpl(Database cx) { Promise loadTime; Promise>> ratesAtKeyCounts; @@ -84,7 +86,9 @@ struct WriteBandwidthWorkload : KVWorkload { this->loadTime = loadTime.getFuture().get(); } - Future start(Database const& cx) override { + Future start(Database const& cx) override { return startImpl(cx); } + + Future startImpl(Database cx) { for (int i = 0; i < actorCount; i++) { clients.push_back(writeClient(cx, this)); } diff --git a/fdbserver/workloads/pubsub.cpp b/fdbserver/workloads/pubsub.cpp index b372f45f81f..2ecaea625cb 100644 --- a/fdbserver/workloads/pubsub.cpp +++ b/fdbserver/workloads/pubsub.cpp @@ -18,6 +18,7 @@ * limitations under the License. */ +#include "flow/ParseNumber.h" #include #include "fdbclient/NativeAPI.h" #include "pubsub.h" @@ -26,9 +27,7 @@ Value uInt64ToValue(uint64_t v) { return StringRef(format("%016llx", v)); } uint64_t valueToUInt64(const StringRef& v) { - uint64_t x = 0; - sscanf(v.toString().c_str(), "%" SCNx64, &x); - return x; + return parseNumberPrefix(v, 16).orDefault(0); } Key keyForInbox(uint64_t inbox) { diff --git a/flow/CoroTests.cpp b/flow/CoroTests.cpp index 2a07f680ad8..bf38e4e25ee 100644 --- a/flow/CoroTests.cpp +++ b/flow/CoroTests.cpp @@ -2126,6 +2126,8 @@ void assertNoThrowOnCancelDestroyedAfterFirstWait(NoThrowOnCancelRecorder const& } template +// The test log outlives the awaited or explicitly cancelled coroutine. +// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future simple_await_test(std::stringstream& ss, Future f) { ss << "start. "; LifetimeLogger ll(ss, 0); @@ -2138,6 +2140,8 @@ Future simple_await_test(std::stringstream& ss, Future f) { ss << "after co_return. "; } +// The test log outlives the awaited or explicitly cancelled coroutine. +// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future actor_cancel_test(std::stringstream& ss) { ss << "start. "; @@ -2161,6 +2165,8 @@ Future actor_cancel_test(std::stringstream& ss) { ss << "after co_return. "; } +// Test futures finish or cancel before their event recorder is destroyed. +// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future noThrowOnCancelTest(NoThrowOnCancelRecorder& recorder, Future signal, NoThrowOnCancel = {}) { recorder.record(NoThrowOnCancelEvent::Start); @@ -2189,6 +2195,8 @@ Future noThrowOnCancelReentrantCancelTest(Future* result, co_await signal; } +// Test futures finish or cancel before their event recorder is destroyed. +// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future noThrowOnCancelValueTest(NoThrowOnCancelRecorder& recorder, Future signal, NoThrowOnCancel = {}) { recorder.record(NoThrowOnCancelEvent::Start); @@ -2200,6 +2208,8 @@ Future noThrowOnCancelValueTest(NoThrowOnCancelRecorder& recorder, Future noThrowOnCancelSequentialAwaitsTest(NoThrowOnCancelRecorder& recorder, Future firstSignal, Future secondSignal, @@ -2214,6 +2224,8 @@ Future noThrowOnCancelSequentialAwaitsTest(NoThrowOnCancelRecorder& record recorder.record(NoThrowOnCancelEvent::AfterWait); } +// Test futures finish or cancel before their event recorder is destroyed. +// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future noThrowOnCancelFutureStreamTest(NoThrowOnCancelRecorder& recorder, FutureStream stream, NoThrowOnCancel = {}) { @@ -2227,6 +2239,8 @@ Future noThrowOnCancelFutureStreamTest(NoThrowOnCancelRecorder& recorder, co_return value; } +// Test futures finish or cancel before their event recorder is destroyed. +// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future noThrowOnCancelThreadFutureStreamTest(NoThrowOnCancelRecorder& recorder, ThreadFutureStream stream, NoThrowOnCancel = {}) { @@ -2240,6 +2254,8 @@ Future noThrowOnCancelThreadFutureStreamTest(NoThrowOnCancelRecorder& recor co_return value; } +// The test log outlives the awaited or explicitly cancelled coroutine. +// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future actor_throw_test(std::stringstream& ss) { ss << "start. "; diff --git a/flow/IThreadPoolTest.cpp b/flow/IThreadPoolTest.cpp index 0f7a1a3ea34..1169d8a2a80 100644 --- a/flow/IThreadPoolTest.cpp +++ b/flow/IThreadPoolTest.cpp @@ -64,7 +64,7 @@ Future getThreadName(Reference pool) { return fut; } -Future waitForThreadName(Reference pool, std::string const& expectedName) { +Future waitForThreadName(Reference pool, std::string expectedName) { // startThread() sets the pthread name from the creating thread after pthread_create(), so the worker may // briefly report its default name before the requested name is visible. Some environments also report // ENOENT from pthread_setname_np(), which startThread() treats as non-fatal. diff --git a/flow/IndexedSet.cpp b/flow/IndexedSet.cpp index 5582a3d1ac1..e5072f39a2f 100644 --- a/flow/IndexedSet.cpp +++ b/flow/IndexedSet.cpp @@ -372,11 +372,13 @@ TEST_CASE("performance/flow/IndexedSet/strings") { printf("%0.1f Map.KfindStr/sec\n", count / 1000.0 / (end - start)); + tt = 0; start = timer(); for (size_t i = 0; i < count; i++) { - aMap.find(hello); + tt += aMap.find(hello)->second; } end = timer(); + ASSERT(tt == count); printf("%0.1f std::map.KfindStr/sec\n", count / 1000.0 / (end - start)); return Void(); diff --git a/flow/Net2.cpp b/flow/Net2.cpp index 4a6cccf8668..ed8f327d5f8 100644 --- a/flow/Net2.cpp +++ b/flow/Net2.cpp @@ -579,6 +579,8 @@ class Connection final : public IConnection, ReferenceCounted { void closeSocket() { boost::system::error_code error; + // The same error is returned through the output parameter and checked below. + // NOLINTNEXTLINE(bugprone-unused-return-value) socket.close(error); if (error) { TraceEvent(SevWarn, "N2_CloseError", id) @@ -726,6 +728,8 @@ class UDPSocket : public IUDPSocket, ReferenceCounted { void bind(NetworkAddress const& addr) override { boost::system::error_code ec; + // The same error is returned through the output parameter and checked below. + // NOLINTNEXTLINE(bugprone-unused-return-value) socket.bind(udpEndpoint(addr), ec); if (ec) { Error x; @@ -759,6 +763,8 @@ class UDPSocket : public IUDPSocket, ReferenceCounted { void closeSocket() { boost::system::error_code error; + // The same error is returned through the output parameter and checked below. + // NOLINTNEXTLINE(bugprone-unused-return-value) socket.close(error); if (error) { TraceEvent(SevWarn, "N2_CloseError", id) @@ -865,6 +871,8 @@ struct SSLHandshakerThread final : IThreadPoolReceiver { void action(Handshake& h) { try { + // Each operation returns the same error through h.err, which gates the next operation. + // NOLINTBEGIN(bugprone-unused-return-value) h.socket.next_layer().non_blocking(false, h.err); if (!h.err.failed()) { h.socket.handshake(h.type, h.err); @@ -872,6 +880,7 @@ struct SSLHandshakerThread final : IThreadPoolReceiver { if (!h.err.failed()) { h.socket.next_layer().non_blocking(true, h.err); } + // NOLINTEND(bugprone-unused-return-value) if (h.err.failed()) { TraceEvent(SevWarn, h.type == ssl_socket::handshake_type::client ? "N2_ConnectHandshakeError"_audit @@ -1280,12 +1289,15 @@ class SSLConnection final : public IConnection, ReferenceCounted } void closeSocket() { + // Teardown is best effort; errors cannot leave the connection usable. + // NOLINTBEGIN(bugprone-unused-return-value) boost::system::error_code cancelError; socket.cancel(cancelError); boost::system::error_code closeError; socket.close(closeError); boost::system::error_code shutdownError; ssl_sock.shutdown(shutdownError); + // NOLINTEND(bugprone-unused-return-value) } void onReadError(const boost::system::error_code& error) { @@ -2068,18 +2080,24 @@ static Future coordinatorDNSCacheRefresh(Net2* self) { } } -Future> Net2::resolveTCPEndpointWithDNSCache(const std::string& host, - const std::string& service) { +static Future> resolveTCPEndpointWithDNSCacheImpl(Net2* self, + std::string host, + std::string service) { if (FLOW_KNOBS->ENABLE_COORDINATOR_DNS_CACHE) { - Optional> cache = dnsCache.find(host, service); + Optional> cache = self->dnsCache.find(host, service); if (cache.present()) { co_return cache.get(); } - std::vector addresses = co_await resolveTCPEndpoint_impl(this, host, service); - dnsCache.add(host, service, addresses); + std::vector addresses = co_await resolveTCPEndpoint_impl(self, host, service); + self->dnsCache.add(host, service, addresses); co_return addresses; } - co_return co_await resolveTCPEndpoint_impl(this, host, service); + co_return co_await resolveTCPEndpoint_impl(self, host, service); +} + +Future> Net2::resolveTCPEndpointWithDNSCache(const std::string& host, + const std::string& service) { + return resolveTCPEndpointWithDNSCacheImpl(this, host, service); } std::vector Net2::resolveTCPEndpointBlocking(const std::string& host, const std::string& service) { diff --git a/flow/Platform.cpp b/flow/Platform.cpp index 2e8fcd3a235..c6a4a23ca2b 100644 --- a/flow/Platform.cpp +++ b/flow/Platform.cpp @@ -24,6 +24,7 @@ #endif // _WIN32 #include "flow/Platform.h" +#include "flow/ParseNumber.h" #include #include @@ -1527,7 +1528,8 @@ void initPdhStrings(SystemStatisticsState* state, std::string dataFolder) { "PdhEnumObjectItems")) { char* ptr = buf; while (*ptr) { - if (isdigit(*ptr) && atoi(ptr) == storage_device.DeviceNumber) { + auto deviceNumber = parseNumberPrefix(StringRef(static_cast(ptr))); + if (isdigit(*ptr) && deviceNumber.present() && deviceNumber.get() == storage_device.DeviceNumber) { state->pdhStrings.diskDevice = ptr; break; } @@ -2913,7 +2915,10 @@ THREAD_HANDLE startThread(void* (*func)(void*), void* arg, int stackSize, const pthread_t t; pthread_attr_t attr; - pthread_attr_init(&attr); + int attrError = pthread_attr_init(&attr); + if (attrError != 0) { + criticalError(FDB_EXIT_ERROR, "ThreadAttributesError", strerror(attrError)); + } if (stackSize != 0) { if (pthread_attr_setstacksize(&attr, stackSize) != 0) { // If setting the stack size fails the default stack size will be used, so failure to set @@ -2926,8 +2931,13 @@ THREAD_HANDLE startThread(void* (*func)(void*), void* arg, int stackSize, const } auto* args = new ThreadCreateArgs(func, arg); - pthread_create(&t, &attr, &runFunc, args); + int createError = pthread_create(&t, &attr, &runFunc, args); pthread_attr_destroy(&attr); + if (createError != 0) { + delete args; + // Callers transfer thread-owned state without a failed-launch recovery path. + criticalError(FDB_EXIT_ERROR, "ThreadCreationError", strerror(createError)); + } #if defined(__linux__) if (name != nullptr) { diff --git a/flow/Profiler.cpp b/flow/Profiler.cpp index 2fba014e50d..b412aae55f6 100644 --- a/flow/Profiler.cpp +++ b/flow/Profiler.cpp @@ -20,6 +20,7 @@ #include "flow/flow.h" #include "flow/network.h" +#include "flow/ParseNumber.h" #ifdef __linux__ @@ -275,7 +276,7 @@ void startProfiling(INetwork* network, period = maybePeriod.get(); } else { const char* periodEnv = getenv("FLOW_PROFILER_PERIOD"); - period = (periodEnv ? atoi(periodEnv) : 2000); + period = (periodEnv ? parseNumberPrefix(StringRef(periodEnv)).orDefault(0) : 2000); } std::string outputFile; if (maybeOutputFile.present()) { diff --git a/flow/Trace.cpp b/flow/Trace.cpp index 8c3bf1c8cd2..0738e6b4249 100644 --- a/flow/Trace.cpp +++ b/flow/Trace.cpp @@ -26,6 +26,9 @@ #include "flow/flow.h" #include "flow/DeterministicRandom.h" #include "flow/ProcessEvents.h" +#include "flow/UnitTest.h" +#include +#include #include #include #include @@ -1681,56 +1684,52 @@ TraceEventFields::Field& TraceEventFields::mutate(int index) { } namespace { -void parseNumericValue(std::string const& s, double& outValue, bool permissive = false) { - double d = 0; - int consumed = 0; - int r = sscanf(s.c_str(), "%lf%n", &d, &consumed); - if (r == 1 && (consumed == s.size() || permissive)) { - outValue = d; - return; +void checkNumericConversion(std::string const& s, const char* end, bool permissive, int conversionError) { + if (end == s.c_str() || (!permissive && end != s.c_str() + s.size())) { + throw attribute_not_found(); } - - throw attribute_not_found(); -} - -void parseNumericValue(std::string const& s, int& outValue, bool permissive = false) { - long long int iLong = 0; - int consumed = 0; - int r = sscanf(s.c_str(), "%lld%n", &iLong, &consumed); - if (r == 1 && (consumed == s.size() || permissive)) { - if (std::numeric_limits::min() <= iLong && iLong <= std::numeric_limits::max()) { - outValue = (int)iLong; // Downcast definitely safe - return; - } else { - throw attribute_too_large(); - } + if (conversionError == ERANGE) { + throw attribute_too_large(); } +} - throw attribute_not_found(); +void parseNumericValue(std::string const& s, double& outValue, bool permissive = false) { + char* end = nullptr; + errno = 0; + double d = std::strtod(s.c_str(), &end); + checkNumericConversion(s, end, permissive, errno); + outValue = d; } void parseNumericValue(std::string const& s, int64_t& outValue, bool permissive = false) { - long long int i = 0; - int consumed = 0; - int r = sscanf(s.c_str(), "%lld%n", &i, &consumed); - if (r == 1 && (consumed == s.size() || permissive)) { - outValue = i; - return; + char* end = nullptr; + errno = 0; + long long i = std::strtoll(s.c_str(), &end, 10); + checkNumericConversion(s, end, permissive, errno); + if (i < std::numeric_limits::min() || i > std::numeric_limits::max()) { + throw attribute_too_large(); } + outValue = i; +} - throw attribute_not_found(); +void parseNumericValue(std::string const& s, int& outValue, bool permissive = false) { + int64_t i; + parseNumericValue(s, i, permissive); + if (i < std::numeric_limits::min() || i > std::numeric_limits::max()) { + throw attribute_too_large(); + } + outValue = static_cast(i); } void parseNumericValue(std::string const& s, uint64_t& outValue, bool permissive = false) { - unsigned long long int i = 0; - int consumed = 0; - int r = sscanf(s.c_str(), "%llu%n", &i, &consumed); - if (r == 1 && (consumed == s.size() || permissive)) { - outValue = i; - return; + char* end = nullptr; + errno = 0; + unsigned long long i = (std::strtoull)(s.c_str(), &end, 10); + checkNumericConversion(s, end, permissive, errno); + if (i > std::numeric_limits::max()) { + throw attribute_too_large(); } - - throw attribute_not_found(); + outValue = i; } template @@ -1758,6 +1757,72 @@ bool getNumericValue(TraceEventFields const& fields, std::string key, T& outValu } } } + +TEST_CASE("/flow/TraceEventFields/numericParsing") { + auto expectFailure = [](const std::string& text, auto initial, int errorCode, bool permissive = false) { + auto value = initial; + try { + parseNumericValue(text, value, permissive); + } catch (Error& e) { + ASSERT_EQ(e.code(), errorCode); + ASSERT_EQ(value, initial); + return; + } + ASSERT(false); + }; + + int64_t signedValue = 0; + parseNumericValue(" \t+0012", signedValue); + ASSERT_EQ(signedValue, int64_t{ 12 }); + parseNumericValue("-9223372036854775808", signedValue); + ASSERT_EQ(signedValue, std::numeric_limits::min()); + parseNumericValue("9223372036854775807", signedValue); + ASSERT_EQ(signedValue, std::numeric_limits::max()); + expectFailure("9223372036854775808", int64_t{ 123 }, error_code_attribute_too_large); + expectFailure("-9223372036854775809", int64_t{ 123 }, error_code_attribute_too_large); + + int intValue = 0; + parseNumericValue(std::to_string(std::numeric_limits::min()), intValue); + ASSERT_EQ(intValue, std::numeric_limits::min()); + parseNumericValue(std::to_string(std::numeric_limits::max()), intValue); + ASSERT_EQ(intValue, std::numeric_limits::max()); + expectFailure( + std::to_string(static_cast(std::numeric_limits::max()) + 1), 123, error_code_attribute_too_large); + + uint64_t unsignedValue = 0; + parseNumericValue(" \t+18446744073709551615", unsignedValue); + ASSERT_EQ(unsignedValue, std::numeric_limits::max()); + parseNumericValue("-1", unsignedValue); + ASSERT_EQ(unsignedValue, std::numeric_limits::max()); + expectFailure("18446744073709551616", uint64_t{ 123 }, error_code_attribute_too_large); + + double doubleValue = 0; + parseNumericValue(" \t+1.25e2", doubleValue); + ASSERT_EQ(doubleValue, 125.0); + expectFailure("1e9999", 123.0, error_code_attribute_too_large); + expectFailure("1e-9999", 123.0, error_code_attribute_too_large); + expectFailure("", 123.0, error_code_attribute_not_found); + expectFailure(" \t", int64_t{ 123 }, error_code_attribute_not_found, true); + expectFailure("12suffix", 123, error_code_attribute_not_found); + parseNumericValue("12suffix", intValue, true); + ASSERT_EQ(intValue, 12); + expectFailure(std::string("12\0suffix", 9), 123, error_code_attribute_not_found); + parseNumericValue(std::string("12\0suffix", 9), intValue, true); + ASSERT_EQ(intValue, 12); + expectFailure("9223372036854775808suffix", int64_t{ 123 }, error_code_attribute_not_found); + expectFailure("9223372036854775808suffix", int64_t{ 123 }, error_code_attribute_too_large, true); + + TraceEventFields fields; + fields.addField("overflow", "18446744073709551616"); + unsignedValue = 123; + ASSERT(!fields.tryGetUint64("overflow", unsignedValue)); + ASSERT_EQ(unsignedValue, uint64_t{ 123 }); + fields.addField("number", "12suffix"); + ASSERT(!fields.tryGetInt("number", intValue)); + ASSERT_EQ(fields.getInt("number", true), 12); + + return Void(); +} } // namespace bool TraceEventFields::tryGetInt(std::string key, int& outVal, bool permissive) const { diff --git a/flow/UnitTest.cpp b/flow/UnitTest.cpp index 9d3cd67ad5f..189e9fe4f53 100644 --- a/flow/UnitTest.cpp +++ b/flow/UnitTest.cpp @@ -19,6 +19,12 @@ */ #include "flow/UnitTest.h" +#include "flow/ParseNumber.h" + +#include +#include +#include +#include UnitTestCollection g_unittests = { nullptr }; @@ -51,7 +57,11 @@ void UnitTestParameters::set(const std::string& name, double value) { Optional UnitTestParameters::getInt(const std::string& name) const { auto opt = get(name); if (opt.present()) { - return atoll(opt.get().c_str()); + auto parsed = parseNumber(opt.get()); + if (!parsed.present()) { + throw invalid_option_value(); + } + return parsed; } return {}; } @@ -59,7 +69,11 @@ Optional UnitTestParameters::getInt(const std::string& name) const { Optional UnitTestParameters::getDouble(const std::string& name) const { auto opt = get(name); if (opt.present()) { - return atof(opt.get().c_str()); + auto parsed = parseNumber(opt.get()); + if (!parsed.present()) { + throw invalid_option_value(); + } + return parsed; } return {}; } @@ -71,3 +85,123 @@ std::string UnitTestParameters::getDataDir() const { void UnitTestParameters::setDataDir(std::string const& dataDir) { this->dataDir = dataDir; } + +TEST_CASE("/flow/ParseNumber/checked") { + int consumed = -1; + ASSERT_EQ(parseNumberPrefix(" \t+12suffix"_sr, 10, &consumed).get(), 12); + ASSERT_EQ(consumed, 5); + ASSERT(!parseNumber(" \t+12suffix"_sr).present()); + ASSERT_EQ(parseNumber(" \t+12"_sr).get(), 12); + ASSERT(!parseNumber("12 "_sr).present()); + ASSERT(!parseNumber("12\0suffix"_sr).present()); + ASSERT_EQ(parseNumberPrefix("12\0suffix"_sr).get(), 12); + ASSERT_EQ(parseNumber("ffffffffffffffff"_sr, 16).get(), std::numeric_limits::max()); + ASSERT_EQ(parseNumber("0xff"_sr, 0).get(), 255); + ASSERT_EQ(parseNumber("ff"_sr, 16).get(), 255); + ASSERT_EQ(parseNumber("010"_sr).get(), 10); + ASSERT_EQ(parseNumber("-1"_sr).get(), std::numeric_limits::max()); + ASSERT(!parseNumber("-1"_sr).present()); + ASSERT(!parseNumber("256"_sr).present()); + ASSERT(!parseNumber("128"_sr).present()); + ASSERT(!parseNumber("-129"_sr).present()); + ASSERT_EQ(parseNumber("0.1"_sr).get(), 0.1f); + ASSERT_EQ(parseNumber("0.1"_sr).get(), 0.1); + ASSERT(parseNumber("1.25"_sr).get() == 1.25L); + ASSERT_EQ(parseNumber("0x1p2"_sr).get(), 4.0); + ASSERT(!parseNumber("1e9999"_sr).present()); + ASSERT(!parseNumber("1e-9999"_sr).present()); + ASSERT(!parseNumber("1"_sr, 1).present()); + ASSERT(!parseNumber("1"_sr, 16).present()); + for (StringRef text : { ""_sr, "+"_sr, " \t"_sr, "9223372036854775808"_sr }) { + consumed = -1; + ASSERT(!parseNumberPrefix(text, 10, &consumed).present()); + ASSERT_EQ(consumed, -1); + } + const uint8_t backing[] = { '1', '2', '3', 0 }; + ASSERT_EQ(parseNumber(StringRef(backing, 2)).get(), 12); + return Void(); +} + +TEST_CASE("/flow/UnitTestParameters/numericValues") { + UnitTestParameters numericParams; + ASSERT(!numericParams.getInt("missing").present()); + ASSERT(!numericParams.getDouble("missing").present()); + + const int64_t intMin = std::numeric_limits::min(); + const int64_t intMax = std::numeric_limits::max(); + for (const auto& [text, expected] : + std::vector>{ { "0", 0 }, + { " \t+0012", 12 }, + { "-1", -1 }, + { std::to_string(intMin), intMin }, + { std::to_string(intMax), intMax } }) { + numericParams.set("integer", text); + errno = ERANGE; + ASSERT_EQ(numericParams.getInt("integer").get(), expected); + } + for (const std::string& text : std::vector{ + "", " \t", "+", "1x", "1 ", "9223372036854775808", "-9223372036854775809", std::string("12\0junk", 7) }) { + numericParams.set("integer", text); + try { + (void)numericParams.getInt("integer"); + ASSERT(false); + } catch (Error& e) { + ASSERT_EQ(e.code(), error_code_invalid_option_value); + } + } + + for (const auto& [text, expected] : std::vector>{ + { "0", 0.0 }, + { " \t+1.25e2", 125.0 }, + { "-0x1p2", -4.0 }, + { ".5", 0.5 }, + { "1.7976931348623157e308", std::numeric_limits::max() }, + { "2.2250738585072014e-308", std::numeric_limits::min() } }) { + numericParams.set("double", text); + errno = ERANGE; + ASSERT_EQ(numericParams.getDouble("double").get(), expected); + } + numericParams.set("double", std::string("nan")); + ASSERT(std::isnan(numericParams.getDouble("double").get())); + numericParams.set("double", std::string("inf")); + ASSERT(std::isinf(numericParams.getDouble("double").get())); + numericParams.set("double", std::string("-0")); + ASSERT(std::signbit(numericParams.getDouble("double").get())); + for (const std::string& text : + std::vector{ "", " \t", "+", "1x", "1 ", "1e9999", "1e-9999", std::string("12\0junk", 7) }) { + numericParams.set("double", text); + try { + (void)numericParams.getDouble("double"); + ASSERT(false); + } catch (Error& e) { + ASSERT_EQ(e.code(), error_code_invalid_option_value); + } + } + + return Void(); +} + +TEST_CASE("/flow/UnitTestParameters/coroutineOwnership") { + const std::string marker = "unitTestParameterOwnershipProbe"; + if (params.get(marker).present()) { + co_await delay(0.001); + ASSERT_EQ(params.get(marker).get(), std::string("original")); + co_return; + } + + UnitTest* registered = g_unittests.tests; + while (registered != nullptr && StringRef(registered->name) != "/flow/UnitTestParameters/coroutineOwnership"_sr) { + registered = registered->next; + } + ASSERT(registered != nullptr); + + Future pending; + { + UnitTestParameters callerParams; + callerParams.set(marker, std::string("original")); + pending = registered->func(callerParams); + ASSERT(!pending.isReady()); + callerParams.set(marker, std::string("changed")); + } + co_await pending; +} diff --git a/flow/UnitTestRunner.cpp b/flow/UnitTestRunner.cpp index a93e724e146..f5cf9e32d1f 100644 --- a/flow/UnitTestRunner.cpp +++ b/flow/UnitTestRunner.cpp @@ -31,10 +31,13 @@ #include #include +#include +#include #include #include #include #include +#include #include #include #include @@ -112,8 +115,10 @@ void printUsage(const char* program, const UnitTestRunnerConfig& config) { bool parseInt(const char* text, int* value) { char* end = nullptr; + errno = 0; long parsed = strtol(text, &end, 10); - if (*text == '\0' || *end != '\0') { + if (end == text || *end != '\0' || errno == ERANGE || parsed < std::numeric_limits::min() || + parsed > std::numeric_limits::max()) { return false; } *value = static_cast(parsed); @@ -121,15 +126,75 @@ bool parseInt(const char* text, int* value) { } bool parseUInt64(const char* text, uint64_t* value) { + const char* first = text; + while (std::isspace(static_cast(*first))) { + ++first; + } + if (*first == '-') { + return false; + } + char* end = nullptr; - uint64_t parsed = strtoull(text, &end, 10); - if (*text == '\0' || *end != '\0') { + errno = 0; + unsigned long long parsed = strtoull(text, &end, 10); + if (end == text || *end != '\0' || errno == ERANGE || parsed > std::numeric_limits::max()) { return false; } *value = parsed; return true; } +TEST_CASE("/flow/UnitTestRunner/numericOptions") { + const int intMin = std::numeric_limits::min(); + const int intMax = std::numeric_limits::max(); + for (const auto& [text, expected] : + std::vector>{ { "0", 0 }, + { " \t+0012", 12 }, + { "-1", -1 }, + { std::to_string(intMin), intMin }, + { std::to_string(intMax), intMax } }) { + int value = 123; + ASSERT(parseInt(text.c_str(), &value)); + ASSERT_EQ(value, expected); + } + for (const std::string& text : std::vector{ "", + " \t", + "+", + "1x", + "1 ", + std::to_string(static_cast(intMin) - 1), + std::to_string(static_cast(intMax) + 1), + "999999999999999999999999999999" }) { + int value = 123; + ASSERT(!parseInt(text.c_str(), &value)); + ASSERT_EQ(value, 123); + } + + const uint64_t uintMax = std::numeric_limits::max(); + for (const auto& [text, expected] : std::vector>{ + { "0", 0 }, { " \t+0012", 12 }, { std::to_string(uintMax), uintMax } }) { + uint64_t value = 123; + ASSERT(parseUInt64(text.c_str(), &value)); + ASSERT_EQ(value, expected); + } + for (const char* text : { "", + " \t", + "+", + "1x", + "1 ", + "-1", + " \t-1", + "-0", + "18446744073709551616", + "999999999999999999999999999999" }) { + uint64_t value = 123; + ASSERT(!parseUInt64(text, &value)); + ASSERT_EQ(value, uint64_t{ 123 }); + } + + return Void(); +} + bool parseArgs(int argc, char** argv, UnitTestRunnerOptions* options) { CSimpleOpt args(argc, argv, unitTestRunnerOptions, SO_O_EXACT | SO_O_HYPHEN_TO_UNDERSCORE); while (args.Next()) { @@ -284,15 +349,15 @@ std::vector collectTests(const UnitTestRunnerOptions& options, const } } + // The comparator orders tests by their names, independent of their addresses. + // NOLINTNEXTLINE(bugprone-nondeterministic-pointer-iteration-order) std::sort(tests.begin(), tests.end(), [](auto lhs, auto rhs) { return std::string_view(lhs->name) < std::string_view(rhs->name); }); return tests; } -Future runTests(const UnitTestRunnerOptions& options, - const UnitTestRunnerConfig& config, - UnitTestRunnerResult* result) { +Future runTests(UnitTestRunnerOptions options, UnitTestRunnerConfig config, UnitTestRunnerResult* result) { std::vector tests = collectTests(options, config); result->testsAvailable = tests.size(); @@ -364,11 +429,11 @@ Future runTests(const UnitTestRunnerOptions& options, } Future runTestsAfterInitialization(Future initialization, - const UnitTestRunnerOptions& options, - const UnitTestRunnerConfig& config, + UnitTestRunnerOptions options, + UnitTestRunnerConfig config, UnitTestRunnerResult* result) { co_await initialization; - co_await runTests(options, config, result); + co_await runTests(std::move(options), std::move(config), result); } Future stopNetworkAfter(Future what, std::string_view traceName, int* exitCode) { diff --git a/flow/bench/BenchAsyncResult.cpp b/flow/bench/BenchAsyncResult.cpp index 7585e394eac..4e95c71e783 100644 --- a/flow/bench/BenchAsyncResult.cpp +++ b/flow/bench/BenchAsyncResult.cpp @@ -43,10 +43,14 @@ struct ExpensivePayload { } }; +// This eager coroutine copies the payload into its result without suspending. +// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future returnFuturePayload(ExpensivePayload const& payload) { co_return payload; } +// This eager coroutine copies the payload into its result without suspending. +// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) AsyncResult returnAsyncResultPayload(ExpensivePayload const& payload) { co_return payload; } diff --git a/flow/flow.cpp b/flow/flow.cpp index 4dc1e790608..2ddd637b40f 100644 --- a/flow/flow.cpp +++ b/flow/flow.cpp @@ -32,6 +32,7 @@ #include "flow/DeterministicRandom.h" #include "flow/Error.h" #include "flow/Hostname.h" +#include "flow/ParseNumber.h" #include "flow/Util.h" #include "rte_memcpy.h" #include "flow/UnitTest.h" @@ -138,10 +139,10 @@ std::string UID::toString() const { UID UID::fromString(std::string const& s) { ASSERT_EQ(s.size(), 32); - uint64_t a = 0, b = 0; - int r = sscanf(s.c_str(), "%16" SCNx64 "%16" SCNx64, &a, &b); - ASSERT_EQ(r, 2); - return UID(a, b); + auto a = parseNumber(StringRef(s).substr(0, 16), 16); + auto b = parseNumber(StringRef(s).substr(16), 16); + ASSERT(a.present() && b.present()); + return UID(a.get(), b.get()); } UID UID::fromStringThrowsOnFailure(std::string const& s) { @@ -149,20 +150,31 @@ UID UID::fromStringThrowsOnFailure(std::string const& s) { // invalid string size throw operation_failed(); } - // Split into two 16-character hex strings and parse using strtoull - std::string first_half = s.substr(0, 16); - std::string second_half = s.substr(16, 16); - - char* end1; - char* end2; - uint64_t a = strtoull(first_half.c_str(), &end1, 16); - uint64_t b = strtoull(second_half.c_str(), &end2, 16); - - // Verify entire strings were parsed - if (end1 != first_half.c_str() + 16 || end2 != second_half.c_str() + 16) { + auto a = parseNumber(StringRef(s).substr(0, 16), 16); + auto b = parseNumber(StringRef(s).substr(16), 16); + if (!a.present() || !b.present()) { throw operation_failed(); } - return UID(a, b); + return UID(a.get(), b.get()); +} + +TEST_CASE("/flow/UID/parse") { + const UID id(0, std::numeric_limits::max()); + ASSERT(UID::fromString(id.toString()) == id); + ASSERT(UID::fromStringThrowsOnFailure(id.toString()) == id); + ASSERT(UID::fromString("0123456789ABCDEFfedcba9876543210") == UID(0x0123456789abcdef, 0xfedcba9876543210)); + std::string embeddedNul(32, '0'); + embeddedNul[15] = '\0'; + for (const std::string& invalid : + { std::string(31, '0'), std::string(32, 'g'), std::string(31, '0') + "g", embeddedNul }) { + try { + (void)UID::fromStringThrowsOnFailure(invalid); + ASSERT(false); + } catch (Error& e) { + ASSERT_EQ(e.code(), error_code_operation_failed); + } + } + return Void(); } std::string UID::shortString() const { diff --git a/flow/include/flow/IDispatched.h b/flow/include/flow/IDispatched.h index 2ca06bdf26e..9ea9263c9b5 100644 --- a/flow/include/flow/IDispatched.h +++ b/flow/include/flow/IDispatched.h @@ -44,7 +44,7 @@ struct IDispatched { #define REGISTER_DISPATCHED(Type, Instance, Key, Func) \ struct Type##Instance { \ Type##Instance() { \ - ASSERT(Type::dispatches().find(Key) == Type::dispatches().end()); \ + ASSERT(!Type::dispatches().contains(Key)); \ Type::dispatches()[Key] = Func; \ } \ }; \ @@ -52,8 +52,8 @@ struct IDispatched { #define REGISTER_DISPATCHED_ALIAS(Type, Instance, Target, Alias) \ struct Type##Instance { \ Type##Instance() { \ - ASSERT(Type::dispatches().find(Alias) == Type::dispatches().end()); \ - ASSERT(Type::dispatches().find(Target) != Type::dispatches().end()); \ + ASSERT(!Type::dispatches().contains(Alias)); \ + ASSERT(Type::dispatches().contains(Target)); \ Type::dispatches()[Alias] = Type::dispatches()[Target]; \ } \ }; \ diff --git a/flow/include/flow/ParseNumber.h b/flow/include/flow/ParseNumber.h new file mode 100644 index 00000000000..f1a592db5c9 --- /dev/null +++ b/flow/include/flow/ParseNumber.h @@ -0,0 +1,92 @@ +/* + * ParseNumber.h + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#ifndef FLOW_PARSE_NUMBER_H +#define FLOW_PARSE_NUMBER_H +#pragma once + +#include "flow/Arena.h" + +#include +#include +#include +#include + +// Parses a numeric prefix without reading beyond input. Leading C whitespace and signs are accepted. +// Returns absent for a missing number or a conversion outside T's range. On success, consumed includes +// leading whitespace; on failure it is unchanged. Integer bases are 0 or 2..36. Floating-point input +// uses the C strto* grammar and requires base 10. Unsigned conversions retain strtoull's sign wrapping +// before checking T's range, so uint64_t accepts "-1" as UINT64_MAX. +template +Optional parseNumberPrefix(StringRef input, int base = 10, int* consumed = nullptr) { + static_assert((std::is_integral_v && !std::is_same_v) || std::is_floating_point_v); + if constexpr (std::is_integral_v) { + if (base != 0 && (base < 2 || base > 36)) { + return {}; + } + } else if (base != 10) { + return {}; + } + + const std::string text = input.toString(); + char* end = nullptr; + errno = 0; + T result; + if constexpr (std::is_same_v) { + result = std::strtof(text.c_str(), &end); + } else if constexpr (std::is_same_v) { + result = std::strtod(text.c_str(), &end); + } else if constexpr (std::is_same_v) { + result = std::strtold(text.c_str(), &end); + } else if constexpr (std::is_signed_v) { + const long long parsed = std::strtoll(text.c_str(), &end, base); + if (parsed < std::numeric_limits::min() || parsed > std::numeric_limits::max()) { + return {}; + } + result = static_cast(parsed); + } else { + const unsigned long long parsed = (std::strtoull)(text.c_str(), &end, base); + if (parsed > std::numeric_limits::max()) { + return {}; + } + result = static_cast(parsed); + } + if (errno == ERANGE || end == text.c_str()) { + return {}; + } + if (consumed != nullptr) { + *consumed = static_cast(end - text.c_str()); + } + return result; +} + +// Like parseNumberPrefix, but requires all input bytes to be part of the number. Trailing whitespace, +// other suffixes, and embedded NUL bytes are rejected. +template +Optional parseNumber(StringRef input, int base = 10) { + int consumed = 0; + auto result = parseNumberPrefix(input, base, &consumed); + if (!result.present() || consumed != input.size()) { + return {}; + } + return result; +} + +#endif diff --git a/flow/include/flow/UnitTest.h b/flow/include/flow/UnitTest.h index e786f4fb5d2..eac90a88778 100644 --- a/flow/include/flow/UnitTest.h +++ b/flow/include/flow/UnitTest.h @@ -47,7 +47,8 @@ * } * * The body of a TEST_CASE returns a Future. It may be an ordinary function - * or a C++ coroutine using `co_await` and `co_return`. + * or a C++ coroutine using `co_await` and `co_return`. Each test body receives its + * own const copy of the parameters; a coroutine keeps that copy in its frame. * * Our tools for actually executing tests are external to flow (and use g_unittests to find test cases). * See the `UnitTestWorkload` class. @@ -76,10 +77,12 @@ class UnitTestParameters { // Get a parameter's value, will return !present() if parameter was not set Optional get(const std::string& name) const; - // Get a parameter's value as an integer, will return !present() if parameter was not set + // Get a parameter's value as an integer, returning !present() if it was not set. + // Throws invalid_option_value if a present value is malformed or outside the int64_t range. Optional getInt(const std::string& name) const; - // Get a parameter's value parsed as a double, will return !present() if parameter was not set + // Get a parameter's value parsed as a double, returning !present() if it was not set. + // Throws invalid_option_value if a present value is malformed or outside the double range. Optional getDouble(const std::string& name) const; // This is separate because it assumes data directory has already been set, and doesn't return an optional @@ -120,16 +123,19 @@ extern bool noUnseed; #ifdef FLOW_DISABLE_UNIT_TESTS -#define TEST_CASE(name) static Future FILE_UNIQUE_NAME(disabled_testcase_func)(const UnitTestParameters& params) +#define TEST_CASE(name) static Future FILE_UNIQUE_NAME(disabled_testcase_func)(const UnitTestParameters params) #else #define TEST_CASE(name) \ - static Future FILE_UNIQUE_NAME(testcase_func)(const UnitTestParameters& params); \ + static Future FILE_UNIQUE_NAME(testcase_impl)(const UnitTestParameters params); \ + static Future FILE_UNIQUE_NAME(testcase_func)(const UnitTestParameters& params) { \ + return FILE_UNIQUE_NAME(testcase_impl)(params); \ + } \ namespace { \ static UnitTest FILE_UNIQUE_NAME(testcase)(name, __FILE__, __LINE__, &FILE_UNIQUE_NAME(testcase_func)); \ } \ - static Future FILE_UNIQUE_NAME(testcase_func)(const UnitTestParameters& params) + static Future FILE_UNIQUE_NAME(testcase_impl)(const UnitTestParameters params) #endif diff --git a/flow/network.cpp b/flow/network.cpp index 38f097ca2fb..1ae299d30b7 100644 --- a/flow/network.cpp +++ b/flow/network.cpp @@ -26,6 +26,7 @@ #include "flow/ChaosMetrics.h" #include "flow/UnitTest.h" #include "flow/IConnection.h" +#include "flow/ParseNumber.h" ChaosMetrics::ChaosMetrics() { clear(); @@ -189,11 +190,24 @@ NetworkAddress NetworkAddress::parse(std::string const& s) { } return NetworkAddress(addr.get(), port, true, isTLS, fromHostname); } else { - // TODO: Use IPAddress::parse - int a, b, c, d, port, count = -1; - if (sscanf(f.c_str(), "%d.%d.%d.%d:%d%n", &a, &b, &c, &d, &port, &count) < 5 || count != f.size()) + StringRef remaining(f); + uint32_t ip = 0; + for (int component = 0; component < 4; ++component) { + int consumed = 0; + auto octet = parseNumberPrefix(remaining, 10, &consumed); + const char separator = component == 3 ? ':' : '.'; + if (!octet.present() || octet.get() < 0 || octet.get() > 255 || consumed == remaining.size() || + remaining[consumed] != separator) { + throw connection_string_invalid(); + } + ip = (ip << 8) | static_cast(octet.get()); + remaining = remaining.substr(consumed + 1); + } + auto port = parseNumber(remaining); + if (!port.present() || port.get() < 0 || port.get() > std::numeric_limits::max()) { throw connection_string_invalid(); - return NetworkAddress((a << 24) + (b << 16) + (c << 8) + d, port, true, isTLS, fromHostname); + } + return NetworkAddress(ip, port.get(), true, isTLS, fromHostname); } } @@ -427,6 +441,19 @@ IUDPSocket::~IUDPSocket() = default; const std::vector NetworkMetrics::starvationBins = { 1, 3500, 7000, 7500, 8500, 8900, 10500 }; TEST_CASE("/flow/network/ipaddress") { + ASSERT(NetworkAddress::parse(" \t+127. +0.0. +1: +4800").toString() == "127.0.0.1:4800"); + ASSERT(NetworkAddress::parse("255.255.255.255:65535").toString() == "255.255.255.255:65535"); + for (const char* invalid : { "-1.0.0.1:4800", + "256.0.0.1:4800", + "9223372036854775808.0.0.1:4800", + "127.0.0.1:-1", + "127.0.0.1:65536", + "127.0.0.1:9223372036854775808", + "127.0.0.1:4800suffix", + "127.0.0.1:4800 " }) { + ASSERT(!NetworkAddress::parseOptional(invalid).present()); + } + ASSERT(NetworkAddress::parse("[::1]:4800").toString() == "[::1]:4800"); { From 42aca6f23e8e7fcf706ca66e14dc56927097aa8c Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Fri, 18 Sep 2026 14:01:58 -0700 Subject: [PATCH 115/170] Drop numeric-conversion clang-tidy rollout --- .clang-tidy | 1 - bindings/c/test/mako/mako.cpp | 67 +++--- bindings/flow/tester/Tester.cpp | 10 +- documentation/sphinx/source/clang-tidy.rst | 8 +- fdbbackup/FileConverter.cpp | 15 +- fdbbackup/FileDecoder.cpp | 7 +- fdbbackup/backup.cpp | 64 +++--- fdbcli/AdvanceVersionCommand.cpp | 7 +- fdbcli/MaintenanceCommand.cpp | 9 +- fdbcli/SuspendCommand.cpp | 15 +- fdbcli/VersionEpochCommand.cpp | 6 +- fdbcli/fdbcli.cpp | 8 +- fdbclient/BackupAgentBase.cpp | 38 +--- fdbclient/BackupContainerFileSystem.cpp | 195 ++++++------------ fdbclient/BackupContainerLocalDirectory.cpp | 3 +- fdbclient/DatabaseConfiguration.cpp | 14 +- fdbclient/ManagementAPI.cpp | 16 +- fdbclient/include/fdbclient/FDBTypes.h | 3 +- .../include/fdbclient/RandomKeyValueUtils.h | 8 +- fdbmonitor/fdbmonitor.h | 3 - fdbmonitor/fdbmonitor_lib.cpp | 39 +--- fdbmonitor/fdbmonitor_tests.cpp | 35 ---- fdbrpc/HTTP.cpp | 54 +---- fdbrpc/ReplicationUtils.cpp | 30 ++- fdbserver/SimulatedCluster.cpp | 52 ++--- fdbserver/clustercontroller/Status.cpp | 47 +---- fdbserver/commitproxy/CommitProxyServer.cpp | 3 +- fdbserver/core/QuietDatabase.cpp | 51 +---- fdbserver/core/WorkloadKeys.cpp | 4 +- .../datadistributor/DataDistribution.cpp | 12 +- fdbserver/fdbserver.cpp | 32 +-- fdbserver/networktest.cpp | 5 +- fdbserver/tester/TestSpecParser.cpp | 48 +---- fdbserver/tester/WorkloadUtils.cpp | 85 ++------ fdbserver/worker/worker.cpp | 25 +-- fdbserver/workloads/CommitBugCheck.cpp | 3 +- fdbserver/workloads/ConfigureDatabase.cpp | 7 +- fdbserver/workloads/DDMetrics.cpp | 3 +- fdbserver/workloads/FileSystem.cpp | 7 +- fdbserver/workloads/HTTPKeyValueStore.cpp | 10 +- fdbserver/workloads/Inventory.cpp | 5 +- fdbserver/workloads/QueuePush.cpp | 10 +- fdbserver/workloads/SnapTest.cpp | 9 +- fdbserver/workloads/Storefront.cpp | 7 +- fdbserver/workloads/TaskBucketCorrectness.cpp | 7 +- fdbserver/workloads/pubsub.cpp | 5 +- flow/Platform.cpp | 4 +- flow/Profiler.cpp | 3 +- flow/Trace.cpp | 141 ++++--------- flow/UnitTest.cpp | 113 +--------- flow/UnitTestRunner.cpp | 71 +------ flow/flow.cpp | 44 ++-- flow/include/flow/ParseNumber.h | 92 --------- flow/include/flow/UnitTest.h | 6 +- flow/network.cpp | 35 +--- 55 files changed, 383 insertions(+), 1218 deletions(-) delete mode 100644 flow/include/flow/ParseNumber.h diff --git a/.clang-tidy b/.clang-tidy index b4b3693f055..21a96f20d88 100644 --- a/.clang-tidy +++ b/.clang-tidy @@ -39,7 +39,6 @@ Checks: > bugprone-unused-return-value, bugprone-use-after-move, bugprone-virtual-near-miss, - cert-err34-c, cppcoreguidelines-avoid-capturing-lambda-coroutines, cppcoreguidelines-avoid-reference-coroutine-parameters, misc-coroutine-hostile-raii, diff --git a/bindings/c/test/mako/mako.cpp b/bindings/c/test/mako/mako.cpp index 877f1c652f2..0d8f50b0914 100644 --- a/bindings/c/test/mako/mako.cpp +++ b/bindings/c/test/mako/mako.cpp @@ -21,7 +21,6 @@ #include #include #include -#include #include #include #include @@ -29,7 +28,6 @@ #include #include #include -#include #include #include #include @@ -1156,25 +1154,6 @@ void usage() { "Maximum estimated GRV proxy queue delay in milliseconds. Set as transaction option in run mode."); } -// Numeric options retain their historical prefix parsing and zero default for invalid input. -static int parseIntegerArgument(const char* text) { - char* end = nullptr; - errno = 0; - const long long value = std::strtoll(text, &end, 10); - if (end == text || errno == ERANGE || value < std::numeric_limits::min() || - value > std::numeric_limits::max()) { - return 0; - } - return static_cast(value); -} - -static double parseDoubleArgument(const char* text) { - char* end = nullptr; - errno = 0; - const double value = std::strtod(text, &end); - return end == text || errno == ERANGE ? 0 : value; -} - /* parse benchmark parameters */ int parseArguments(int argc, char* argv[], Arguments& args) { int rc; @@ -1266,7 +1245,7 @@ int parseArguments(int argc, char* argv[], Arguments& args) { usage(); return -1; case 'a': - args.api_version = parseIntegerArgument(optarg); + args.api_version = atoi(optarg); break; case 'c': { const char delim[] = ","; @@ -1278,26 +1257,26 @@ int parseArguments(int argc, char* argv[], Arguments& args) { break; } case 'd': - args.num_databases = parseIntegerArgument(optarg); + args.num_databases = atoi(optarg); break; case 'p': - args.num_processes = parseIntegerArgument(optarg); + args.num_processes = atoi(optarg); break; case 't': - args.num_threads = parseIntegerArgument(optarg); + args.num_threads = atoi(optarg); break; case 'r': - args.rows = parseIntegerArgument(optarg); + args.rows = atoi(optarg); args.row_digits = digits(args.rows); break; case 'l': - args.load_factor = parseDoubleArgument(optarg); + args.load_factor = atof(optarg); break; case 's': - args.seconds = parseIntegerArgument(optarg); + args.seconds = atoi(optarg); break; case 'i': - args.iteration = parseIntegerArgument(optarg); + args.iteration = atoi(optarg); break; case 'x': rc = parseTransaction(args, optarg); @@ -1305,7 +1284,7 @@ int parseArguments(int argc, char* argv[], Arguments& args) { return -1; break; case 'v': - args.verbose = parseIntegerArgument(optarg); + args.verbose = atoi(optarg); break; case 'z': args.zipf = 1; @@ -1332,23 +1311,23 @@ int parseArguments(int argc, char* argv[], Arguments& args) { } break; case ARG_ASYNC: - args.async_xacts = parseIntegerArgument(optarg); + args.async_xacts = atoi(optarg); break; case ARG_KEYLEN: - args.key_length = parseIntegerArgument(optarg); + args.key_length = atoi(optarg); break; case ARG_VALLEN: - args.value_length = parseIntegerArgument(optarg); + args.value_length = atoi(optarg); break; case ARG_TPS: case ARG_TPSMAX: - args.tpsmax = parseIntegerArgument(optarg); + args.tpsmax = atoi(optarg); break; case ARG_TPSMIN: - args.tpsmin = parseIntegerArgument(optarg); + args.tpsmin = atoi(optarg); break; case ARG_TPSINTERVAL: - args.tpsinterval = parseIntegerArgument(optarg); + args.tpsinterval = atoi(optarg); break; case ARG_TPSCHANGE: if (strcmp(optarg, "sin") == 0) @@ -1363,7 +1342,7 @@ int parseArguments(int argc, char* argv[], Arguments& args) { } break; case ARG_SAMPLING: - args.sampling = parseIntegerArgument(optarg); + args.sampling = atoi(optarg); break; case ARG_VERSION: logr.error("Version: {}", FDB_API_VERSION); @@ -1420,11 +1399,11 @@ int parseArguments(int argc, char* argv[], Arguments& args) { } break; case ARG_TXNTRACE: - args.txntrace = parseIntegerArgument(optarg); + args.txntrace = atoi(optarg); break; case ARG_TXNTAGGING: - args.txntagging = parseIntegerArgument(optarg); + args.txntagging = atoi(optarg); if (args.txntagging > 1000) { args.txntagging = 1000; } @@ -1437,7 +1416,7 @@ int parseArguments(int argc, char* argv[], Arguments& args) { memcpy(args.txntagging_prefix, optarg, strlen(optarg)); break; case ARG_CLIENT_THREADS_PER_VERSION: - args.client_threads_per_version = parseIntegerArgument(optarg); + args.client_threads_per_version = atoi(optarg); break; case ARG_DISABLE_CLIENT_BYPASS: args.disable_client_bypass = true; @@ -1498,16 +1477,16 @@ int parseArguments(int argc, char* argv[], Arguments& args) { args.private_key_pem = oss.str(); } break; case ARG_TRANSACTION_TIMEOUT_TX: - args.transaction_timeout_tx = parseIntegerArgument(optarg); + args.transaction_timeout_tx = atoi(optarg); break; case ARG_TRANSACTION_TIMEOUT_DB: - args.transaction_timeout_db = parseIntegerArgument(optarg); + args.transaction_timeout_db = atoi(optarg); break; case ARG_MAX_GRV_QUEUE_DELAY: - args.max_grv_queue_delay_ms = parseIntegerArgument(optarg); + args.max_grv_queue_delay_ms = atoi(optarg); break; case ARG_WARMUP_SECONDS: - args.warmup_seconds = parseIntegerArgument(optarg); + args.warmup_seconds = atoi(optarg); break; } } diff --git a/bindings/flow/tester/Tester.cpp b/bindings/flow/tester/Tester.cpp index 2c86f783d8d..f9678cf2ca8 100644 --- a/bindings/flow/tester/Tester.cpp +++ b/bindings/flow/tester/Tester.cpp @@ -28,7 +28,6 @@ #include "bindings/flow/FDBLoanerTypes.h" #include "fdbrpc/fdbrpc.h" #include "flow/DeterministicRandom.h" -#include "flow/ParseNumber.h" #include "flow/TLSConfig.h" // Otherwise we have to type setupNetwork(), FDB::open(), etc. @@ -1857,18 +1856,15 @@ int main(int argc, char** argv) { flushAndExit(FDB_EXIT_SUCCESS);*/ } StringRef prefix((const uint8_t*)argv[1], strlen(argv[1])); - auto apiVersion = parseNumberPrefix(StringRef(static_cast(argv[2]))); - if (!apiVersion.present()) { - fprintf(stderr, "Invalid API version: %s\n", argv[2]); - return 1; - } + int apiVersion; + sscanf(argv[2], "%d", &apiVersion); std::string clusterFilename; if (argc > 3) { clusterFilename = std::string(argv[3]); } // start test - startTest(Uncancellable(), clusterFilename, prefix, apiVersion.get()); + startTest(Uncancellable(), clusterFilename, prefix, apiVersion); // Run the network until someone tells us to stop g_network->run(); diff --git a/documentation/sphinx/source/clang-tidy.rst b/documentation/sphinx/source/clang-tidy.rst index 5dc80dd33f1..de952516f74 100644 --- a/documentation/sphinx/source/clang-tidy.rst +++ b/documentation/sphinx/source/clang-tidy.rst @@ -10,12 +10,11 @@ This guide explains how to run ``clang-tidy`` locally so you can fix issues befo What clang-tidy checks ====================== -FoundationDB configures 59 named checks in the ``.clang-tidy`` file at the repository root. The +FoundationDB configures 58 named checks in the ``.clang-tidy`` file at the repository root. The active set depends on the clang-tidy version and can be inspected with ``clang-tidy --list-checks``. The intent is to enable more as we go forward. Here are some example rules: * **38 Bugprone rules** -- catch potential runtime errors, including unsafe self-assignment, forwarding constructors that hide copy or move constructors, narrow accumulation initializers, mismatched argument comments, obvious infinite loops, chained comparisons, swapped arguments, integer division in floating-point calculations, missed base-class copy construction, repeated macro argument evaluation, near-miss virtual overrides, dangling returned references, incorrect erase/remove calls, ignored return values, incorrect POSIX error checks, and pointer-dependent iteration order -* **1 CERT rule** -- identify numeric conversion APIs that cannot report invalid input (``cert-err34-c``) * **2 C++ Core Guidelines rules** -- catch unsafe captures in coroutine lambdas and borrowed coroutine parameters (``cppcoreguidelines-avoid-capturing-lambda-coroutines`` and ``cppcoreguidelines-avoid-reference-coroutine-parameters``) * **2 Misc rules** -- catch redundant expressions and RAII objects held across coroutine suspension points * **4 Modernize rules** -- encourage modern C++ practices (e.g., ``modernize-use-auto``, ``modernize-use-override``) @@ -48,11 +47,6 @@ LLVM 20 or newer. CI installs the packaged ``clang-tidy`` separately from the build image's compiler so that the warning is active without changing the compiler or C++ standard library. -``cert-err34-c`` is the compatible name for the check called -``bugprone-unchecked-string-to-number-conversion`` in newer LLVM versions. -Replacing ``atoi`` with ``strtol`` alone is insufficient: validate the conversion, -range, and the caller's policy for trailing input. - Basic examples of ``clang-tidy`` style and performance improvement changes: .. code-block:: cpp diff --git a/fdbbackup/FileConverter.cpp b/fdbbackup/FileConverter.cpp index c5d9db4c1fe..8a710a7cff6 100644 --- a/fdbbackup/FileConverter.cpp +++ b/fdbbackup/FileConverter.cpp @@ -30,7 +30,6 @@ #include "fdbclient/BackupContainer.h" #include "fdbclient/MutationList.h" #include "flow/flow.h" -#include "flow/ParseNumber.h" #include "flow/serialize.h" #include "fdbclient/BuildFlags.h" @@ -513,27 +512,21 @@ int parseCommandLine(ConvertParams* param, CSimpleOpt* args) { printConvertUsage(); return FDB_EXIT_ERROR; - case OPT_BEGIN_VERSION: { - auto version = parseNumberPrefix(StringRef(arg)); - if (!version.present()) { + case OPT_BEGIN_VERSION: + if (!sscanf(arg, "%" SCNd64, ¶m->begin)) { std::cerr << "ERROR: could not parse begin version " << arg << "\n"; printConvertUsage(); return FDB_EXIT_ERROR; } - param->begin = version.get(); break; - } - case OPT_END_VERSION: { - auto version = parseNumberPrefix(StringRef(arg)); - if (!version.present()) { + case OPT_END_VERSION: + if (!sscanf(arg, "%" SCNd64, ¶m->end)) { std::cerr << "ERROR: could not parse end version " << arg << "\n"; printConvertUsage(); return FDB_EXIT_ERROR; } - param->end = version.get(); break; - } case OPT_CONTAINER: param->container_url = args->OptionArg(); diff --git a/fdbbackup/FileDecoder.cpp b/fdbbackup/FileDecoder.cpp index ef0cf034174..0fa7f94cf20 100644 --- a/fdbbackup/FileDecoder.cpp +++ b/fdbbackup/FileDecoder.cpp @@ -52,7 +52,6 @@ #include "flow/Platform.h" #include "flow/Trace.h" #include "flow/flow.h" -#include "flow/ParseNumber.h" #include "flow/serialize.h" #define SevDecodeInfo SevVerbose @@ -347,13 +346,11 @@ int parseDecodeCommandLine(Reference param, CSimpleOpt* args) { break; case OPT_BEGIN_VERSION_FILTER: - param->beginVersionFilter = - parseNumberPrefix(StringRef(static_cast(args->OptionArg()))).orDefault(0); + param->beginVersionFilter = std::atoll(args->OptionArg()); break; case OPT_END_VERSION_FILTER: - param->endVersionFilter = - parseNumberPrefix(StringRef(static_cast(args->OptionArg()))).orDefault(0); + param->endVersionFilter = std::atoll(args->OptionArg()); break; case OPT_CRASHONERROR: diff --git a/fdbbackup/backup.cpp b/fdbbackup/backup.cpp index 0e6e05c57cf..3e08d531fc9 100644 --- a/fdbbackup/backup.cpp +++ b/fdbbackup/backup.cpp @@ -55,7 +55,6 @@ #include "fdbclient/ManagementAPI.h" #include "flow/Platform.h" -#include "flow/ParseNumber.h" #include #include @@ -3038,24 +3037,20 @@ Version parseVersion(const char* str) { StringRef s((const uint8_t*)str, strlen(str)); if (s.endsWith("days"_sr) || s.endsWith("d"_sr)) { - auto days = parseNumberPrefix(s); - if (days.present()) { - double version = (double)CLIENT_KNOBS->CORE_VERSIONSPERSECOND * 24 * 3600 * -days.get(); - if (version >= (double)std::numeric_limits::min() && - version < -(double)std::numeric_limits::min()) { - return static_cast(version); - } - } - } else { - auto version = parseNumberPrefix(s); - if (version.present()) { - return version.get(); + float days; + if (sscanf(str, "%f", &days) != 1) { + fprintf(stderr, "Could not parse version: %s\n", str); + flushAndExit(FDB_EXIT_ERROR); } + return (double)CLIENT_KNOBS->CORE_VERSIONSPERSECOND * 24 * 3600 * -days; } - fprintf(stderr, "Could not parse version: %s\n", str); - flushAndExit(FDB_EXIT_ERROR); - return invalidVersion; + Version ver; + if (sscanf(str, "%" SCNd64, &ver) != 1) { + fprintf(stderr, "Could not parse version: %s\n", str); + flushAndExit(FDB_EXIT_ERROR); + } + return ver; } // Creates a connection to a cluster. Optionally prints an error if the connection fails. @@ -3596,15 +3591,13 @@ int main(int argc, char* argv[]) { case OPT_EXPIRE_MIN_RESTORABLE_DAYS: case OPT_EXPIRE_DELETE_BEFORE_DAYS: { const char* a = args->OptionArg(); - auto parsedVersion = parseNumberPrefix(StringRef(a)); - if (!parsedVersion.present()) { + long long ver = 0; + if (!sscanf(a, "%lld", &ver)) { fprintf(stderr, "ERROR: Could not parse expiration version `%s'\n", a); printHelpTeaser(newArgV[0]); return FDB_EXIT_ERROR; } - Version ver = parsedVersion.get(); - // Interpret the value as days worth of versions relative to now (negative) if (optId == OPT_EXPIRE_MIN_RESTORABLE_DAYS || optId == OPT_EXPIRE_DELETE_BEFORE_DAYS) { ver = -ver * 24 * 60 * 60 * CLIENT_KNOBS->CORE_VERSIONSPERSECOND; @@ -3694,13 +3687,12 @@ int main(int argc, char* argv[]) { case OPT_INITIAL_SNAPSHOT_INTERVAL: case OPT_MOD_ACTIVE_INTERVAL: { const char* a = args->OptionArg(); - auto parsedSeconds = parseNumberPrefix(StringRef(a)); - if (!parsedSeconds.present()) { + int seconds; + if (!sscanf(a, "%d", &seconds)) { fprintf(stderr, "ERROR: Could not parse snapshot interval `%s'\n", a); printHelpTeaser(newArgV[0]); return FDB_EXIT_ERROR; } - int seconds = parsedSeconds.get(); if (optId == OPT_SNAPSHOTINTERVAL) { snapshotIntervalSeconds = seconds; modifyOptions.snapshotIntervalSeconds = seconds; @@ -3788,46 +3780,44 @@ int main(int argc, char* argv[]) { } case OPT_ERRORLIMIT: { const char* a = args->OptionArg(); - auto parsedMaxErrors = parseNumberPrefix(StringRef(a)); - if (!parsedMaxErrors.present()) { + if (!sscanf(a, "%d", &maxErrors)) { fprintf(stderr, "ERROR: Could not parse max number of errors `%s'\n", a); printHelpTeaser(newArgV[0]); return FDB_EXIT_ERROR; } - maxErrors = parsedMaxErrors.get(); break; } case OPT_RESTORE_BEGIN_VERSION: { const char* a = args->OptionArg(); - auto parsedVersion = parseNumberPrefix(StringRef(a)); - if (!parsedVersion.present()) { + long long ver = 0; + if (!sscanf(a, "%lld", &ver)) { fprintf(stderr, "ERROR: Could not parse database beginVersion `%s'\n", a); printHelpTeaser(newArgV[0]); return FDB_EXIT_ERROR; } - beginVersion = parsedVersion.get(); + beginVersion = ver; break; } case OPT_RESTORE_VERSION: { const char* a = args->OptionArg(); - auto parsedVersion = parseNumberPrefix(StringRef(a)); - if (!parsedVersion.present()) { + long long ver = 0; + if (!sscanf(a, "%lld", &ver)) { fprintf(stderr, "ERROR: Could not parse database version `%s'\n", a); printHelpTeaser(newArgV[0]); return FDB_EXIT_ERROR; } - restoreVersion = parsedVersion.get(); + restoreVersion = ver; break; } case OPT_RESTORE_SNAPSHOT_VERSION: { const char* a = args->OptionArg(); - auto parsedVersion = parseNumberPrefix(StringRef(a)); - if (!parsedVersion.present()) { + long long ver = 0; + if (!sscanf(a, "%lld", &ver)) { fprintf(stderr, "ERROR: Could not parse database version `%s'\n", a); printHelpTeaser(newArgV[0]); return FDB_EXIT_ERROR; } - snapshotVersion = parsedVersion.get(); + snapshotVersion = ver; break; } case OPT_RESTORE_USER_DATA: { @@ -3844,8 +3834,8 @@ int main(int argc, char* argv[]) { } #ifdef _WIN32 case OPT_PARENTPID: { - const char* pid_str = args->OptionArg(); - int parent_pid = parseNumberPrefix(StringRef(pid_str)).orDefault(0); + auto pid_str = args->OptionArg(); + int parent_pid = atoi(pid_str); auto pHandle = OpenProcess(SYNCHRONIZE, FALSE, parent_pid); if (!pHandle) { TraceEvent("ParentProcessOpenError").GetLastError(); diff --git a/fdbcli/AdvanceVersionCommand.cpp b/fdbcli/AdvanceVersionCommand.cpp index 1bc704bebf6..c6647efd4dc 100644 --- a/fdbcli/AdvanceVersionCommand.cpp +++ b/fdbcli/AdvanceVersionCommand.cpp @@ -18,7 +18,6 @@ * limitations under the License. */ -#include "flow/ParseNumber.h" #include "boost/lexical_cast.hpp" #include "fmt/format.h" #include "fdbcli/fdbcli.h" @@ -38,12 +37,12 @@ Future advanceVersionCommandActor(Reference db, std::vector(tokens[1]); - if (!parsed.present()) { + Version v; + int n = 0; + if (sscanf(tokens[1].toString().c_str(), "%" PRId64 "%n", &v, &n) != 1 || n != tokens[1].size()) { printUsage(tokens[0]); co_return false; } else { - Version v = parsed.get(); Reference tr = db->createTransaction(); while (true) { tr->setOption(FDBTransactionOptions::SPECIAL_KEY_SPACE_ENABLE_WRITES); diff --git a/fdbcli/MaintenanceCommand.cpp b/fdbcli/MaintenanceCommand.cpp index 6ec838ec45d..b3495ba6e5b 100644 --- a/fdbcli/MaintenanceCommand.cpp +++ b/fdbcli/MaintenanceCommand.cpp @@ -18,7 +18,6 @@ * limitations under the License. */ -#include "flow/ParseNumber.h" #include #include "boost/lexical_cast.hpp" @@ -148,12 +147,14 @@ Future maintenanceCommandActor(Reference db, std::vector(tokens[3]); - if (!seconds.present()) { + double seconds; + int n = 0; + auto secondsStr = tokens[3].toString(); + if (sscanf(secondsStr.c_str(), "%lf%n", &seconds, &n) != 1 || n != secondsStr.size()) { printUsage(tokens[0]); result = false; } else { - bool setResult = co_await setHealthyZone(db, tokens[2], seconds.get(), true); + bool setResult = co_await setHealthyZone(db, tokens[2], seconds, true); result = setResult; } } else { diff --git a/fdbcli/SuspendCommand.cpp b/fdbcli/SuspendCommand.cpp index 34ae46ae033..8bbfad5f1e3 100644 --- a/fdbcli/SuspendCommand.cpp +++ b/fdbcli/SuspendCommand.cpp @@ -18,10 +18,6 @@ * limitations under the License. */ -#include "flow/ParseNumber.h" -#include -#include - #include "boost/algorithm/string.hpp" #include "fdbcli/fdbcli.h" @@ -72,10 +68,11 @@ Future suspendCommandActor(Reference db, } if (result) { - auto seconds = parseNumber(tokens[1]); + double seconds{ 0 }; + int n = 0; int i{ 0 }; - if (!seconds.present() || !std::isfinite(seconds.get()) || - seconds.get() < std::numeric_limits::min() || seconds.get() > std::numeric_limits::max()) { + auto secondsStr = tokens[1].toString(); + if (sscanf(secondsStr.c_str(), "%lf%n", &seconds, &n) != 1 || n != secondsStr.size()) { printUsage(tokens[0]); result = false; } else { @@ -84,8 +81,8 @@ Future suspendCommandActor(Reference db, addressesVec.push_back(tokens[i].toString()); } addressesStr = boost::algorithm::join(addressesVec, ","); - int64_t suspendRequestSent = co_await safeThreadFutureToFuture( - db->rebootWorker(addressesStr, false, static_cast(seconds.get()))); + int64_t suspendRequestSent = + co_await safeThreadFutureToFuture(db->rebootWorker(addressesStr, false, static_cast(seconds))); if (!suspendRequestSent) { result = false; fprintf( diff --git a/fdbcli/VersionEpochCommand.cpp b/fdbcli/VersionEpochCommand.cpp index 691f2e4d314..6044b0d71e7 100644 --- a/fdbcli/VersionEpochCommand.cpp +++ b/fdbcli/VersionEpochCommand.cpp @@ -18,7 +18,6 @@ * limitations under the License. */ -#include "flow/ParseNumber.h" #include "boost/lexical_cast.hpp" #include "fdbcli/fdbcli.h" @@ -122,12 +121,11 @@ Future versionEpochCommandActor(Reference db, Database cx, std: (tokens.size() == 3 && tokencmp(tokens[1], "set"))) { int64_t v; if (tokens.size() == 3) { - auto parsed = parseNumber(tokens[2]); - if (!parsed.present()) { + int n = 0; + if (sscanf(tokens[2].toString().c_str(), "%" SCNd64 "%n", &v, &n) != 1 || n != tokens[2].size()) { printUsage(tokens[0]); co_return false; } - v = parsed.get(); } else { v = 0; // default version epoch } diff --git a/fdbcli/fdbcli.cpp b/fdbcli/fdbcli.cpp index 17ce0f9801c..b8d4b591e97 100644 --- a/fdbcli/fdbcli.cpp +++ b/fdbcli/fdbcli.cpp @@ -18,7 +18,6 @@ * limitations under the License. */ -#include "flow/ParseNumber.h" #include "boost/lexical_cast.hpp" #include "fmt/format.h" #include "fdbclient/ClusterConnectionFile.h" @@ -1312,12 +1311,13 @@ Future cli(CLIOptions opt, LineNoise* plinenoise, Reference(tokens[1]); - if (!v.present()) { + double v; + int n = 0; + if (sscanf(tokens[1].toString().c_str(), "%lf%n", &v, &n) != 1 || n != tokens[1].size()) { printUsage(tokens[0]); is_error = true; } else { - co_await delay(v.get()); + co_await delay(v); } } continue; diff --git a/fdbclient/BackupAgentBase.cpp b/fdbclient/BackupAgentBase.cpp index 06f06370e4e..ef3406d6450 100644 --- a/fdbclient/BackupAgentBase.cpp +++ b/fdbclient/BackupAgentBase.cpp @@ -30,7 +30,6 @@ #include "fdbclient/SystemData.h" #include "fdbrpc/simulator.h" #include "flow/ActorCollection.h" -#include "flow/ParseNumber.h" #include "flow/DeterministicRandom.h" #include "flow/network.h" @@ -67,19 +66,11 @@ int64_t BackupAgentBase::parseTime(std::string timestamp) { #endif // Read timezone offset in +/-HHMM format then convert to seconds - StringRef timezone(timestamp); - if (timezone.size() < 24) { + int tzHH; + int tzMM; + if (sscanf(timestamp.substr(19, 5).c_str(), "%3d%2d", &tzHH, &tzMM) != 2) { return -1; } - timezone = timezone.substr(19, 5); - int consumed = 0; - auto hours = parseNumberPrefix(timezone.substr(0, 3), 10, &consumed); - auto minutes = parseNumberPrefix(timezone.substr(consumed, 2)); - if (!hours.present() || !minutes.present()) { - return -1; - } - int tzHH = hours.get(); - int tzMM = minutes.get(); if (tzHH < 0) { tzMM = -tzMM; } @@ -149,28 +140,13 @@ bool copyParameter(Reference source, Reference dest, Key key) { } Version getVersionFromString(std::string const& value) { - auto version = parseNumber(StringRef(value)); - if (!version.present()) { + Version version = invalidVersion; + int n = 0; + if (sscanf(value.c_str(), "%lld%n", (long long*)&version, &n) != 1 || n != value.size()) { TraceEvent(SevWarnAlways, "GetVersionFromString").detail("InvalidVersion", value); throw restore_invalid_version(); } - return version.get(); -} - -TEST_CASE("/backup/versionparsing") { - ASSERT(getVersionFromString("9223372036854775807") == std::numeric_limits::max()); - ASSERT(getVersionFromString("-9223372036854775808") == std::numeric_limits::min()); - for (const auto& value : { "", " ", "9223372036854775808", "-9223372036854775809", "123suffix", "123 " }) { - bool rejected = false; - try { - getVersionFromString(value); - } catch (Error& e) { - ASSERT(e.code() == error_code_restore_invalid_version); - rejected = true; - } - ASSERT(rejected); - } - return Void(); + return version; } // Transaction log data is stored by the FoundationDB core in the diff --git a/fdbclient/BackupContainerFileSystem.cpp b/fdbclient/BackupContainerFileSystem.cpp index 4b4ec772c22..2b9030e8095 100644 --- a/fdbclient/BackupContainerFileSystem.cpp +++ b/fdbclient/BackupContainerFileSystem.cpp @@ -29,11 +29,9 @@ #include "fdbrpc/AsyncFileEncrypted.h" #include "flow/StreamCipher.h" #include "flow/UnitTest.h" -#include "flow/ParseNumber.h" #include #include -#include class BackupContainerFileSystemImpl { public: @@ -172,24 +170,16 @@ class BackupContainerFileSystemImpl { static bool pathToRangeFile(RangeFile& out, const std::string& path, int64_t size) { std::string name = fileNameOnly(path); - StringRef fields(name); - if (name.find('\0') != std::string::npos || fields.eat(","_sr) != "range"_sr) { - return false; - } - auto version = parseNumber(fields.eat(","_sr)); - bool foundSeparator = false; - auto uid = fields.eat(","_sr, &foundSeparator); - auto blockSize = parseNumber(fields); - if (!version.present() || uid.empty() || !foundSeparator || !blockSize.present()) { - return false; - } RangeFile f; f.fileName = path; f.fileSize = size; - f.version = version.get(); - f.blockSize = blockSize.get(); - out = f; - return true; + int len; + if (sscanf(name.c_str(), "range,%" SCNd64 ",%*[^,],%u%n", &f.version, &f.blockSize, &len) == 2 && + len == name.size()) { + out = f; + return true; + } + return false; } static Future writeKeyspaceSnapshotFile(Reference bc, @@ -1394,16 +1384,11 @@ class BackupContainerFileSystemImpl { // Extract the snapshot begin version from a path static Version extractSnapshotBeginVersion(const std::string& path) { - StringRef input(path); - const StringRef prefix = "kvranges/snapshot."_sr; - if (!input.startsWith(prefix)) { - return invalidVersion; - } - input = input.substr(prefix.size()); - while (!input.empty() && std::isspace(static_cast(input[0]))) { - input = input.substr(1); + Version snapshotBeginVersion; + if (sscanf(path.c_str(), "kvranges/snapshot.%018" SCNd64, &snapshotBeginVersion) == 1) { + return snapshotBeginVersion; } - return parseNumberPrefix(input.substr(0, std::min(input.size(), 18))).orDefault(invalidVersion); + return invalidVersion; } // The innermost folder covers 100,000 seconds (1e11 versions) which is 5,000 mutation log files at current @@ -1424,63 +1409,67 @@ class BackupContainerFileSystemImpl { static bool pathToLogFile(LogFile& out, const std::string& path, int64_t size) { std::string name = fileNameOnly(path); - StringRef fields(name); - if (name.find('\0') != std::string::npos || fields.eat(","_sr) != "log"_sr) { - return false; - } - auto beginVersion = parseNumber(fields.eat(","_sr)); - auto endVersion = parseNumber(fields.eat(","_sr)); - bool foundSeparator = false; - auto uid = fields.eat(","_sr, &foundSeparator); - if (!beginVersion.present() || !endVersion.present() || uid.empty() || !foundSeparator) { - return false; - } LogFile f; f.fileName = path; f.fileSize = size; - f.beginVersion = beginVersion.get(); - f.endVersion = endVersion.get(); - auto blockSize = parseNumber(fields); - if (!blockSize.present()) { - auto tagId = parseNumber(fields.eat("-of-"_sr, &foundSeparator)); - if (!tagId.present() || tagId.get() < 0 || !foundSeparator) { - return false; - } - auto totalTags = parseNumber(fields.eat(","_sr)); - blockSize = parseNumber(fields); - if (!totalTags.present() || !blockSize.present()) { - return false; - } - f.tagId = tagId.get(); - f.totalTags = totalTags.get(); + int len; + if (sscanf(name.c_str(), + "log,%" SCNd64 ",%" SCNd64 ",%*[^,],%u%n", + &f.beginVersion, + &f.endVersion, + &f.blockSize, + &len) == 3 && + len == name.size()) { + out = f; + return true; + } else if (sscanf(name.c_str(), + "log,%" SCNd64 ",%" SCNd64 ",%*[^,],%d-of-%d,%u%n", + &f.beginVersion, + &f.endVersion, + &f.tagId, + &f.totalTags, + &f.blockSize, + &len) == 5 && + len == name.size() && f.tagId >= 0) { + out = f; + return true; } - f.blockSize = blockSize.get(); - out = f; - return true; + return false; } static bool pathToKeyspaceSnapshotFile(KeyspaceSnapshotFile& out, const std::string& path) { std::string name = fileNameOnly(path); - StringRef fields(name); - if (name.find('\0') != std::string::npos || fields.eat(","_sr) != "snapshot"_sr) { - return false; - } - auto beginVersion = parseNumber(fields.eat(","_sr)); - auto endVersion = parseNumber(fields.eat(","_sr)); - bool hasType = false; - auto totalSize = parseNumber(fields.eat(","_sr, &hasType)); - if (!beginVersion.present() || !endVersion.present() || !totalSize.present() || - (hasType && (fields.empty() || fields.size() > 63 || fields.toString().find(',') != std::string::npos))) { - return false; - } KeyspaceSnapshotFile f; f.fileName = path; - f.beginVersion = beginVersion.get(); - f.endVersion = endVersion.get(); - f.totalSize = totalSize.get(); - f.snapshotType = fields.toString(); - out = f; - return true; + int len; + char typeBuf[64] = {}; + + // Try new format with type suffix: snapshot,beginVersion,endVersion,totalSize,type + if (sscanf(name.c_str(), + "snapshot,%" SCNd64 ",%" SCNd64 ",%" SCNd64 ",%63[^,]%n", + &f.beginVersion, + &f.endVersion, + &f.totalSize, + typeBuf, + &len) == 4 && + len == name.size()) { + f.snapshotType = typeBuf; + out = f; + return true; + } + + // Try original format: snapshot,beginVersion,endVersion,totalSize + if (sscanf(name.c_str(), + "snapshot,%" SCNd64 ",%" SCNd64 ",%" SCNd64 "%n", + &f.beginVersion, + &f.endVersion, + &f.totalSize, + &len) == 3 && + len == name.size()) { + out = f; + return true; + } + return false; } // fallback for using existing write api if the underlying blob store doesn't support efficient writeEntireFile @@ -1861,9 +1850,10 @@ static Future> readVersionProperty(Referenceread((uint8_t*)s.data(), size, 0); - auto version = parseNumber(StringRef(s)); - if (rs == size && version.present()) - co_return version.get(); + Version v; + int len; + if (rs == size && sscanf(s.c_str(), "%" SCNd64 "%n", &v, &len) == 1 && len == size) + co_return v; TraceEvent(SevWarn, "BackupContainerInvalidProperty").detail("URL", bc->getURL()).detail("Path", path); @@ -2228,59 +2218,6 @@ TEST_CASE("/backup/containers_list") { } } -TEST_CASE("/backup/filenameparsing") { - RangeFile range; - ASSERT(BackupContainerFileSystemImpl::pathToRangeFile(range, "ranges/range,123,uid,4096", 100)); - ASSERT(range.version == 123 && range.blockSize == 4096 && range.fileSize == 100); - for (const auto& name : { "ranges/range,9223372036854775808,uid,4096", - "ranges/range,123,uid,4294967296", - "ranges/range,,uid,4096", - "ranges/range,123,,4096", - "ranges/range,123,uid,4096suffix" }) { - ASSERT(!BackupContainerFileSystemImpl::pathToRangeFile(range, name, 0)); - } - - LogFile log; - ASSERT(BackupContainerFileSystemImpl::pathToLogFile(log, "logs/log,123,456,uid,4096", 100)); - ASSERT(log.beginVersion == 123 && log.endVersion == 456 && log.blockSize == 4096 && log.tagId == -1); - ASSERT(BackupContainerFileSystemImpl::pathToLogFile(log, "plogs/log,123,456,uid,2-of-4,4096", 100)); - ASSERT(log.tagId == 2 && log.totalTags == 4 && log.blockSize == 4096); - for (const auto& name : { "logs/log,123,9223372036854775808,uid,4096", - "plogs/log,123,456,uid,2147483648-of-4,4096", - "plogs/log,123,456,uid,2-of-2147483648,4096", - "plogs/log,123,456,uid,-1-of-4,4096", - "plogs/log,123,456,uid,2-of-4,4294967296", - "plogs/log,123,456,uid,2-of-4,4096suffix" }) { - ASSERT(!BackupContainerFileSystemImpl::pathToLogFile(log, name, 0)); - } - - KeyspaceSnapshotFile snapshot; - ASSERT(BackupContainerFileSystemImpl::pathToKeyspaceSnapshotFile(snapshot, "snapshots/snapshot,123,456,789")); - ASSERT(snapshot.beginVersion == 123 && snapshot.endVersion == 456 && snapshot.totalSize == 789 && - snapshot.snapshotType.empty()); - ASSERT(BackupContainerFileSystemImpl::pathToKeyspaceSnapshotFile(snapshot, "snapshots/snapshot,123,456,789,bulk")); - ASSERT(snapshot.snapshotType == "bulk"); - ASSERT(BackupContainerFileSystemImpl::pathToKeyspaceSnapshotFile( - snapshot, "snapshots/snapshot,123,456,789," + std::string(63, 'x'))); - ASSERT(!BackupContainerFileSystemImpl::pathToKeyspaceSnapshotFile( - snapshot, "snapshots/snapshot,123,456,789," + std::string(64, 'x'))); - for (const auto& name : { "snapshots/snapshot,123,456,9223372036854775808", - "snapshots/snapshot,123,456,", - "snapshots/snapshot,123,456,789,", - "snapshots/snapshot,123,456,789,bulk,extra" }) { - ASSERT(!BackupContainerFileSystemImpl::pathToKeyspaceSnapshotFile(snapshot, name)); - } - - ASSERT(BackupContainerFileSystemImpl::extractSnapshotBeginVersion("kvranges/snapshot.000000000000000123/range") == - 123); - ASSERT(BackupContainerFileSystemImpl::extractSnapshotBeginVersion( - "kvranges/snapshot. \t000000000000000123/range") == 123); - ASSERT(BackupContainerFileSystemImpl::extractSnapshotBeginVersion("kvranges/snapshot. \t") == invalidVersion); - ASSERT(BackupContainerFileSystemImpl::extractSnapshotBeginVersion("kvranges/snapshot.invalid/range") == - invalidVersion); - return Void(); -} - TEST_CASE("/backup/time") { // test formatTime() for (int i = 0; i < 1000; ++i) { diff --git a/fdbclient/BackupContainerLocalDirectory.cpp b/fdbclient/BackupContainerLocalDirectory.cpp index 40088c54de4..78f91eb9acf 100644 --- a/fdbclient/BackupContainerLocalDirectory.cpp +++ b/fdbclient/BackupContainerLocalDirectory.cpp @@ -24,7 +24,6 @@ #include "flow/IAsyncFile.h" #include "flow/FaultInjection.h" #include "flow/Platform.h" -#include "flow/ParseNumber.h" #include "fdbrpc/simulator.h" #include "fdbrpc/SimulatorProcessInfo.h" @@ -272,7 +271,7 @@ Future> BackupContainerLocalDirectory::readFile(const std: // Extract block size from the filename, if present size_t lastComma = path.find_last_of(','); if (lastComma != path.npos) { - blockSize = parseNumberPrefix(StringRef(path).substr(lastComma + 1)).orDefault(0); + blockSize = atoi(path.substr(lastComma + 1).c_str()); } if (blockSize <= 0) { blockSize = deterministicRandom()->randomInt(1e4, 1e6); diff --git a/fdbclient/DatabaseConfiguration.cpp b/fdbclient/DatabaseConfiguration.cpp index 95826a76e34..f25072085d8 100644 --- a/fdbclient/DatabaseConfiguration.cpp +++ b/fdbclient/DatabaseConfiguration.cpp @@ -20,7 +20,6 @@ #include #include "fdbclient/DatabaseConfiguration.h" -#include "flow/ParseNumber.h" #include "fdbclient/FDBTypes.h" #include "fdbclient/SystemData.h" #include "flow/ITrace.h" @@ -60,19 +59,22 @@ void DatabaseConfiguration::resetInternal() { } int toInt(ValueRef const& v) { - return parseNumberPrefix(v).orDefault(0); + return atoi(v.toString().c_str()); } void parse(int* i, ValueRef const& v) { - *i = parseNumberPrefix(v).orDefault(0); + // FIXME: Sanity checking + *i = atoi(v.toString().c_str()); } void parse(int64_t* i, ValueRef const& v) { - *i = parseNumberPrefix(v).orDefault(0); + // FIXME: Sanity checking + *i = atoll(v.toString().c_str()); } void parse(double* i, ValueRef const& v) { - *i = parseNumberPrefix(v).orDefault(0); + // FIXME: Sanity checking + *i = atof(v.toString().c_str()); } void parseReplicationPolicy(Reference* policy, ValueRef const& v) { @@ -886,7 +888,7 @@ bool DatabaseConfiguration::isOverridden(std::string key) const { key = configKeysPrefix.toString() + std::move(key); if (mutableConfiguration.present()) { - return mutableConfiguration.get().contains(key); + return mutableConfiguration.get().find(key) != mutableConfiguration.get().end(); } const int keyLen = key.size(); diff --git a/fdbclient/ManagementAPI.cpp b/fdbclient/ManagementAPI.cpp index 1b58507ee4a..c83b44233ed 100644 --- a/fdbclient/ManagementAPI.cpp +++ b/fdbclient/ManagementAPI.cpp @@ -48,7 +48,6 @@ #include "fdbrpc/simulator.h" #include "fdbclient/StatusClient.h" #include "flow/Trace.h" -#include "flow/ParseNumber.h" #include "flow/UnitTest.h" #include "fdbrpc/ReplicationPolicy.h" #include "fdbrpc/Replication.h" @@ -104,12 +103,11 @@ std::map configForToken(std::string const& mode) { std::string key = mode.substr(0, pos); std::string value = mode.substr(pos + 1); - auto specifiedProxiesCount = key == "proxies" ? parseNumber(StringRef(value)) : Optional(); - if (key == "proxies" && specifiedProxiesCount.present()) { + if (key == "proxies" && isInteger(value)) { printf("Warning: Proxy role is being split into GRV Proxy and Commit Proxy, now prefer configuring " "'grv_proxies' and 'commit_proxies' separately. Generally we should follow that 'commit_proxies'" " is three times of 'grv_proxies' count and 'grv_proxies' should be not more than 4.\n"); - int proxiesCount = specifiedProxiesCount.get(); + int proxiesCount = atoi(value.c_str()); if (proxiesCount == -1) { proxiesCount = CLIENT_KNOBS->DEFAULT_AUTO_GRV_PROXIES + CLIENT_KNOBS->DEFAULT_AUTO_COMMIT_PROXIES; ASSERT_WE_THINK(proxiesCount >= 2); @@ -134,7 +132,7 @@ std::map configForToken(std::string const& mode) { commitProxyCount); TraceEvent("DatabaseConfigurationProxiesSpecified") - .detail("SpecifiedProxies", specifiedProxiesCount.get()) + .detail("SpecifiedProxies", atoi(value.c_str())) .detail("EffectiveSpecifiedProxies", proxiesCount) .detail("ConvertedGrvProxies", grvProxyCount) .detail("ConvertedCommitProxies", commitProxyCount); @@ -1203,8 +1201,8 @@ struct AutoQuorumChange final : IQuorumChange { Future> fStorageReplicas = tr->get("storage_replicas"_sr.withPrefix(configKeysPrefix)); Future> fLogReplicas = tr->get("log_replicas"_sr.withPrefix(configKeysPrefix)); co_await (success(fStorageReplicas) && success(fLogReplicas)); - int redundancy = std::min(parseNumberPrefix(fStorageReplicas.get().get()).orDefault(0), - parseNumberPrefix(fLogReplicas.get().get()).orDefault(0)); + int redundancy = std::min(atoi(fStorageReplicas.get().get().toString().c_str()), + atoi(fLogReplicas.get().get().toString().c_str())); co_return redundancy; } @@ -1468,7 +1466,7 @@ Future excludeServers(Transaction* tr, std::vector serve std::set exclusions(excl.begin(), excl.end()); bool containNewExclusion = false; for (auto& s : servers) { - if (exclusions.contains(s)) { + if (exclusions.find(s) != exclusions.end()) { continue; } containNewExclusion = true; @@ -1542,7 +1540,7 @@ Future excludeLocalities(Transaction* tr, std::unordered_set std::set exclusion(excl.begin(), excl.end()); bool containNewExclusion = false; for (const auto& l : localities) { - if (exclusion.contains(l)) { + if (exclusion.find(l) != exclusion.end()) { continue; } containNewExclusion = true; diff --git a/fdbclient/include/fdbclient/FDBTypes.h b/fdbclient/include/fdbclient/FDBTypes.h index cb913dadfb6..5ad9fc5c3a9 100644 --- a/fdbclient/include/fdbclient/FDBTypes.h +++ b/fdbclient/include/fdbclient/FDBTypes.h @@ -35,7 +35,6 @@ #include "flow/FastRef.h" #include "flow/ProtocolVersion.h" -#include "flow/ParseNumber.h" #include "flow/flow.h" #include "fdbclient/ProcessClass.h" #include "fdbclient/ProcessData.h" @@ -1515,7 +1514,7 @@ struct EncryptionAtRestModeDeprecated { } // A failed parsing returns 0 (DISABLED) - int num = parseNumberPrefix(val.get()).orDefault(0); + int num = atoi(val.get().toString().c_str()); if (num < 0 || num >= END) { return DISABLED; } diff --git a/fdbclient/include/fdbclient/RandomKeyValueUtils.h b/fdbclient/include/fdbclient/RandomKeyValueUtils.h index 89b6e9ec6d8..e0ad845db62 100644 --- a/fdbclient/include/fdbclient/RandomKeyValueUtils.h +++ b/fdbclient/include/fdbclient/RandomKeyValueUtils.h @@ -27,7 +27,6 @@ #include "flow/Arena.h" #include "flow/Error.h" #include "flow/IRandom.h" -#include "flow/ParseNumber.h" #include "fdbclient/FDBTypes.h" template @@ -65,7 +64,7 @@ struct RandomIntGenerator : IGenerator { alpha = true; return (unsigned int)s[0]; } else { - return parseNumberPrefix(s).orDefault(0); + return atol(s.toString().c_str()); } } @@ -221,11 +220,10 @@ struct RandomStringSetGeneratorBase : IKeyGenerator { maxKeyLen = keyGen.getMaxKeyLen(); ASSERT(indexGenerator.max > 0); std::set uniqueKeys; - uint64_t inserts = 0; + int inserts = 0; // for smaller indexGenerator.max, give it more insert try, as it may not find enough unique keys with 3 * max. // It adds roughly log * 100. For example, even for max is 1, it will try at least 100 times. - const uint64_t maxInsertTry = - uint64_t{ 3 } * indexGenerator.max + (((sizeof(uint) * 8) - clz(indexGenerator.max)) * 100); + const uint maxInsertTry = 3 * indexGenerator.max + (((sizeof(uint) * 8) - clz(indexGenerator.max)) * 100); while (uniqueKeys.size() < indexGenerator.max) { auto k = keyGen.next(); uniqueKeys.insert(k); diff --git a/fdbmonitor/fdbmonitor.h b/fdbmonitor/fdbmonitor.h index 956fa2eeedc..0081a6ec99d 100644 --- a/fdbmonitor/fdbmonitor.h +++ b/fdbmonitor/fdbmonitor.h @@ -29,7 +29,6 @@ #include #include #include -#include #include #include #include @@ -89,8 +88,6 @@ void print_usage(const char* name); std::unordered_map> set_watches(std::string path, int ifd); void load_conf(const char* confpath, uid_t& uid, gid_t& gid, sigset_t* mask, fdb_fd_set rfds, int* maxfd); uint64_t getRss(ProcessID id); -// Reads the first two /proc/statm fields and converts resident pages to bytes. Leaves rss unchanged on failure. -bool parseRss(std::string_view statm, long pageSize, uint64_t& rss); void kill_process(ProcessID id, bool wait = true, bool cleanup = true); struct Command; diff --git a/fdbmonitor/fdbmonitor_lib.cpp b/fdbmonitor/fdbmonitor_lib.cpp index ac1ea346df6..1c6b3a288ea 100644 --- a/fdbmonitor/fdbmonitor_lib.cpp +++ b/fdbmonitor/fdbmonitor_lib.cpp @@ -19,8 +19,6 @@ */ #include -#include -#include #include #include #ifndef _WIN32 @@ -419,32 +417,6 @@ std::unordered_map> id_command; std::unordered_map pid_id; std::unordered_map id_pid; -bool parseRss(std::string_view statm, long pageSize, uint64_t& rss) { - auto parsePages = [&statm](uint64_t& pages) { - while (!statm.empty() && std::isspace(static_cast(statm.front()))) { - statm.remove_prefix(1); - } - if (statm.empty()) { - return false; - } - const char* end = statm.data() + statm.size(); - auto result = std::from_chars(statm.data(), statm.data() + statm.size(), pages); - if (result.ec != std::errc() || (result.ptr != end && !std::isspace(static_cast(*result.ptr)))) { - return false; - } - statm.remove_prefix(result.ptr - statm.data()); - return true; - }; - uint64_t sizePages; - uint64_t residentPages; - if (pageSize <= 0 || !parsePages(sizePages) || !parsePages(residentPages) || - residentPages > std::numeric_limits::max() / static_cast(pageSize)) { - return false; - } - rss = residentPages * static_cast(pageSize); - return true; -} - // Return resident memory in bytes for the given process, or 0 if error. uint64_t getRss(ProcessID id) { #ifndef __linux__ @@ -459,15 +431,14 @@ uint64_t getRss(ProcessID id) { log_msg(SevWarn, "Unable to open stat file for %s\n", id.c_str()); return 0; } - char stat_buf[256]; - bool read = fgets(stat_buf, sizeof(stat_buf), stat_file) != nullptr; - fclose(stat_file); - uint64_t rss; - if (!read || !parseRss(stat_buf, sysconf(_SC_PAGESIZE), rss)) { + long rss = 0; + int ret = fscanf(stat_file, "%*s%ld", &rss); + if (ret == 0) { log_msg(SevWarn, "Unable to parse rss size for %s\n", id.c_str()); return 0; } - return rss; + fclose(stat_file); + return static_cast(rss) * sysconf(_SC_PAGESIZE); #endif } diff --git a/fdbmonitor/fdbmonitor_tests.cpp b/fdbmonitor/fdbmonitor_tests.cpp index 61b606fb09a..01d22a52227 100644 --- a/fdbmonitor/fdbmonitor_tests.cpp +++ b/fdbmonitor/fdbmonitor_tests.cpp @@ -3,7 +3,6 @@ #include #include #include -#include namespace fdbmonitor { namespace tests { @@ -155,39 +154,6 @@ void testEnvVarUtils() { assert_msg(!EnvVarUtils::keyValueValid("=BAZ", "FOO=BAR =BAZ"), "Key must be non-empty"); } -void testRssParsing() { - uint64_t rss = 123; - assert_msg(parseRss("100 20 3 4 5 6 7\n", 4096, rss) && rss == 81920, "Resident pages must convert to bytes"); - assert_msg(parseRss(" \t100 0\n", 4096, rss) && rss == 0, "Zero resident pages must be valid"); - const char bounded[] = { '1', '0', '0', ' ', '2', '0' }; - assert_msg(parseRss(std::string_view(bounded, 5), 4096, rss) && rss == 8192, "Parsing must honor the input length"); - const uint64_t maxBytes = std::numeric_limits::max(); - assert_msg(parseRss("100 " + std::to_string(maxBytes), 1, rss) && rss == maxBytes, - "The largest byte count must be valid"); - const uint64_t maxPages = maxBytes / 4096; - assert_msg(parseRss("100 " + std::to_string(maxPages), 4096, rss) && rss == maxPages * 4096, - "The largest whole-page byte count must be valid"); - for (const char* invalid : { "", - " \t", - "100", - "100 ", - "x 20", - "100 x", - "100 2x", - "-1 20", - "100 -1", - "18446744073709551616 20", - "100 18446744073709551616" }) { - rss = 123; - assert_msg(!parseRss(invalid, 4096, rss) && rss == 123, "Invalid statm must leave RSS unchanged"); - } - rss = 123; - assert_msg(!parseRss("100 " + std::to_string(maxPages + 1), 4096, rss) && rss == 123, - "Resident byte-count overflow must be rejected"); - assert_msg(!parseRss("100 20", 0, rss) && rss == 123, "Zero page size must be rejected"); - assert_msg(!parseRss("100 20", -1, rss) && rss == 123, "Failed page-size lookup must be rejected"); -} - } // namespace tests } // namespace fdbmonitor @@ -196,5 +162,4 @@ int main(int argc, char** argv) { testPathOps(); testEnvVarUtils(); - testRssParsing(); } diff --git a/fdbrpc/HTTP.cpp b/fdbrpc/HTTP.cpp index fad5985eec4..7de0e52c23d 100644 --- a/fdbrpc/HTTP.cpp +++ b/fdbrpc/HTTP.cpp @@ -26,8 +26,6 @@ #include "flow/Trace.h" #include "flow/Knobs.h" #include "flow/CodeProbe.h" -#include "flow/ParseNumber.h" -#include "flow/UnitTest.h" #include "md5/md5.h" #include "libb64/encode.h" #include @@ -478,33 +476,6 @@ Future readHTTPData(HTTPData* r, } } -static Optional parseHTTPVersion(StringRef text, int* consumed = nullptr) { - if (!text.startsWith("HTTP/"_sr)) { - return {}; - } - int versionLength = 0; - auto version = parseNumberPrefix(text.substr(5), 10, &versionLength); - if (version.present() && consumed != nullptr) { - *consumed = 5 + versionLength; - } - return version; -} - -static bool parseHTTPResponseLine(StringRef text, float& version, int& code) { - int versionLength = 0; - auto parsedVersion = parseHTTPVersion(text, &versionLength); - if (!parsedVersion.present()) { - return false; - } - auto parsedCode = parseNumberPrefix(text.substr(versionLength)); - if (!parsedCode.present()) { - return false; - } - version = parsedVersion.get(); - code = parsedCode.get(); - return true; -} - // Reads an HTTP request from a network connection // If the connection fails while being read the exception will emitted // If the response is not parsable or complete in some way, http_bad_response will be thrown @@ -559,8 +530,9 @@ Future read_http_request(Reference r, Reference read_http_response(Reference r, Referenceversion, r->code)) { + int reachedEnd = -1; + if (sscanf(buf.c_str() + pos, "HTTP/%f %d%n", &r->version, &r->code, &reachedEnd) < 2 || reachedEnd < 0) { TraceEvent(SevWarn, "HTTPResponseParseFailure") .detail("Buffer", buf.substr(pos, std::min(lineLen, (size_t)100))) .detail("Pos", pos) @@ -615,23 +588,6 @@ Future read_http_response(Reference r, Referencedata, conn, &buf, &pos, header_only, skipCheckMD5); } -TEST_CASE("/fdbrpc/HTTP/NumericFields") { - ASSERT(!parseHTTPVersion("not-http"_sr).present()); - ASSERT(!parseHTTPVersion("HTTP/"_sr).present()); - ASSERT(!parseHTTPVersion("HTTP/1e1000"_sr).present()); - ASSERT(parseHTTPVersion("HTTP/1.1"_sr).get() == 1.1f); - - float version = 0; - int code = 0; - ASSERT(parseHTTPResponseLine("HTTP/1.1 200 OK"_sr, version, code)); - ASSERT(version == 1.1f && code == 200); - ASSERT(parseHTTPResponseLine("HTTP/1.0\t404 Not Found"_sr, version, code)); - ASSERT(version == 1.0f && code == 404); - ASSERT(!parseHTTPResponseLine("HTTP/1.1 "_sr, version, code)); - ASSERT(!parseHTTPResponseLine("HTTP/1.1 2147483648"_sr, version, code)); - return Void(); -} - Future HTTP::IncomingResponse::read(Reference conn, bool header_only) { return read_http_response(Reference::addRef(this), conn, header_only); } diff --git a/fdbrpc/ReplicationUtils.cpp b/fdbrpc/ReplicationUtils.cpp index 3d9ba022476..38566e3c5e0 100644 --- a/fdbrpc/ReplicationUtils.cpp +++ b/fdbrpc/ReplicationUtils.cpp @@ -22,7 +22,6 @@ #include "flow/Hash3.h" #include "flow/UnitTest.h" #include "flow/Platform.h" -#include "flow/ParseNumber.h" #include "fdbrpc/ReplicationPolicy.h" #include "fdbrpc/Replication.h" @@ -762,9 +761,6 @@ Reference randomAcrossPolicy(LocalitySet const& serverSet) { } int testReplication() { - auto parseEnvInt = [](const char* value, int defaultValue) { - return value ? parseNumberPrefix(StringRef(value)).orDefault(0) : defaultValue; - }; const char* testTotalEnv = getenv("REPLICATION_TESTTOTAL"); const char* debugLevelEnv = getenv("REPLICATION_DEBUGLEVEL"); const char* policyTotalEnv = getenv("REPLICATION_POLICYTOTAL"); @@ -777,16 +773,16 @@ int testReplication() { const char* rateSampleEnv = getenv("REPLICATION_RATESAMPLE"); const char* policySampleEnv = getenv("REPLICATION_POLICYSAMPLE"); const char* policyMinEnv = getenv("REPLICATION_POLICYEXTRA"); - int totalTests = parseEnvInt(testTotalEnv, 10000); - int skipTotal = parseEnvInt(skipTotalEnv, 0); - int findBest = parseEnvInt(findBestEnv, 0); - int policyIndexStatic = parseEnvInt(policyIndexEnv, -1); - int policyTotal = parseEnvInt(policyTotalEnv, 100); - bool stopOnError = parseEnvInt(stopOnErrorEnv, 0) > 0; - bool validate = parseEnvInt(validateEnv, 1) > 0; - int rateSample = parseEnvInt(rateSampleEnv, 1000); - int policySample = parseEnvInt(policySampleEnv, 100); - int policyMin = parseEnvInt(policyMinEnv, 2); + int totalTests = testTotalEnv ? atoi(testTotalEnv) : 10000; + int skipTotal = skipTotalEnv ? atoi(skipTotalEnv) : 0; + int findBest = findBestEnv ? atoi(findBestEnv) : 0; + int policyIndexStatic = policyIndexEnv ? atoi(policyIndexEnv) : -1; + int policyTotal = policyTotalEnv ? atoi(policyTotalEnv) : 100; + bool stopOnError = stopOnErrorEnv ? (atoi(stopOnErrorEnv) > 0) : false; + bool validate = validateEnv ? (atoi(validateEnv) > 0) : true; + int rateSample = rateSampleEnv ? atoi(rateSampleEnv) : 1000; + int policySample = policySampleEnv ? atoi(policySampleEnv) : 100; + int policyMin = policyMinEnv ? atoi(policyMinEnv) : 2; int policyIndex, testCounter, alsoSize, debugBackup, maxAlsoSize; std::vector serverIndexes; Reference testServers; @@ -795,7 +791,7 @@ int testReplication() { int totalErrors = 0; if (debugLevelEnv) - g_replicationdebug = parseEnvInt(debugLevelEnv, 0); + g_replicationdebug = atoi(debugLevelEnv); debugBackup = g_replicationdebug; testServers = createTestLocalityMap(serverIndexes, @@ -873,7 +869,7 @@ int testReplication() { } if (g_replicationdebug >= 0) printf("Succeeded in completing %d of %d policies\n", testCounter - totalErrors, totalTests); - if ((g_replicationdebug > 0) || parseEnvInt(reportCacheEnv, 0) > 0) { + if ((g_replicationdebug > 0) || ((reportCacheEnv) && (atoi(reportCacheEnv) > 0))) { testServers->cacheReport(); } @@ -885,7 +881,7 @@ void filterLocalityDataForPolicy(const std::set& keys, LocalityData for (auto iter = ld->_data.begin(); iter != ld->_data.end();) { auto prev = iter; iter++; - if (!keys.contains(prev->first.toString())) { + if (keys.find(prev->first.toString()) == keys.end()) { ld->_data.erase(prev); } } diff --git a/fdbserver/SimulatedCluster.cpp b/fdbserver/SimulatedCluster.cpp index 1ddefa72aa8..090b41777bb 100644 --- a/fdbserver/SimulatedCluster.cpp +++ b/fdbserver/SimulatedCluster.cpp @@ -28,7 +28,6 @@ #include -#include "flow/ParseNumber.h" #include "fdbclient/DatabaseConfiguration.h" #include "fdbclient/FDBTypes.h" #include "fdbrpc/Locality.h" @@ -383,19 +382,20 @@ class TestConfig : public BasicTestConfig { } if (attrib == "extraDatabaseCount") { - extraDatabaseCount = parseNumberPrefix(StringRef(value)).orDefault(extraDatabaseCount); + sscanf(value.c_str(), "%d", &extraDatabaseCount); } if (attrib == "minimumReplication") { - minimumReplication = parseNumberPrefix(StringRef(value)).orDefault(minimumReplication); + sscanf(value.c_str(), "%d", &minimumReplication); } if (attrib == "minimumRegions") { - minimumRegions = parseNumberPrefix(StringRef(value)).orDefault(minimumRegions); + sscanf(value.c_str(), "%d", &minimumRegions); } if (attrib == "configureLocked") { - int configureLockedInt = parseNumberPrefix(StringRef(value)).orDefault(0); + int configureLockedInt; + sscanf(value.c_str(), "%d", &configureLockedInt); configureLocked = (configureLockedInt != 0); } @@ -404,7 +404,7 @@ class TestConfig : public BasicTestConfig { } if (attrib == "logAntiQuorum") { - logAntiQuorum = parseNumberPrefix(StringRef(value)).orDefault(logAntiQuorum); + sscanf(value.c_str(), "%d", &logAntiQuorum); } if (attrib == "storageEngineExcludeTypes") { @@ -418,7 +418,7 @@ class TestConfig : public BasicTestConfig { } } if (attrib == "maxTLogVersion") { - maxTLogVersion = parseNumberPrefix(StringRef(value)).orDefault(maxTLogVersion); + sscanf(value.c_str(), "%d", &maxTLogVersion); } if (attrib == "disableTss") { disableTss = strcmp(value.c_str(), "true") == 0; @@ -442,12 +442,10 @@ class TestConfig : public BasicTestConfig { longRunningTest = strcmp(value.c_str(), "true") == 0; } if (attrib == "simulationNormalRunTestsTimeoutSeconds") { - simulationNormalRunTestsTimeoutSeconds = - parseNumberPrefix(StringRef(value)).orDefault(simulationNormalRunTestsTimeoutSeconds); + sscanf(value.c_str(), "%d", &simulationNormalRunTestsTimeoutSeconds); } if (attrib == "simulationBuggifyRunTestsTimeoutSeconds") { - simulationBuggifyRunTestsTimeoutSeconds = - parseNumberPrefix(StringRef(value)).orDefault(simulationBuggifyRunTestsTimeoutSeconds); + sscanf(value.c_str(), "%d", &simulationBuggifyRunTestsTimeoutSeconds); } } @@ -667,8 +665,8 @@ Future runDr(Reference connRecord) { .detail("ConnectionString", connRecord->getConnectionString().toString()) .detail("ExtraString", fdbSimulationPolicyState().extraDatabases[0]); - DatabaseBackupAgent dbAgent(cx); - DatabaseBackupAgent extraAgent(drDatabase); + DatabaseBackupAgent dbAgent = DatabaseBackupAgent(cx); + DatabaseBackupAgent extraAgent = DatabaseBackupAgent(drDatabase); auto drPollDelay = 1.0 / CLIENT_KNOBS->BACKUP_AGGREGATE_POLL_RATE; @@ -1362,27 +1360,19 @@ Future restartSimulatedSystem(std::vector>* systemActors, // allows multiple ipAddr entries ini.SetMultiKey(); - auto parseRestartInteger = [](const char* text) { - Optional value = text == nullptr ? Optional() : parseNumberPrefix(StringRef(text)); - if (!value.present()) { - throw test_specification_invalid(); - } - return value.get(); - }; - try { - int machineCount = parseRestartInteger(ini.GetValue("META", "machineCount")); - int processesPerMachine = parseRestartInteger(ini.GetValue("META", "processesPerMachine")); + int machineCount = atoi(ini.GetValue("META", "machineCount")); + int processesPerMachine = atoi(ini.GetValue("META", "processesPerMachine")); int listenersPerProcess = 1; auto listenersPerProcessStr = ini.GetValue("META", "listenersPerProcess"); if (listenersPerProcessStr != nullptr) { - listenersPerProcess = parseRestartInteger(listenersPerProcessStr); + listenersPerProcess = atoi(listenersPerProcessStr); } - int desiredCoordinators = parseRestartInteger(ini.GetValue("META", "desiredCoordinators")); - int testerCount = parseRestartInteger(ini.GetValue("META", "testerCount")); + int desiredCoordinators = atoi(ini.GetValue("META", "desiredCoordinators")); + int testerCount = atoi(ini.GetValue("META", "testerCount")); auto tssModeStr = ini.GetValue("META", "tssMode"); if (tssModeStr != nullptr) { - fdbSimulationPolicyState().tssMode = static_cast(parseRestartInteger(tssModeStr)); + fdbSimulationPolicyState().tssMode = static_cast(atoi(tssModeStr)); } ClusterConnectionString conn(ini.GetValue("META", "connectionString")); if (testConfig->extraDatabaseMode == FDBExtraDatabaseMode::Local) { @@ -1422,8 +1412,7 @@ Future restartSimulatedSystem(std::vector>* systemActors, zoneId = Standalone(zoneIdStr); } - auto cType = static_cast( - parseRestartInteger(ini.GetValue(machineIdString.c_str(), "mClass"))); + auto cType = static_cast(atoi(ini.GetValue(machineIdString.c_str(), "mClass"))); // using specialized class types can lead to nondeterministic recruitment if (cType == ProcessClass::MasterClass || cType == ProcessClass::ResolutionClass) { cType = ProcessClass::StatelessClass; @@ -1435,7 +1424,7 @@ Future restartSimulatedSystem(std::vector>* systemActors, } std::vector ipAddrs; - int processes = parseRestartInteger(ini.GetValue(machineIdString.c_str(), "processes")); + int processes = atoi(ini.GetValue(machineIdString.c_str(), "processes")); auto ip = ini.GetValue(machineIdString.c_str(), "ipAddr"); @@ -1742,7 +1731,8 @@ SimulationStorageEngine chooseSimulationStorageEngine(const TestConfig& testConf } } else if (SERVER_KNOBS->ENFORCE_SHARDED_ROCKSDB_SIM_IF_AVALIABLE && - !testConfig.storageEngineExcludeTypes.contains(SimulationStorageEngine::SHARDED_ROCKSDB)) { + testConfig.storageEngineExcludeTypes.find(SimulationStorageEngine::SHARDED_ROCKSDB) == + testConfig.storageEngineExcludeTypes.end()) { reason = "ENFORCE_SHARDED_ROCKSDB_SIM_IF_AVALIABLE is enabled"_sr; result = SimulationStorageEngine::SHARDED_ROCKSDB; diff --git a/fdbserver/clustercontroller/Status.cpp b/fdbserver/clustercontroller/Status.cpp index bdfdad9938c..cd91c939c5c 100644 --- a/fdbserver/clustercontroller/Status.cpp +++ b/fdbserver/clustercontroller/Status.cpp @@ -18,7 +18,6 @@ * limitations under the License. */ -#include "flow/ParseNumber.h" #include #include "fdbclient/json_spirit/json_spirit_value.h" #include "flow/genericactors.h" @@ -182,20 +181,7 @@ class StatusCounter { explicit(false) StatusCounter(const std::string& parsableText) { parseText(parsableText); } StatusCounter& parseText(const std::string& parsableText) { - StringRef remaining(parsableText); - int consumed = 0; - auto parsedHz = parseNumberPrefix(remaining, 10, &consumed); - remaining = remaining.substr(consumed); - consumed = 0; - auto parsedRoughness = parseNumberPrefix(remaining, 10, &consumed); - remaining = remaining.substr(consumed); - auto parsedCounter = parseNumberPrefix(remaining); - if (!parsedHz.present() || !parsedRoughness.present() || !parsedCounter.present()) { - throw attribute_not_found(); - } - hz = parsedHz.get(); - roughness = parsedRoughness.get(); - counter = parsedCounter.get(); + sscanf(parsableText.c_str(), "%lf %lf %" SCNd64 "", &hz, &roughness, &counter); return *this; } @@ -231,32 +217,11 @@ class StatusCounter { int64_t counter; }; -TEST_CASE("/status/counterParsing") { - StatusCounter counter("1.25 2.5 42"); - ASSERT_EQ(counter.getHz(), 1.25); - ASSERT_EQ(counter.getRoughness(), 2.5); - ASSERT_EQ(counter.getCounter(), 42); - for (const char* text : { "", "1.25 2.5", "1.25 x 42", "1.25 2.5 9223372036854775808" }) { - bool rejected = false; - try { - counter.parseText(text); - } catch (Error& e) { - ASSERT_EQ(e.code(), error_code_attribute_not_found); - rejected = true; - } - ASSERT(rejected); - ASSERT_EQ(counter.getHz(), 1.25); - ASSERT_EQ(counter.getRoughness(), 2.5); - ASSERT_EQ(counter.getCounter(), 42); - } - return Void(); -} - static JsonBuilderObject getError(const TraceEventFields& errorFields) { JsonBuilderObject statusObj; try { if (errorFields.size()) { - double time = errorFields.getDouble("Time"); + double time = atof(errorFields.getValue("Time").c_str()); statusObj["time"] = time; statusObj["raw_log_message"] = errorFields.toString(); @@ -1249,10 +1214,10 @@ static AsyncResult recoveryStateStatusFetcher(Database cx, // Add additional metadata for certain statuses if (mStatusCode == RecoveryStatus::recruiting_transaction_servers) { - int requiredLogs = md.getInt("RequiredTLogs"); - int requiredCommitProxies = md.getInt("RequiredCommitProxies"); - int requiredGrvProxies = md.getInt("RequiredGrvProxies"); - int requiredResolvers = md.getInt("RequiredResolvers"); + int requiredLogs = atoi(md.getValue("RequiredTLogs").c_str()); + int requiredCommitProxies = atoi(md.getValue("RequiredCommitProxies").c_str()); + int requiredGrvProxies = atoi(md.getValue("RequiredGrvProxies").c_str()); + int requiredResolvers = atoi(md.getValue("RequiredResolvers").c_str()); // int requiredProcesses = std::max(requiredLogs, std::max(requiredResolvers, requiredCommitProxies)); // int requiredMachines = std::max(requiredLogs, 1); diff --git a/fdbserver/commitproxy/CommitProxyServer.cpp b/fdbserver/commitproxy/CommitProxyServer.cpp index 1ae3e695eae..5b2ecc821eb 100644 --- a/fdbserver/commitproxy/CommitProxyServer.cpp +++ b/fdbserver/commitproxy/CommitProxyServer.cpp @@ -18,7 +18,6 @@ * limitations under the License. */ -#include "flow/ParseNumber.h" #include #include #include @@ -2496,7 +2495,7 @@ Future proxySnapCreate(ProxySnapRequest snapReq, ProxyCommitData* commitDa auto result = commitData->txnStateStore->readValue("log_anti_quorum"_sr.withPrefix(configKeysPrefix)).get(); int logAntiQuorum = 0; if (result.present()) { - logAntiQuorum = parseNumberPrefix(result.get()).orDefault(0); + logAntiQuorum = atoi(result.get().toString().c_str()); } // FIXME: logAntiQuorum not supported, remove it later, // In version2, we probably don't need this limitation, but this needs to be tested. diff --git a/fdbserver/core/QuietDatabase.cpp b/fdbserver/core/QuietDatabase.cpp index e41e88adb15..878ae842d40 100644 --- a/fdbserver/core/QuietDatabase.cpp +++ b/fdbserver/core/QuietDatabase.cpp @@ -18,8 +18,6 @@ * limitations under the License. */ -#include "flow/UnitTest.h" -#include "flow/ParseNumber.h" #include #include #include @@ -145,43 +143,14 @@ Future getDataInFlight(Database cx, Reference co // Computes the queue size for storage servers and tlogs using the bytesInput and bytesDurable attributes int64_t getQueueSize(const TraceEventFields& md) { - auto parseBytes = [](const std::string& text) { - StringRef remaining(text); - for (int i = 0; i < 2; ++i) { - int consumed = 0; - if (!parseNumberPrefix(remaining, 10, &consumed).present()) { - throw attribute_not_found(); - } - remaining = remaining.substr(consumed); - } - auto bytes = parseNumberPrefix(remaining); - if (!bytes.present()) { - throw attribute_not_found(); - } - return bytes.get(); - }; - return parseBytes(md.getValue("BytesInput")) - parseBytes(md.getValue("BytesDurable")); -} + double inputRate, durableRate; + double inputRoughness, durableRoughness; + int64_t inputBytes, durableBytes; -TEST_CASE("/fdbserver/QuietDatabase/queueCounterParsing") { - TraceEventFields fields; - fields.addField("BytesInput", "1.25 2.5 100"); - fields.addField("BytesDurable", "1.0 2.0 40"); - ASSERT_EQ(getQueueSize(fields), 60); - for (const char* text : { "", "1.0 2.0", "1.0 2.0 9223372036854775808" }) { - TraceEventFields invalid; - invalid.addField("BytesInput", text); - invalid.addField("BytesDurable", "1.0 2.0 40"); - bool rejected = false; - try { - getQueueSize(invalid); - } catch (Error& e) { - ASSERT_EQ(e.code(), error_code_attribute_not_found); - rejected = true; - } - ASSERT(rejected); - } - return Void(); + sscanf(md.getValue("BytesInput").c_str(), "%lf %lf %" SCNd64, &inputRate, &inputRoughness, &inputBytes); + sscanf(md.getValue("BytesDurable").c_str(), "%lf %lf %" SCNd64, &durableRate, &durableRoughness, &durableBytes); + + return inputBytes - durableBytes; } int64_t getDurableVersion(const TraceEventFields& md) { @@ -220,8 +189,8 @@ Future> getCoordWorkers(Database cx, Reference secondary = worker.interf.tLog.getEndpoint().addresses.secondaryAddress; - if (coordinatorsAddrSet.contains(primary) || - (secondary.present() && coordinatorsAddrSet.contains(secondary.get()))) { + if (coordinatorsAddrSet.find(primary) != coordinatorsAddrSet.end() || + (secondary.present() && (coordinatorsAddrSet.find(secondary.get()) != coordinatorsAddrSet.end()))) { result.push_back(worker.interf); } } @@ -316,7 +285,7 @@ Future, int>> getStorageWorkers(Database }); int usableRegions = 1; if (regionsValue.present()) { - usableRegions = parseNumberPrefix(regionsValue.get()).orDefault(0); + usableRegions = atoi(regionsValue.get().toString().c_str()); } auto masterDcId = dbInfo->get().master.locality.dcId(); diff --git a/fdbserver/core/WorkloadKeys.cpp b/fdbserver/core/WorkloadKeys.cpp index b6994b8d379..3a165908186 100644 --- a/fdbserver/core/WorkloadKeys.cpp +++ b/fdbserver/core/WorkloadKeys.cpp @@ -1,4 +1,3 @@ -#include "flow/ParseNumber.h" #include #include @@ -10,7 +9,8 @@ Key doubleToTestKey(double p) { } double testKeyToDouble(const KeyRef& p) { - uint64_t x = parseNumberPrefix(p, 16).orDefault(0); + uint64_t x = 0; + sscanf(p.toString().c_str(), "%" SCNx64, &x); return *(double*)&x; } diff --git a/fdbserver/datadistributor/DataDistribution.cpp b/fdbserver/datadistributor/DataDistribution.cpp index 5c7d97ab462..9365b10bbfa 100644 --- a/fdbserver/datadistributor/DataDistribution.cpp +++ b/fdbserver/datadistributor/DataDistribution.cpp @@ -18,7 +18,6 @@ * limitations under the License. */ -#include "flow/ParseNumber.h" #include #include #include @@ -2243,7 +2242,8 @@ Future scheduleBulkLoadJob(Reference self, Promise // No matter whether the task range is aligned with the manifest entry range, the task // begin key must be in the manifestEntryMap. See manifestEntryMap definition for more // details. - ASSERT(self->bulkLoadJobManager.get().manifestEntryMap->contains(task.getRange().begin)); + ASSERT(self->bulkLoadJobManager.get().manifestEntryMap->find(task.getRange().begin) != + self->bulkLoadJobManager.get().manifestEntryMap->end()); if (task.onAnyPhase( { BulkLoadPhase::Complete, BulkLoadPhase::Acknowledged, BulkLoadPhase::Error })) { ASSERT(task.getRange().end == res[i + 1].key); @@ -3455,7 +3455,7 @@ Future>> getSta Optional regionsValue = co_await tr.get("usable_regions"_sr.withPrefix(configKeysPrefix)); int usableRegions = 1; if (regionsValue.present()) { - usableRegions = parseNumberPrefix(regionsValue.get()).orDefault(0); + usableRegions = atoi(regionsValue.get().toString().c_str()); } auto masterDcId = dbInfo->get().master.locality.dcId(); int storageFailures = 0; @@ -3493,7 +3493,7 @@ Future>> getSta for (const auto& tlog : *tlogs) { TraceEvent(SevDebug, "GetStatefulWorkersTLog").detail("Addr", tlog.address()); - if (!workersMap.contains(tlog.address())) { + if (workersMap.find(tlog.address()) == workersMap.end()) { TraceEvent(SevWarn, "MissingTLogWorkerInterface").detail("TlogAddress", tlog.address()); throw snap_tlog_failed(); } @@ -3518,8 +3518,8 @@ Future>> getSta // as we use primary addresses from storage and tlog interfaces above NetworkAddress primary = worker.interf.address(); Optional secondary = worker.interf.tLog.getEndpoint().addresses.secondaryAddress; - if (coordinatorsAddrSet.contains(primary) || - (secondary.present() && coordinatorsAddrSet.contains(secondary.get()))) { + if (coordinatorsAddrSet.find(primary) != coordinatorsAddrSet.end() || + (secondary.present() && (coordinatorsAddrSet.find(secondary.get()) != coordinatorsAddrSet.end()))) { if (result.contains(primary)) { ASSERT(workersMap[primary].id() == result[primary].first.id()); result[primary].second.append(",coord"); diff --git a/fdbserver/fdbserver.cpp b/fdbserver/fdbserver.cpp index 7afbe781456..8e4f8f37de8 100644 --- a/fdbserver/fdbserver.cpp +++ b/fdbserver/fdbserver.cpp @@ -22,7 +22,6 @@ // a macro that makes boost interprocess break on Windows. #define BOOST_DATE_TIME_NO_LIB -#include "flow/ParseNumber.h" #include #include #include @@ -1341,13 +1340,11 @@ struct CLIOptions { } case OPT_NUMTESTERS: { const char* a = args.OptionArg(); - auto parsed = parseNumberPrefix(StringRef(a)); - if (!parsed.present()) { + if (!sscanf(a, "%d", &minTesterCount)) { fprintf(stderr, "ERROR: Could not parse numtesters `%s'\n", a); printHelpTeaser(argv[0]); flushAndExit(FDB_EXIT_ERROR); } - minTesterCount = parsed.get(); break; } case OPT_ROLLSIZE: { @@ -1388,13 +1385,8 @@ struct CLIOptions { } #ifdef _WIN32 case OPT_PARENTPID: { - const char* pid_str = args.OptionArg(); - auto parsedPid = parseNumberPrefix(StringRef(pid_str)); - if (!parsedPid.present() || parsedPid.get() <= 0) { - fprintf(stderr, "ERROR: Invalid parent process id `%s'\n", pid_str); - flushAndExit(FDB_EXIT_ERROR); - } - int parent_pid = parsedPid.get(); + auto pid_str = args.OptionArg(); + int parent_pid = atoi(pid_str); auto pHandle = OpenProcess(SYNCHRONIZE, FALSE, parent_pid); if (!pHandle) { TraceEvent("ParentProcessOpenError").GetLastError(); @@ -1416,13 +1408,9 @@ struct CLIOptions { break; #else case OPT_PARENTPID: { - const char* pid_str = args.OptionArg(); - auto parsedPid = parseNumberPrefix(StringRef(pid_str)); - if (!parsedPid.present() || parsedPid.get() <= 0) { - fprintf(stderr, "ERROR: Invalid parent process id `%s'\n", pid_str); - flushAndExit(FDB_EXIT_ERROR); - } - int* parent_pid = new int(parsedPid.get()); + auto pid_str = args.OptionArg(); + int* parent_pid = new (int); + *parent_pid = atoi(pid_str); startThread(&parentWatcher, parent_pid, 0, "fdb-parentwatch"); break; } @@ -1576,13 +1564,11 @@ struct CLIOptions { break; case OPT_IO_TRUST_SECONDS: { const char* a = args.OptionArg(); - auto parsed = parseNumberPrefix(StringRef(a)); - if (!parsed.present()) { + if (!sscanf(a, "%lf", &fileIoTimeout)) { fprintf(stderr, "ERROR: Could not parse io_trust_seconds `%s'\n", a); printHelpTeaser(argv[0]); flushAndExit(FDB_EXIT_ERROR); } - fileIoTimeout = parsed.get(); break; } case OPT_IO_TRUST_WARN_ONLY: @@ -2193,10 +2179,10 @@ int main(int argc, char* argv[]) { int backupFailed = true; const char* isRestoringStr = ini.GetValue("RESTORE", "isRestoring", nullptr); if (isRestoringStr) { - isRestoring = parseNumberPrefix(StringRef(isRestoringStr)).orDefault(0); + isRestoring = atoi(isRestoringStr); const char* backupFailedStr = ini.GetValue("RESTORE", "BackupFailed", nullptr); if (isRestoring && backupFailedStr) { - backupFailed = parseNumberPrefix(StringRef(backupFailedStr)).orDefault(0); + backupFailed = atoi(backupFailedStr); } } if (isRestoring && !backupFailed) { diff --git a/fdbserver/networktest.cpp b/fdbserver/networktest.cpp index 98d1932bd10..4122777ae40 100644 --- a/fdbserver/networktest.cpp +++ b/fdbserver/networktest.cpp @@ -18,7 +18,6 @@ * limitations under the License. */ -#include "flow/ParseNumber.h" #include "fmt/format.h" #include "fdbserver/NetworkTest.h" #include "flow/ActorCollection.h" @@ -263,8 +262,8 @@ struct RandomIntRange { if (high.empty()) { high = low; } - min = low.empty() ? 0 : parseNumberPrefix(low).orDefault(0); - max = high.empty() ? 0 : parseNumberPrefix(high).orDefault(0); + min = low.empty() ? 0 : atol(low.toString().c_str()); + max = high.empty() ? 0 : atol(high.toString().c_str()); if (min > max) { std::swap(min, max); } diff --git a/fdbserver/tester/TestSpecParser.cpp b/fdbserver/tester/TestSpecParser.cpp index 31528fc67a0..86799505310 100644 --- a/fdbserver/tester/TestSpecParser.cpp +++ b/fdbserver/tester/TestSpecParser.cpp @@ -28,7 +28,6 @@ #include #include "flow/Platform.h" -#include "flow/ParseNumber.h" #include "flow/Trace.h" #include "flow/UnitTest.h" #include "fdbclient/NativeAPI.h" @@ -37,15 +36,6 @@ namespace { -template -T parseTestNumber(const std::string& value) { - auto parsed = parseNumberPrefix(value); - if (!parsed.present()) { - throw test_specification_invalid(); - } - return parsed.get(); -} - std::map> testSpecGlobalKeys = { // These are read by SimulatedCluster and used before testers exist. Thus, they must // be recognized and accepted, but there's no point in placing them into a testSpec. @@ -98,13 +88,14 @@ std::maptimeout = parseTestNumber(value); + sscanf(value.c_str(), "%d", &(spec->timeout)); ASSERT(spec->timeout > 0); TraceEvent("TestParserTest").detail("ParsedTimeout", spec->timeout); } }, { "databasePingDelay", [](const std::string& value, TestSpec* spec) { - double databasePingDelay = parseTestNumber(value); + double databasePingDelay; + sscanf(value.c_str(), "%lf", &databasePingDelay); ASSERT(databasePingDelay >= 0); if (!spec->useDB && databasePingDelay > 0) { TraceEvent(SevError, "TestParserError") @@ -142,7 +133,7 @@ std::mapstartDelay = parseTestNumber(value); + sscanf(value.c_str(), "%lf", &spec->startDelay); TraceEvent("TestParserTest").detail("ParsedStartDelay", spec->startDelay); } }, { "runConsistencyCheck", @@ -157,7 +148,7 @@ std::mapmaxDDRunTime = parseTestNumber(value); + sscanf(value.c_str(), "%lf", &(spec->maxDDRunTime)); ASSERT(spec->maxDDRunTime >= 0); TraceEvent("TestParserTest").detail("ParsedMaxDDRunTime", spec->maxDDRunTime); } }, @@ -187,7 +178,8 @@ std::map(value); + double connectionFailuresDisableDuration; + sscanf(value.c_str(), "%lf", &connectionFailuresDisableDuration); ASSERT(connectionFailuresDisableDuration >= 0); spec->simConnectionFailuresDisableDuration = connectionFailuresDisableDuration; TraceEvent("TestParserTest") @@ -261,26 +253,6 @@ std::string toml_to_string(const T& value) { } } -TEST_CASE("/fdbserver/tester/TestSpecParser/NumericOptions") { - TestSpec spec; - testSpecTestKeys.at("timeout")("200.0", &spec); - ASSERT_EQ(spec.timeout, 200); - testSpecTestKeys.at("startDelay")(" \t+1.25suffix", &spec); - ASSERT_EQ(spec.startDelay, 1.25); - for (const auto& [option, value] : std::vector>{ - { "timeout", "99999999999999999999" }, { "startDelay", "1e9999" }, { "databasePingDelay", "invalid" } }) { - try { - testSpecTestKeys.at(option)(value, &spec); - ASSERT(false); - } catch (Error& e) { - ASSERT_EQ(e.code(), error_code_test_specification_invalid); - } - } - ASSERT_EQ(spec.timeout, 200); - ASSERT_EQ(spec.startDelay, 1.25); - return Void(); -} - TEST_CASE("/fdbserver/tester/TestSpecParser/TOMLArrayToString") { std::istringstream input(R"( strings = ['a', 'b'] @@ -332,11 +304,11 @@ std::vector readTests(std::ifstream& ifs) { } testSpecTestKeys[attrib](value, &spec); - } else if (testSpecTestKeys.contains(attrib)) { + } else if (testSpecTestKeys.find(attrib) != testSpecTestKeys.end()) { if (parsingWorkloads) TraceEvent(SevError, "TestSpecTestParamInWorkload").detail("Attrib", attrib).detail("Value", value); testSpecTestKeys[attrib](value, &spec); - } else if (testSpecGlobalKeys.contains(attrib)) { + } else if (testSpecGlobalKeys.find(attrib) != testSpecGlobalKeys.end()) { if (!beforeFirstTest) TraceEvent(SevError, "TestSpecGlobalParamInTest").detail("Attrib", attrib).detail("Value", value); testSpecGlobalKeys[attrib](value); @@ -421,7 +393,7 @@ TestSet readTOMLTests_(std::string fileName) { if (k == "workload" || k == "knobs") { continue; } - if (testSpecTestKeys.contains(k)) { + if (testSpecTestKeys.find(k) != testSpecTestKeys.end()) { testSpecTestKeys[k](toml_to_string(v), &spec); } else { TraceEvent(SevError, "TestSpecUnrecognizedTestParam") diff --git a/fdbserver/tester/WorkloadUtils.cpp b/fdbserver/tester/WorkloadUtils.cpp index c68eaef2868..d84414b8287 100644 --- a/fdbserver/tester/WorkloadUtils.cpp +++ b/fdbserver/tester/WorkloadUtils.cpp @@ -21,28 +21,19 @@ #include #include #include -#include #include #include #include "flow/CoroUtils.h" #include "flow/DeterministicRandom.h" -#include "flow/ParseNumber.h" #include "flow/Trace.h" -#include "flow/UnitTest.h" #include "flow/genericactors.h" #include "fdbserver/core/ServerDBInfo.h" #include "fdbserver/tester/workloads.h" namespace { -template -Optional parseNumericOption(StringRef value) { - // Legacy options accept numeric prefixes, including "100000.0" for integers. - return parseNumberPrefix(value); -} - constexpr char HEX_CHAR_LOOKUP[16] = { '0', '1', '2', '3', '4', '5', '6', '7', '8', '9', 'a', 'b', 'c', 'd', 'e', 'f' }; Future> getMetricsCompoundWorkload(CompoundWorkload* self) { @@ -159,10 +150,10 @@ Value getOption(VectorRef options, Key key, Value defaultValue) { int getOption(VectorRef options, Key key, int defaultValue) { for (int i = 0; i < options.size(); i++) { if (options[i].key == key) { - auto r = parseNumericOption(options[i].value); - if (r.present()) { + int r; + if (sscanf(options[i].value.toString().c_str(), "%d", &r)) { options[i].value = ""_sr; - return r.get(); + return r; } else { TraceEvent(SevError, "InvalidTestOption").detail("OptionName", key); throw test_specification_invalid(); @@ -176,10 +167,10 @@ int getOption(VectorRef options, Key key, int defaultValue) { uint64_t getOption(VectorRef options, Key key, uint64_t defaultValue) { for (int i = 0; i < options.size(); i++) { if (options[i].key == key) { - auto r = parseNumericOption(options[i].value); - if (r.present()) { + uint64_t r; + if (sscanf(options[i].value.toString().c_str(), "%" SCNd64, &r)) { options[i].value = ""_sr; - return r.get(); + return r; } else { TraceEvent(SevError, "InvalidTestOption").detail("OptionName", key); throw test_specification_invalid(); @@ -193,10 +184,10 @@ uint64_t getOption(VectorRef options, Key key, uint64_t defaultValu int64_t getOption(VectorRef options, Key key, int64_t defaultValue) { for (int i = 0; i < options.size(); i++) { if (options[i].key == key) { - auto r = parseNumericOption(options[i].value); - if (r.present()) { + int64_t r; + if (sscanf(options[i].value.toString().c_str(), "%" SCNd64, &r)) { options[i].value = ""_sr; - return r.get(); + return r; } else { TraceEvent(SevError, "InvalidTestOption").detail("OptionName", key); throw test_specification_invalid(); @@ -210,11 +201,10 @@ int64_t getOption(VectorRef options, Key key, int64_t defaultValue) double getOption(VectorRef options, Key key, double defaultValue) { for (int i = 0; i < options.size(); i++) { if (options[i].key == key) { - // Preserve the float rounding used by existing simulation configurations. - auto r = parseNumericOption(options[i].value); - if (r.present()) { + float r; + if (sscanf(options[i].value.toString().c_str(), "%f", &r)) { options[i].value = ""_sr; - return r.get(); + return r; } } } @@ -255,22 +245,14 @@ std::vector getOption(VectorRef options, Key key, std::vector< for (int i = 0; i < options.size(); i++) { if (options[i].key == key) { std::vector v; - auto appendValue = [&](StringRef value) { - auto parsed = parseNumericOption(value); - if (!parsed.present()) { - TraceEvent(SevError, "InvalidTestOption").detail("OptionName", key); - throw test_specification_invalid(); - } - v.push_back(parsed.get()); - }; int begin = 0; for (int c = 0; c < options[i].value.size(); c++) { if (options[i].value[c] == ',') { - appendValue(options[i].value.substr(begin, c - begin)); + v.push_back(atoi((char*)options[i].value.begin() + begin)); begin = c + 1; } } - appendValue(options[i].value.substr(begin)); + v.push_back(atoi((char*)options[i].value.begin() + begin)); options[i].value = ""_sr; return v; } @@ -278,45 +260,6 @@ std::vector getOption(VectorRef options, Key key, std::vector< return defaultValue; } -TEST_CASE("/fdbserver/WorkloadUtils/numericOptions") { - ASSERT_EQ(parseNumericOption(" \t+0012 \n"_sr).get(), 12); - ASSERT_EQ(parseNumericOption("100000.0"_sr).get(), 100000); - ASSERT_EQ(parseNumericOption("12suffix"_sr).get(), 12); - ASSERT_EQ(parseNumericOption("12\0suffix"_sr).get(), 12); - ASSERT_EQ(parseNumericOption("-9223372036854775808"_sr).get(), std::numeric_limits::min()); - ASSERT_EQ(parseNumericOption("9223372036854775807"_sr).get(), std::numeric_limits::max()); - ASSERT_EQ(parseNumericOption("18446744073709551615"_sr).get(), std::numeric_limits::max()); - ASSERT_EQ(parseNumericOption("-1"_sr).get(), std::numeric_limits::max()); - ASSERT_EQ(parseNumericOption(" \t+1.25e2 \n"_sr).get(), 125.0f); - ASSERT_EQ(parseNumericOption("1.25suffix"_sr).get(), 1.25f); - ASSERT(!parseNumericOption("9223372036854775808"_sr).present()); - ASSERT(!parseNumericOption("-9223372036854775809"_sr).present()); - ASSERT(!parseNumericOption("18446744073709551616"_sr).present()); - ASSERT( - !parseNumericOption(std::to_string(static_cast(std::numeric_limits::max()) + 1)).present()); - ASSERT(!parseNumericOption("1e9999"_sr).present()); - ASSERT(!parseNumericOption("1e-9999"_sr).present()); - for (StringRef text : { ""_sr, " \t"_sr, "+"_sr, "suffix12"_sr }) { - ASSERT(!parseNumericOption(text).present()); - ASSERT(!parseNumericOption(text).present()); - } - - const uint8_t backing[] = { '1', ',', '2', '3', 0 }; - Standalone> options; - options.push_back(options.arena(), KeyValueRef("integers"_sr, StringRef(backing, 3))); - const std::vector parsed = getOption(options, "integers"_sr, std::vector{}); - ASSERT(parsed == std::vector({ 1, 2 })); - ASSERT(options[0].value.empty()); - - options.push_back(options.arena(), KeyValueRef("double"_sr, "0.1"_sr)); - ASSERT_EQ(getOption(options, "double"_sr, 0.0), static_cast(0.1f)); - options.push_back(options.arena(), KeyValueRef("invalidDouble"_sr, "invalid"_sr)); - ASSERT_EQ(getOption(options, "invalidDouble"_sr, 3.5), 3.5); - ASSERT(options[2].value == "invalid"_sr); - - return Void(); -} - bool hasOption(VectorRef options, Key key) { for (const auto& option : options) { if (option.key == key) { diff --git a/fdbserver/worker/worker.cpp b/fdbserver/worker/worker.cpp index 1c543cd51f5..f47596402e9 100644 --- a/fdbserver/worker/worker.cpp +++ b/fdbserver/worker/worker.cpp @@ -18,7 +18,6 @@ * limitations under the License. */ -#include "flow/ParseNumber.h" #include #include #include @@ -1492,13 +1491,7 @@ Future runProfiler(ProfilerRequest req) { bool checkHighMemory(int64_t threshold, bool* error) { #if defined(__linux__) && defined(USE_GPERFTOOLS) && !defined(VALGRIND) *error = false; - const long pageSizeResult = sysconf(_SC_PAGESIZE); - if (pageSizeResult <= 0) { - TraceEvent("GetPageSizeFailure").log(); - *error = true; - return false; - } - const uint64_t page_size = static_cast(pageSizeResult); + uint64_t page_size = sysconf(_SC_PAGESIZE); int fd = open("/proc/self/statm", O_RDONLY | O_CLOEXEC); if (fd < 0) { TraceEvent("OpenStatmFileFailure").log(); @@ -1509,23 +1502,15 @@ bool checkHighMemory(int64_t threshold, bool* error) { const int buf_sz = 256; char stat_buf[buf_sz]; ssize_t stat_nread = read(fd, stat_buf, buf_sz); - close(fd); - if (stat_nread <= 0) { + if (stat_nread < 0) { TraceEvent("ReadStatmFileFailure").log(); *error = true; return false; } - StringRef statText(reinterpret_cast(stat_buf), stat_nread); - int consumed = 0; - auto vmsize = parseNumberPrefix(statText, 10, &consumed); - auto rssPages = parseNumberPrefix(statText.substr(consumed)); - if (!vmsize.present() || !rssPages.present() || rssPages.get() > std::numeric_limits::max() / page_size) { - TraceEvent("ParseStatmFileFailure").log(); - *error = true; - return false; - } - uint64_t rss = rssPages.get() * page_size; + uint64_t vmsize, rss; + sscanf(stat_buf, "%lu %lu", &vmsize, &rss); + rss *= page_size; if (rss >= threshold) { return true; } diff --git a/fdbserver/workloads/CommitBugCheck.cpp b/fdbserver/workloads/CommitBugCheck.cpp index 0eac6af5e66..fc4c7549b09 100644 --- a/fdbserver/workloads/CommitBugCheck.cpp +++ b/fdbserver/workloads/CommitBugCheck.cpp @@ -18,7 +18,6 @@ * limitations under the License. */ -#include "flow/ParseNumber.h" #include "fdbserver/core/TesterInterface.h" #include "fdbserver/tester/workloads.h" @@ -116,7 +115,7 @@ struct CommitBugWorkload : TestWorkload { Optional val = co_await tr.get(key); int num = 0; if (val.present()) { - num = parseNumberPrefix(val.get()).orDefault(0); + num = atoi(val.get().toString().c_str()); if (num != i) { TraceEvent(SevError, "CommitBug2Failed").detail("Value", num).detail("Expected", i); self->success = false; diff --git a/fdbserver/workloads/ConfigureDatabase.cpp b/fdbserver/workloads/ConfigureDatabase.cpp index 8f76c54918b..3574995a549 100644 --- a/fdbserver/workloads/ConfigureDatabase.cpp +++ b/fdbserver/workloads/ConfigureDatabase.cpp @@ -18,7 +18,6 @@ * limitations under the License. */ -#include "flow/ParseNumber.h" #include #include "fdbclient/FDBTypes.h" @@ -257,7 +256,11 @@ struct ConfigureDatabaseWorkload : TestWorkload { void getMetrics(std::vector& m) override { m.push_back(retries.getMetric()); } - static inline uint64_t valueToUInt64(const StringRef& v) { return parseNumberPrefix(v, 16).orDefault(0); } + static inline uint64_t valueToUInt64(const StringRef& v) { + long long unsigned int x = 0; + sscanf(v.toString().c_str(), "%llx", &x); + return x; + } inline Standalone getDatabaseName(int dbIndex) { return StringRef(format("DestroyDB%d", dbIndex)); } diff --git a/fdbserver/workloads/DDMetrics.cpp b/fdbserver/workloads/DDMetrics.cpp index 6415958fbfe..6094184695c 100644 --- a/fdbserver/workloads/DDMetrics.cpp +++ b/fdbserver/workloads/DDMetrics.cpp @@ -38,7 +38,8 @@ struct DDMetricsWorkload : TestWorkload { TraceEvent("GetHighPriorityReliocationsInFlight").detail("Stage", "ContactingMaster"); TraceEventFields md = co_await timeoutError(masterWorker.eventLogRequest.getReply(EventLogRequest("MovingData"_sr)), 1.0); - int relocations = md.getInt("UnhealthyRelocations"); + int relocations; + sscanf(md.getValue("UnhealthyRelocations").c_str(), "%d", &relocations); co_return relocations; } diff --git a/fdbserver/workloads/FileSystem.cpp b/fdbserver/workloads/FileSystem.cpp index 76725ee6e19..6702892440b 100644 --- a/fdbserver/workloads/FileSystem.cpp +++ b/fdbserver/workloads/FileSystem.cpp @@ -18,7 +18,6 @@ * limitations under the License. */ -#include "flow/ParseNumber.h" #include "fdbrpc/DDSketch.h" #include "fdbclient/NativeAPI.h" #include "fdbserver/core/TesterInterface.h" @@ -221,7 +220,11 @@ struct FileSystemWorkload : TestWorkload { } } - static int testKeyToInt(const KeyRef& p) { return parseNumberPrefix(p).orDefault(0); } + static int testKeyToInt(const KeyRef& p) { + int x = 0; + sscanf(p.toString().c_str(), "%d", &x); + return x; + } Future writeClient(Database cx, FileSystemWorkload* self) { double clientBegin = now(); diff --git a/fdbserver/workloads/HTTPKeyValueStore.cpp b/fdbserver/workloads/HTTPKeyValueStore.cpp index 1b79b7e1e1e..eff3c1ff0a8 100644 --- a/fdbserver/workloads/HTTPKeyValueStore.cpp +++ b/fdbserver/workloads/HTTPKeyValueStore.cpp @@ -18,7 +18,6 @@ * limitations under the License. */ -#include "flow/ParseNumber.h" #include "flow/Arena.h" #include "flow/IRandom.h" #include "flow/Trace.h" @@ -113,13 +112,8 @@ Future httpKVRequestCallback(Reference kvStore, ASSERT(req->data.headers.contains("UID")); ASSERT(req->data.headers.contains("SeqNo")); - auto clientIdValue = parseNumberPrefix(StringRef(req->data.headers["ClientID"])); - auto seqNoValue = parseNumberPrefix(StringRef(req->data.headers["SeqNo"])); - if (!clientIdValue.present() || !seqNoValue.present()) { - throw http_request_failed(); - } - int clientId = clientIdValue.get(); - int seqNo = seqNoValue.get(); + int clientId = atoi(req->data.headers["ClientID"].c_str()); + int seqNo = atoi(req->data.headers["SeqNo"].c_str()); ASSERT(req->data.headers.contains("Content-Length")); ASSERT_EQ(req->data.headers["Content-Length"], std::to_string(req->data.content.size())); diff --git a/fdbserver/workloads/Inventory.cpp b/fdbserver/workloads/Inventory.cpp index f1460502304..81ca552a2be 100644 --- a/fdbserver/workloads/Inventory.cpp +++ b/fdbserver/workloads/Inventory.cpp @@ -18,7 +18,6 @@ * limitations under the License. */ -#include "flow/ParseNumber.h" #include "fdbclient/NativeAPI.h" #include "fdbserver/core/TesterInterface.h" #include "fdbserver/tester/workloads.h" @@ -122,7 +121,7 @@ struct InventoryTestWorkload : TestWorkload { std::map actualResults; for (int i = 0; i < data.size(); i++) - actualResults[data[i].key] = parseNumberPrefix(data[i].value).orDefault(0); + actualResults[data[i].key] = atoi(data[i].value.toString().c_str()); for (auto i = self->minExpectedResults.begin(); i != self->minExpectedResults.end(); ++i) actualResults[i->first]; bool error = false; @@ -156,7 +155,7 @@ struct InventoryTestWorkload : TestWorkload { Future inventoryTestWrite(Transaction* tr, Key key) { Optional val = co_await tr->get(key); - int count = !val.present() ? 0 : parseNumberPrefix(val.get()).orDefault(0); + int count = !val.present() ? 0 : atoi(val.get().toString().c_str()); ASSERT(count >= 0 && count < 1000000); tr->set(key, format("%d", count + 1)); } diff --git a/fdbserver/workloads/QueuePush.cpp b/fdbserver/workloads/QueuePush.cpp index 9439f0122fa..8a376477247 100644 --- a/fdbserver/workloads/QueuePush.cpp +++ b/fdbserver/workloads/QueuePush.cpp @@ -17,7 +17,6 @@ * See the License for the specific language governing permissions and * limitations under the License. */ -#include "flow/ParseNumber.h" #include #include "fdbclient/FDBTypes.h" @@ -81,11 +80,12 @@ struct QueuePushWorkload : TestWorkload { static Key keyForIndex(int base, int offset) { return StringRef(format("%08x%08x", base, offset)); } static std::pair valuesForKey(KeyRef value) { + int base, offset; ASSERT(value.size() == 16); - auto base = parseNumberPrefix(value.substr(0, 8), 16); - auto offset = parseNumberPrefix(value.substr(8, 8), 16); - if (base.present() && offset.present()) { - return std::make_pair(static_cast(base.get()), static_cast(offset.get())); + + if (sscanf(value.substr(0, 8).toString().c_str(), "%x", &base) && + sscanf(value.substr(8, 8).toString().c_str(), "%x", &offset)) { + return std::make_pair(base, offset); } else { // SOMEDAY: what should this really be? Should we rely on exceptions for control flow here? throw client_invalid_operation(); diff --git a/fdbserver/workloads/SnapTest.cpp b/fdbserver/workloads/SnapTest.cpp index b7c30a2cf8f..1dde81408a5 100644 --- a/fdbserver/workloads/SnapTest.cpp +++ b/fdbserver/workloads/SnapTest.cpp @@ -18,7 +18,6 @@ * limitations under the License. */ -#include "flow/ParseNumber.h" #include #include "fdbclient/ManagementAPI.h" #include "fdbclient/NativeAPI.h" @@ -219,13 +218,7 @@ struct SnapTestWorkload : TestWorkload { CSimpleIni ini; ini.SetUnicode(); ini.LoadFile(restartInfoLocation.c_str()); - const char* backupFailedText = ini.GetValue("RESTORE", "BackupFailed"); - auto backupFailedValue = - backupFailedText == nullptr ? Optional() : parseNumberPrefix(StringRef(backupFailedText)); - if (!backupFailedValue.present()) { - throw test_specification_invalid(); - } - bool backupFailed = backupFailedValue.get(); + bool backupFailed = atoi(ini.GetValue("RESTORE", "BackupFailed")); if (backupFailed) { // since backup failed, skip the restore checking TraceEvent(SevWarnAlways, "BackupFailedSkippingRestoreCheck").log(); diff --git a/fdbserver/workloads/Storefront.cpp b/fdbserver/workloads/Storefront.cpp index 1e14042821a..9c7c8589615 100644 --- a/fdbserver/workloads/Storefront.cpp +++ b/fdbserver/workloads/Storefront.cpp @@ -18,7 +18,6 @@ * limitations under the License. */ -#include "flow/ParseNumber.h" #include "fdbclient/NativeAPI.h" #include "fdbserver/core/TesterInterface.h" #include "fdbserver/tester/workloads.h" @@ -88,7 +87,11 @@ struct StorefrontWorkload : TestWorkload { return x; }*/ - static inline int valueToInt(const StringRef& v) { return parseNumberPrefix(v).orDefault(0); } + static inline int valueToInt(const StringRef& v) { + int x = 0; + sscanf(v.toString().c_str(), "%d", &x); + return x; + } Key keyForIndex(int n) { return itemKey(n); } Key itemKey(int item) { return StringRef(format("/items/%016d", item)); } diff --git a/fdbserver/workloads/TaskBucketCorrectness.cpp b/fdbserver/workloads/TaskBucketCorrectness.cpp index 3bd6e1abb48..f51cd58f6f0 100644 --- a/fdbserver/workloads/TaskBucketCorrectness.cpp +++ b/fdbserver/workloads/TaskBucketCorrectness.cpp @@ -18,7 +18,6 @@ * limitations under the License. */ -#include "flow/ParseNumber.h" #include "flow/UnitTest.h" #include "flow/Error.h" #include "fdbclient/Tuple.h" @@ -77,8 +76,8 @@ struct SayHelloTaskFunc : TaskFuncBase { if (!task->params["chained"_sr].compare("false"_sr)) { co_await done->set(tr, taskBucket); } else { - int subtaskCount = parseNumberPrefix(task->params["subtaskCount"_sr]).orDefault(0); - int currTaskNumber = parseNumberPrefix(value.removePrefix("task_"_sr)).orDefault(0); + int subtaskCount = atoi(task->params["subtaskCount"_sr].toString().c_str()); + int currTaskNumber = atoi(value.removePrefix("task_"_sr).toString().c_str()); TraceEvent("TaskBucketCorrectnessSayHello") .detail("SubtaskCount", subtaskCount) .detail("CurrTaskNumber", currTaskNumber); @@ -135,7 +134,7 @@ struct SayHelloToEveryoneTaskFunc : TaskFuncBase { int subtaskCount = 1; if (!task->params["chained"_sr].compare("false"_sr)) { - subtaskCount = parseNumberPrefix(task->params["subtaskCount"_sr]).orDefault(0); + subtaskCount = atoi(task->params["subtaskCount"_sr].toString().c_str()); } for (int i = 0; i < subtaskCount; ++i) { auto new_task = makeReference( diff --git a/fdbserver/workloads/pubsub.cpp b/fdbserver/workloads/pubsub.cpp index 2ecaea625cb..b372f45f81f 100644 --- a/fdbserver/workloads/pubsub.cpp +++ b/fdbserver/workloads/pubsub.cpp @@ -18,7 +18,6 @@ * limitations under the License. */ -#include "flow/ParseNumber.h" #include #include "fdbclient/NativeAPI.h" #include "pubsub.h" @@ -27,7 +26,9 @@ Value uInt64ToValue(uint64_t v) { return StringRef(format("%016llx", v)); } uint64_t valueToUInt64(const StringRef& v) { - return parseNumberPrefix(v, 16).orDefault(0); + uint64_t x = 0; + sscanf(v.toString().c_str(), "%" SCNx64, &x); + return x; } Key keyForInbox(uint64_t inbox) { diff --git a/flow/Platform.cpp b/flow/Platform.cpp index c6a4a23ca2b..2fbdafdfc41 100644 --- a/flow/Platform.cpp +++ b/flow/Platform.cpp @@ -24,7 +24,6 @@ #endif // _WIN32 #include "flow/Platform.h" -#include "flow/ParseNumber.h" #include #include @@ -1528,8 +1527,7 @@ void initPdhStrings(SystemStatisticsState* state, std::string dataFolder) { "PdhEnumObjectItems")) { char* ptr = buf; while (*ptr) { - auto deviceNumber = parseNumberPrefix(StringRef(static_cast(ptr))); - if (isdigit(*ptr) && deviceNumber.present() && deviceNumber.get() == storage_device.DeviceNumber) { + if (isdigit(*ptr) && atoi(ptr) == storage_device.DeviceNumber) { state->pdhStrings.diskDevice = ptr; break; } diff --git a/flow/Profiler.cpp b/flow/Profiler.cpp index b412aae55f6..2fba014e50d 100644 --- a/flow/Profiler.cpp +++ b/flow/Profiler.cpp @@ -20,7 +20,6 @@ #include "flow/flow.h" #include "flow/network.h" -#include "flow/ParseNumber.h" #ifdef __linux__ @@ -276,7 +275,7 @@ void startProfiling(INetwork* network, period = maybePeriod.get(); } else { const char* periodEnv = getenv("FLOW_PROFILER_PERIOD"); - period = (periodEnv ? parseNumberPrefix(StringRef(periodEnv)).orDefault(0) : 2000); + period = (periodEnv ? atoi(periodEnv) : 2000); } std::string outputFile; if (maybeOutputFile.present()) { diff --git a/flow/Trace.cpp b/flow/Trace.cpp index 0738e6b4249..8c3bf1c8cd2 100644 --- a/flow/Trace.cpp +++ b/flow/Trace.cpp @@ -26,9 +26,6 @@ #include "flow/flow.h" #include "flow/DeterministicRandom.h" #include "flow/ProcessEvents.h" -#include "flow/UnitTest.h" -#include -#include #include #include #include @@ -1684,52 +1681,56 @@ TraceEventFields::Field& TraceEventFields::mutate(int index) { } namespace { -void checkNumericConversion(std::string const& s, const char* end, bool permissive, int conversionError) { - if (end == s.c_str() || (!permissive && end != s.c_str() + s.size())) { - throw attribute_not_found(); - } - if (conversionError == ERANGE) { - throw attribute_too_large(); +void parseNumericValue(std::string const& s, double& outValue, bool permissive = false) { + double d = 0; + int consumed = 0; + int r = sscanf(s.c_str(), "%lf%n", &d, &consumed); + if (r == 1 && (consumed == s.size() || permissive)) { + outValue = d; + return; } -} -void parseNumericValue(std::string const& s, double& outValue, bool permissive = false) { - char* end = nullptr; - errno = 0; - double d = std::strtod(s.c_str(), &end); - checkNumericConversion(s, end, permissive, errno); - outValue = d; + throw attribute_not_found(); } -void parseNumericValue(std::string const& s, int64_t& outValue, bool permissive = false) { - char* end = nullptr; - errno = 0; - long long i = std::strtoll(s.c_str(), &end, 10); - checkNumericConversion(s, end, permissive, errno); - if (i < std::numeric_limits::min() || i > std::numeric_limits::max()) { - throw attribute_too_large(); +void parseNumericValue(std::string const& s, int& outValue, bool permissive = false) { + long long int iLong = 0; + int consumed = 0; + int r = sscanf(s.c_str(), "%lld%n", &iLong, &consumed); + if (r == 1 && (consumed == s.size() || permissive)) { + if (std::numeric_limits::min() <= iLong && iLong <= std::numeric_limits::max()) { + outValue = (int)iLong; // Downcast definitely safe + return; + } else { + throw attribute_too_large(); + } } - outValue = i; + + throw attribute_not_found(); } -void parseNumericValue(std::string const& s, int& outValue, bool permissive = false) { - int64_t i; - parseNumericValue(s, i, permissive); - if (i < std::numeric_limits::min() || i > std::numeric_limits::max()) { - throw attribute_too_large(); +void parseNumericValue(std::string const& s, int64_t& outValue, bool permissive = false) { + long long int i = 0; + int consumed = 0; + int r = sscanf(s.c_str(), "%lld%n", &i, &consumed); + if (r == 1 && (consumed == s.size() || permissive)) { + outValue = i; + return; } - outValue = static_cast(i); + + throw attribute_not_found(); } void parseNumericValue(std::string const& s, uint64_t& outValue, bool permissive = false) { - char* end = nullptr; - errno = 0; - unsigned long long i = (std::strtoull)(s.c_str(), &end, 10); - checkNumericConversion(s, end, permissive, errno); - if (i > std::numeric_limits::max()) { - throw attribute_too_large(); + unsigned long long int i = 0; + int consumed = 0; + int r = sscanf(s.c_str(), "%llu%n", &i, &consumed); + if (r == 1 && (consumed == s.size() || permissive)) { + outValue = i; + return; } - outValue = i; + + throw attribute_not_found(); } template @@ -1757,72 +1758,6 @@ bool getNumericValue(TraceEventFields const& fields, std::string key, T& outValu } } } - -TEST_CASE("/flow/TraceEventFields/numericParsing") { - auto expectFailure = [](const std::string& text, auto initial, int errorCode, bool permissive = false) { - auto value = initial; - try { - parseNumericValue(text, value, permissive); - } catch (Error& e) { - ASSERT_EQ(e.code(), errorCode); - ASSERT_EQ(value, initial); - return; - } - ASSERT(false); - }; - - int64_t signedValue = 0; - parseNumericValue(" \t+0012", signedValue); - ASSERT_EQ(signedValue, int64_t{ 12 }); - parseNumericValue("-9223372036854775808", signedValue); - ASSERT_EQ(signedValue, std::numeric_limits::min()); - parseNumericValue("9223372036854775807", signedValue); - ASSERT_EQ(signedValue, std::numeric_limits::max()); - expectFailure("9223372036854775808", int64_t{ 123 }, error_code_attribute_too_large); - expectFailure("-9223372036854775809", int64_t{ 123 }, error_code_attribute_too_large); - - int intValue = 0; - parseNumericValue(std::to_string(std::numeric_limits::min()), intValue); - ASSERT_EQ(intValue, std::numeric_limits::min()); - parseNumericValue(std::to_string(std::numeric_limits::max()), intValue); - ASSERT_EQ(intValue, std::numeric_limits::max()); - expectFailure( - std::to_string(static_cast(std::numeric_limits::max()) + 1), 123, error_code_attribute_too_large); - - uint64_t unsignedValue = 0; - parseNumericValue(" \t+18446744073709551615", unsignedValue); - ASSERT_EQ(unsignedValue, std::numeric_limits::max()); - parseNumericValue("-1", unsignedValue); - ASSERT_EQ(unsignedValue, std::numeric_limits::max()); - expectFailure("18446744073709551616", uint64_t{ 123 }, error_code_attribute_too_large); - - double doubleValue = 0; - parseNumericValue(" \t+1.25e2", doubleValue); - ASSERT_EQ(doubleValue, 125.0); - expectFailure("1e9999", 123.0, error_code_attribute_too_large); - expectFailure("1e-9999", 123.0, error_code_attribute_too_large); - expectFailure("", 123.0, error_code_attribute_not_found); - expectFailure(" \t", int64_t{ 123 }, error_code_attribute_not_found, true); - expectFailure("12suffix", 123, error_code_attribute_not_found); - parseNumericValue("12suffix", intValue, true); - ASSERT_EQ(intValue, 12); - expectFailure(std::string("12\0suffix", 9), 123, error_code_attribute_not_found); - parseNumericValue(std::string("12\0suffix", 9), intValue, true); - ASSERT_EQ(intValue, 12); - expectFailure("9223372036854775808suffix", int64_t{ 123 }, error_code_attribute_not_found); - expectFailure("9223372036854775808suffix", int64_t{ 123 }, error_code_attribute_too_large, true); - - TraceEventFields fields; - fields.addField("overflow", "18446744073709551616"); - unsignedValue = 123; - ASSERT(!fields.tryGetUint64("overflow", unsignedValue)); - ASSERT_EQ(unsignedValue, uint64_t{ 123 }); - fields.addField("number", "12suffix"); - ASSERT(!fields.tryGetInt("number", intValue)); - ASSERT_EQ(fields.getInt("number", true), 12); - - return Void(); -} } // namespace bool TraceEventFields::tryGetInt(std::string key, int& outVal, bool permissive) const { diff --git a/flow/UnitTest.cpp b/flow/UnitTest.cpp index 189e9fe4f53..e13f0aeed5b 100644 --- a/flow/UnitTest.cpp +++ b/flow/UnitTest.cpp @@ -19,12 +19,6 @@ */ #include "flow/UnitTest.h" -#include "flow/ParseNumber.h" - -#include -#include -#include -#include UnitTestCollection g_unittests = { nullptr }; @@ -57,11 +51,7 @@ void UnitTestParameters::set(const std::string& name, double value) { Optional UnitTestParameters::getInt(const std::string& name) const { auto opt = get(name); if (opt.present()) { - auto parsed = parseNumber(opt.get()); - if (!parsed.present()) { - throw invalid_option_value(); - } - return parsed; + return atoll(opt.get().c_str()); } return {}; } @@ -69,11 +59,7 @@ Optional UnitTestParameters::getInt(const std::string& name) const { Optional UnitTestParameters::getDouble(const std::string& name) const { auto opt = get(name); if (opt.present()) { - auto parsed = parseNumber(opt.get()); - if (!parsed.present()) { - throw invalid_option_value(); - } - return parsed; + return atof(opt.get().c_str()); } return {}; } @@ -86,101 +72,6 @@ void UnitTestParameters::setDataDir(std::string const& dataDir) { this->dataDir = dataDir; } -TEST_CASE("/flow/ParseNumber/checked") { - int consumed = -1; - ASSERT_EQ(parseNumberPrefix(" \t+12suffix"_sr, 10, &consumed).get(), 12); - ASSERT_EQ(consumed, 5); - ASSERT(!parseNumber(" \t+12suffix"_sr).present()); - ASSERT_EQ(parseNumber(" \t+12"_sr).get(), 12); - ASSERT(!parseNumber("12 "_sr).present()); - ASSERT(!parseNumber("12\0suffix"_sr).present()); - ASSERT_EQ(parseNumberPrefix("12\0suffix"_sr).get(), 12); - ASSERT_EQ(parseNumber("ffffffffffffffff"_sr, 16).get(), std::numeric_limits::max()); - ASSERT_EQ(parseNumber("0xff"_sr, 0).get(), 255); - ASSERT_EQ(parseNumber("ff"_sr, 16).get(), 255); - ASSERT_EQ(parseNumber("010"_sr).get(), 10); - ASSERT_EQ(parseNumber("-1"_sr).get(), std::numeric_limits::max()); - ASSERT(!parseNumber("-1"_sr).present()); - ASSERT(!parseNumber("256"_sr).present()); - ASSERT(!parseNumber("128"_sr).present()); - ASSERT(!parseNumber("-129"_sr).present()); - ASSERT_EQ(parseNumber("0.1"_sr).get(), 0.1f); - ASSERT_EQ(parseNumber("0.1"_sr).get(), 0.1); - ASSERT(parseNumber("1.25"_sr).get() == 1.25L); - ASSERT_EQ(parseNumber("0x1p2"_sr).get(), 4.0); - ASSERT(!parseNumber("1e9999"_sr).present()); - ASSERT(!parseNumber("1e-9999"_sr).present()); - ASSERT(!parseNumber("1"_sr, 1).present()); - ASSERT(!parseNumber("1"_sr, 16).present()); - for (StringRef text : { ""_sr, "+"_sr, " \t"_sr, "9223372036854775808"_sr }) { - consumed = -1; - ASSERT(!parseNumberPrefix(text, 10, &consumed).present()); - ASSERT_EQ(consumed, -1); - } - const uint8_t backing[] = { '1', '2', '3', 0 }; - ASSERT_EQ(parseNumber(StringRef(backing, 2)).get(), 12); - return Void(); -} - -TEST_CASE("/flow/UnitTestParameters/numericValues") { - UnitTestParameters numericParams; - ASSERT(!numericParams.getInt("missing").present()); - ASSERT(!numericParams.getDouble("missing").present()); - - const int64_t intMin = std::numeric_limits::min(); - const int64_t intMax = std::numeric_limits::max(); - for (const auto& [text, expected] : - std::vector>{ { "0", 0 }, - { " \t+0012", 12 }, - { "-1", -1 }, - { std::to_string(intMin), intMin }, - { std::to_string(intMax), intMax } }) { - numericParams.set("integer", text); - errno = ERANGE; - ASSERT_EQ(numericParams.getInt("integer").get(), expected); - } - for (const std::string& text : std::vector{ - "", " \t", "+", "1x", "1 ", "9223372036854775808", "-9223372036854775809", std::string("12\0junk", 7) }) { - numericParams.set("integer", text); - try { - (void)numericParams.getInt("integer"); - ASSERT(false); - } catch (Error& e) { - ASSERT_EQ(e.code(), error_code_invalid_option_value); - } - } - - for (const auto& [text, expected] : std::vector>{ - { "0", 0.0 }, - { " \t+1.25e2", 125.0 }, - { "-0x1p2", -4.0 }, - { ".5", 0.5 }, - { "1.7976931348623157e308", std::numeric_limits::max() }, - { "2.2250738585072014e-308", std::numeric_limits::min() } }) { - numericParams.set("double", text); - errno = ERANGE; - ASSERT_EQ(numericParams.getDouble("double").get(), expected); - } - numericParams.set("double", std::string("nan")); - ASSERT(std::isnan(numericParams.getDouble("double").get())); - numericParams.set("double", std::string("inf")); - ASSERT(std::isinf(numericParams.getDouble("double").get())); - numericParams.set("double", std::string("-0")); - ASSERT(std::signbit(numericParams.getDouble("double").get())); - for (const std::string& text : - std::vector{ "", " \t", "+", "1x", "1 ", "1e9999", "1e-9999", std::string("12\0junk", 7) }) { - numericParams.set("double", text); - try { - (void)numericParams.getDouble("double"); - ASSERT(false); - } catch (Error& e) { - ASSERT_EQ(e.code(), error_code_invalid_option_value); - } - } - - return Void(); -} - TEST_CASE("/flow/UnitTestParameters/coroutineOwnership") { const std::string marker = "unitTestParameterOwnershipProbe"; if (params.get(marker).present()) { diff --git a/flow/UnitTestRunner.cpp b/flow/UnitTestRunner.cpp index f5cf9e32d1f..42448812780 100644 --- a/flow/UnitTestRunner.cpp +++ b/flow/UnitTestRunner.cpp @@ -31,13 +31,10 @@ #include #include -#include -#include #include #include #include #include -#include #include #include #include @@ -115,10 +112,8 @@ void printUsage(const char* program, const UnitTestRunnerConfig& config) { bool parseInt(const char* text, int* value) { char* end = nullptr; - errno = 0; long parsed = strtol(text, &end, 10); - if (end == text || *end != '\0' || errno == ERANGE || parsed < std::numeric_limits::min() || - parsed > std::numeric_limits::max()) { + if (*text == '\0' || *end != '\0') { return false; } *value = static_cast(parsed); @@ -126,75 +121,15 @@ bool parseInt(const char* text, int* value) { } bool parseUInt64(const char* text, uint64_t* value) { - const char* first = text; - while (std::isspace(static_cast(*first))) { - ++first; - } - if (*first == '-') { - return false; - } - char* end = nullptr; - errno = 0; - unsigned long long parsed = strtoull(text, &end, 10); - if (end == text || *end != '\0' || errno == ERANGE || parsed > std::numeric_limits::max()) { + uint64_t parsed = strtoull(text, &end, 10); + if (*text == '\0' || *end != '\0') { return false; } *value = parsed; return true; } -TEST_CASE("/flow/UnitTestRunner/numericOptions") { - const int intMin = std::numeric_limits::min(); - const int intMax = std::numeric_limits::max(); - for (const auto& [text, expected] : - std::vector>{ { "0", 0 }, - { " \t+0012", 12 }, - { "-1", -1 }, - { std::to_string(intMin), intMin }, - { std::to_string(intMax), intMax } }) { - int value = 123; - ASSERT(parseInt(text.c_str(), &value)); - ASSERT_EQ(value, expected); - } - for (const std::string& text : std::vector{ "", - " \t", - "+", - "1x", - "1 ", - std::to_string(static_cast(intMin) - 1), - std::to_string(static_cast(intMax) + 1), - "999999999999999999999999999999" }) { - int value = 123; - ASSERT(!parseInt(text.c_str(), &value)); - ASSERT_EQ(value, 123); - } - - const uint64_t uintMax = std::numeric_limits::max(); - for (const auto& [text, expected] : std::vector>{ - { "0", 0 }, { " \t+0012", 12 }, { std::to_string(uintMax), uintMax } }) { - uint64_t value = 123; - ASSERT(parseUInt64(text.c_str(), &value)); - ASSERT_EQ(value, expected); - } - for (const char* text : { "", - " \t", - "+", - "1x", - "1 ", - "-1", - " \t-1", - "-0", - "18446744073709551616", - "999999999999999999999999999999" }) { - uint64_t value = 123; - ASSERT(!parseUInt64(text, &value)); - ASSERT_EQ(value, uint64_t{ 123 }); - } - - return Void(); -} - bool parseArgs(int argc, char** argv, UnitTestRunnerOptions* options) { CSimpleOpt args(argc, argv, unitTestRunnerOptions, SO_O_EXACT | SO_O_HYPHEN_TO_UNDERSCORE); while (args.Next()) { diff --git a/flow/flow.cpp b/flow/flow.cpp index 2ddd637b40f..4dc1e790608 100644 --- a/flow/flow.cpp +++ b/flow/flow.cpp @@ -32,7 +32,6 @@ #include "flow/DeterministicRandom.h" #include "flow/Error.h" #include "flow/Hostname.h" -#include "flow/ParseNumber.h" #include "flow/Util.h" #include "rte_memcpy.h" #include "flow/UnitTest.h" @@ -139,10 +138,10 @@ std::string UID::toString() const { UID UID::fromString(std::string const& s) { ASSERT_EQ(s.size(), 32); - auto a = parseNumber(StringRef(s).substr(0, 16), 16); - auto b = parseNumber(StringRef(s).substr(16), 16); - ASSERT(a.present() && b.present()); - return UID(a.get(), b.get()); + uint64_t a = 0, b = 0; + int r = sscanf(s.c_str(), "%16" SCNx64 "%16" SCNx64, &a, &b); + ASSERT_EQ(r, 2); + return UID(a, b); } UID UID::fromStringThrowsOnFailure(std::string const& s) { @@ -150,31 +149,20 @@ UID UID::fromStringThrowsOnFailure(std::string const& s) { // invalid string size throw operation_failed(); } - auto a = parseNumber(StringRef(s).substr(0, 16), 16); - auto b = parseNumber(StringRef(s).substr(16), 16); - if (!a.present() || !b.present()) { + // Split into two 16-character hex strings and parse using strtoull + std::string first_half = s.substr(0, 16); + std::string second_half = s.substr(16, 16); + + char* end1; + char* end2; + uint64_t a = strtoull(first_half.c_str(), &end1, 16); + uint64_t b = strtoull(second_half.c_str(), &end2, 16); + + // Verify entire strings were parsed + if (end1 != first_half.c_str() + 16 || end2 != second_half.c_str() + 16) { throw operation_failed(); } - return UID(a.get(), b.get()); -} - -TEST_CASE("/flow/UID/parse") { - const UID id(0, std::numeric_limits::max()); - ASSERT(UID::fromString(id.toString()) == id); - ASSERT(UID::fromStringThrowsOnFailure(id.toString()) == id); - ASSERT(UID::fromString("0123456789ABCDEFfedcba9876543210") == UID(0x0123456789abcdef, 0xfedcba9876543210)); - std::string embeddedNul(32, '0'); - embeddedNul[15] = '\0'; - for (const std::string& invalid : - { std::string(31, '0'), std::string(32, 'g'), std::string(31, '0') + "g", embeddedNul }) { - try { - (void)UID::fromStringThrowsOnFailure(invalid); - ASSERT(false); - } catch (Error& e) { - ASSERT_EQ(e.code(), error_code_operation_failed); - } - } - return Void(); + return UID(a, b); } std::string UID::shortString() const { diff --git a/flow/include/flow/ParseNumber.h b/flow/include/flow/ParseNumber.h deleted file mode 100644 index f1a592db5c9..00000000000 --- a/flow/include/flow/ParseNumber.h +++ /dev/null @@ -1,92 +0,0 @@ -/* - * ParseNumber.h - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -#ifndef FLOW_PARSE_NUMBER_H -#define FLOW_PARSE_NUMBER_H -#pragma once - -#include "flow/Arena.h" - -#include -#include -#include -#include - -// Parses a numeric prefix without reading beyond input. Leading C whitespace and signs are accepted. -// Returns absent for a missing number or a conversion outside T's range. On success, consumed includes -// leading whitespace; on failure it is unchanged. Integer bases are 0 or 2..36. Floating-point input -// uses the C strto* grammar and requires base 10. Unsigned conversions retain strtoull's sign wrapping -// before checking T's range, so uint64_t accepts "-1" as UINT64_MAX. -template -Optional parseNumberPrefix(StringRef input, int base = 10, int* consumed = nullptr) { - static_assert((std::is_integral_v && !std::is_same_v) || std::is_floating_point_v); - if constexpr (std::is_integral_v) { - if (base != 0 && (base < 2 || base > 36)) { - return {}; - } - } else if (base != 10) { - return {}; - } - - const std::string text = input.toString(); - char* end = nullptr; - errno = 0; - T result; - if constexpr (std::is_same_v) { - result = std::strtof(text.c_str(), &end); - } else if constexpr (std::is_same_v) { - result = std::strtod(text.c_str(), &end); - } else if constexpr (std::is_same_v) { - result = std::strtold(text.c_str(), &end); - } else if constexpr (std::is_signed_v) { - const long long parsed = std::strtoll(text.c_str(), &end, base); - if (parsed < std::numeric_limits::min() || parsed > std::numeric_limits::max()) { - return {}; - } - result = static_cast(parsed); - } else { - const unsigned long long parsed = (std::strtoull)(text.c_str(), &end, base); - if (parsed > std::numeric_limits::max()) { - return {}; - } - result = static_cast(parsed); - } - if (errno == ERANGE || end == text.c_str()) { - return {}; - } - if (consumed != nullptr) { - *consumed = static_cast(end - text.c_str()); - } - return result; -} - -// Like parseNumberPrefix, but requires all input bytes to be part of the number. Trailing whitespace, -// other suffixes, and embedded NUL bytes are rejected. -template -Optional parseNumber(StringRef input, int base = 10) { - int consumed = 0; - auto result = parseNumberPrefix(input, base, &consumed); - if (!result.present() || consumed != input.size()) { - return {}; - } - return result; -} - -#endif diff --git a/flow/include/flow/UnitTest.h b/flow/include/flow/UnitTest.h index eac90a88778..ec2b1161bb0 100644 --- a/flow/include/flow/UnitTest.h +++ b/flow/include/flow/UnitTest.h @@ -77,12 +77,10 @@ class UnitTestParameters { // Get a parameter's value, will return !present() if parameter was not set Optional get(const std::string& name) const; - // Get a parameter's value as an integer, returning !present() if it was not set. - // Throws invalid_option_value if a present value is malformed or outside the int64_t range. + // Get a parameter's value as an integer, will return !present() if parameter was not set Optional getInt(const std::string& name) const; - // Get a parameter's value parsed as a double, returning !present() if it was not set. - // Throws invalid_option_value if a present value is malformed or outside the double range. + // Get a parameter's value parsed as a double, will return !present() if parameter was not set Optional getDouble(const std::string& name) const; // This is separate because it assumes data directory has already been set, and doesn't return an optional diff --git a/flow/network.cpp b/flow/network.cpp index 1ae299d30b7..38f097ca2fb 100644 --- a/flow/network.cpp +++ b/flow/network.cpp @@ -26,7 +26,6 @@ #include "flow/ChaosMetrics.h" #include "flow/UnitTest.h" #include "flow/IConnection.h" -#include "flow/ParseNumber.h" ChaosMetrics::ChaosMetrics() { clear(); @@ -190,24 +189,11 @@ NetworkAddress NetworkAddress::parse(std::string const& s) { } return NetworkAddress(addr.get(), port, true, isTLS, fromHostname); } else { - StringRef remaining(f); - uint32_t ip = 0; - for (int component = 0; component < 4; ++component) { - int consumed = 0; - auto octet = parseNumberPrefix(remaining, 10, &consumed); - const char separator = component == 3 ? ':' : '.'; - if (!octet.present() || octet.get() < 0 || octet.get() > 255 || consumed == remaining.size() || - remaining[consumed] != separator) { - throw connection_string_invalid(); - } - ip = (ip << 8) | static_cast(octet.get()); - remaining = remaining.substr(consumed + 1); - } - auto port = parseNumber(remaining); - if (!port.present() || port.get() < 0 || port.get() > std::numeric_limits::max()) { + // TODO: Use IPAddress::parse + int a, b, c, d, port, count = -1; + if (sscanf(f.c_str(), "%d.%d.%d.%d:%d%n", &a, &b, &c, &d, &port, &count) < 5 || count != f.size()) throw connection_string_invalid(); - } - return NetworkAddress(ip, port.get(), true, isTLS, fromHostname); + return NetworkAddress((a << 24) + (b << 16) + (c << 8) + d, port, true, isTLS, fromHostname); } } @@ -441,19 +427,6 @@ IUDPSocket::~IUDPSocket() = default; const std::vector NetworkMetrics::starvationBins = { 1, 3500, 7000, 7500, 8500, 8900, 10500 }; TEST_CASE("/flow/network/ipaddress") { - ASSERT(NetworkAddress::parse(" \t+127. +0.0. +1: +4800").toString() == "127.0.0.1:4800"); - ASSERT(NetworkAddress::parse("255.255.255.255:65535").toString() == "255.255.255.255:65535"); - for (const char* invalid : { "-1.0.0.1:4800", - "256.0.0.1:4800", - "9223372036854775808.0.0.1:4800", - "127.0.0.1:-1", - "127.0.0.1:65536", - "127.0.0.1:9223372036854775808", - "127.0.0.1:4800suffix", - "127.0.0.1:4800 " }) { - ASSERT(!NetworkAddress::parseOptional(invalid).present()); - } - ASSERT(NetworkAddress::parse("[::1]:4800").toString() == "[::1]:4800"); { From ad6444475a3c2939220b0d957a3e19270667ca1d Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Fri, 18 Sep 2026 14:50:21 -0700 Subject: [PATCH 116/170] Drop pointer-order check and clang-tidy tool upgrade --- .clang-tidy | 1 - .github/workflows/tidy.yml | 14 +--- bindings/flow/tester/Tester.cpp | 2 +- documentation/sphinx/source/clang-tidy.rst | 11 +--- fdbcli/HotRangeCommand.cpp | 2 +- fdbcli/SuspendCommand.cpp | 2 +- .../clustercontroller/ClusterController.cpp | 66 ++++++++++--------- fdbserver/core/MoveKeys.cpp | 2 +- fdbserver/kvstore/VersionedBTree.cpp | 2 - fdbserver/tester/TesterServer.cpp | 2 - fdbserver/worker/worker.cpp | 2 +- fdbserver/workloads/BulkDumping.cpp | 2 +- fdbserver/workloads/RangeLock.cpp | 4 +- fdbserver/workloads/SaveAndKill.cpp | 6 +- fdbserver/workloads/UnitTests.cpp | 2 - flow/UnitTestRunner.cpp | 2 - flow/include/flow/IDispatched.h | 6 +- 17 files changed, 52 insertions(+), 76 deletions(-) diff --git a/.clang-tidy b/.clang-tidy index 21a96f20d88..6d116142af0 100644 --- a/.clang-tidy +++ b/.clang-tidy @@ -15,7 +15,6 @@ Checks: > bugprone-macro-repeated-side-effects, bugprone-misplaced-widening-cast, bugprone-move-forwarding-reference, - bugprone-nondeterministic-pointer-iteration-order, bugprone-posix-return, bugprone-redundant-branch-condition, bugprone-return-const-ref-from-parameter, diff --git a/.github/workflows/tidy.yml b/.github/workflows/tidy.yml index 42befff6df5..eccacfd0a8b 100644 --- a/.github/workflows/tidy.yml +++ b/.github/workflows/tidy.yml @@ -55,18 +55,6 @@ jobs: ) ninja -v $PB_HEADERS - - name: Install clang-tidy with pointer-order checks - run: | - # The tool bundled in the build image is older than the pointer-order check. - # Keep its compiler and libc++ for the build; install only the packaged tidy tool. - dnf --disablerepo='*' --enablerepo=baseos,appstream \ - --setopt=baseos.mirrorlist= --setopt=appstream.mirrorlist= \ - --setopt=baseos.baseurl='https://dl.rockylinux.org/pub/rocky/9/BaseOS/$basearch/os/' \ - --setopt=appstream.baseurl='https://dl.rockylinux.org/pub/rocky/9/AppStream/$basearch/os/' \ - --setopt=timeout=15 --setopt=retries=1 install -y clang-tools-extra - /usr/bin/clang-tidy --version - /usr/bin/clang-tidy --list-checks | grep -q bugprone-nondeterministic-pointer-iteration-order - - name: clang-tidy # clang-tidy is roughly as slow as compiling per-file, so check only touched files shell: bash @@ -121,7 +109,7 @@ jobs: if [[ $FILE == fdbserver/*/*RocksDB*.cpp ]]; then ninja -C build_output fdbserver/rocksdb-prefix/src/rocksdb-stamp/rocksdb-download fi - if ! /usr/bin/clang-tidy -p build_output --warnings-as-errors='*' "$FILE"; then + if ! clang-tidy -p build_output --warnings-as-errors='*' "$FILE"; then PASSED=false fi echo diff --git a/bindings/flow/tester/Tester.cpp b/bindings/flow/tester/Tester.cpp index f9678cf2ca8..f7788645e8d 100644 --- a/bindings/flow/tester/Tester.cpp +++ b/bindings/flow/tester/Tester.cpp @@ -1514,7 +1514,7 @@ struct AtomicOPFunc : InstructionFunc { Standalone s3 = co_await items[2].value; Standalone value = Tuple::unpack(s3).getString(0); - ASSERT(optionInfo.contains(op.toString())); + ASSERT(optionInfo.find(op.toString()) != optionInfo.end()); FDBMutationType atomicOp = optionInfo[op.toString()]; diff --git a/documentation/sphinx/source/clang-tidy.rst b/documentation/sphinx/source/clang-tidy.rst index de952516f74..d1ee69557a9 100644 --- a/documentation/sphinx/source/clang-tidy.rst +++ b/documentation/sphinx/source/clang-tidy.rst @@ -10,11 +10,11 @@ This guide explains how to run ``clang-tidy`` locally so you can fix issues befo What clang-tidy checks ====================== -FoundationDB configures 58 named checks in the ``.clang-tidy`` file at the repository root. The +FoundationDB configures 57 named checks in the ``.clang-tidy`` file at the repository root. The active set depends on the clang-tidy version and can be inspected with ``clang-tidy --list-checks``. The intent is to enable more as we go forward. Here are some example rules: -* **38 Bugprone rules** -- catch potential runtime errors, including unsafe self-assignment, forwarding constructors that hide copy or move constructors, narrow accumulation initializers, mismatched argument comments, obvious infinite loops, chained comparisons, swapped arguments, integer division in floating-point calculations, missed base-class copy construction, repeated macro argument evaluation, near-miss virtual overrides, dangling returned references, incorrect erase/remove calls, ignored return values, incorrect POSIX error checks, and pointer-dependent iteration order +* **37 Bugprone rules** -- catch potential runtime errors, including unsafe self-assignment, forwarding constructors that hide copy or move constructors, narrow accumulation initializers, mismatched argument comments, obvious infinite loops, chained comparisons, swapped arguments, integer division in floating-point calculations, missed base-class copy construction, repeated macro argument evaluation, near-miss virtual overrides, dangling returned references, incorrect erase/remove calls, ignored return values, and incorrect POSIX error checks * **2 C++ Core Guidelines rules** -- catch unsafe captures in coroutine lambdas and borrowed coroutine parameters (``cppcoreguidelines-avoid-capturing-lambda-coroutines`` and ``cppcoreguidelines-avoid-reference-coroutine-parameters``) * **2 Misc rules** -- catch redundant expressions and RAII objects held across coroutine suspension points * **4 Modernize rules** -- encourage modern C++ practices (e.g., ``modernize-use-auto``, ``modernize-use-override``) @@ -40,13 +40,6 @@ arguments in coroutine frames. A synchronous forwarding wrapper can preserve an interface that accepts references while its coroutine implementation takes values. Copying a ``StringRef`` or ``KeyRef`` still does not retain the referenced bytes. -``bugprone-nondeterministic-pointer-iteration-order`` helps protect simulation -reproducibility. It checks some pointer-keyed unordered-container iterations and -pointer sorting; it is not a complete determinism analysis. This check requires -LLVM 20 or newer. CI installs the packaged ``clang-tidy`` separately from the -build image's compiler so that the warning is active without changing the compiler -or C++ standard library. - Basic examples of ``clang-tidy`` style and performance improvement changes: .. code-block:: cpp diff --git a/fdbcli/HotRangeCommand.cpp b/fdbcli/HotRangeCommand.cpp index 3ea6e3eac3f..a48b616aa96 100644 --- a/fdbcli/HotRangeCommand.cpp +++ b/fdbcli/HotRangeCommand.cpp @@ -74,7 +74,7 @@ Future hotRangeCommandActor(Database localdb, } Key address = tokens[1]; // At present we only support one process(IP:Port) at a time - if (!storage_interface->contains(address.toString())) { + if (!storage_interface->count(address.toString())) { fprintf(stderr, "ERROR: storage process `%s' not recognized.\n", printable(address).c_str()); co_return false; } diff --git a/fdbcli/SuspendCommand.cpp b/fdbcli/SuspendCommand.cpp index 8bbfad5f1e3..958ccaf2b88 100644 --- a/fdbcli/SuspendCommand.cpp +++ b/fdbcli/SuspendCommand.cpp @@ -60,7 +60,7 @@ Future suspendCommandActor(Reference db, result = false; } else { for (int i = 2; i < tokens.size(); i++) { - if (!address_interface->contains(tokens[i])) { + if (!address_interface->count(tokens[i])) { fprintf(stderr, "ERROR: process `%s' not recognized.\n", printable(tokens[i]).c_str()); result = false; break; diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index bfd2be72bb5..fb8bd4a64bf 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -3255,8 +3255,10 @@ Future workerHealthMonitor(ClusterControllerData* self) { // recovered. bool hasRecoveredServer = false; for (auto it = self->excludedDegradedServers.begin(); it != self->excludedDegradedServers.end();) { - if (!self->degradationInfo.degradedServers.contains(it->first) && - !self->degradationInfo.disconnectedServers.contains(it->first)) { + if (self->degradationInfo.degradedServers.find(it->first) == + self->degradationInfo.degradedServers.end() && + self->degradationInfo.disconnectedServers.find(it->first) == + self->degradationInfo.disconnectedServers.end()) { self->excludedDegradedServers.erase(it++); hasRecoveredServer = true; } else { @@ -4453,17 +4455,17 @@ TEST_CASE("/fdbserver/clustercontroller/updateWorkerHealth") { req.disconnectedPeers.push_back(badPeer1); req.disconnectedPeers.push_back(badPeer2); data.updateWorkerHealth(req); - ASSERT(data.workerHealth.contains(workerAddress)); + ASSERT(data.workerHealth.find(workerAddress) != data.workerHealth.end()); auto& health = data.workerHealth[workerAddress]; ASSERT_EQ(health.degradedPeers.size(), 2); - ASSERT(health.degradedPeers.contains(badPeer1)); + ASSERT(health.degradedPeers.find(badPeer1) != health.degradedPeers.end()); ASSERT_EQ(health.degradedPeers[badPeer1].startTime, health.degradedPeers[badPeer1].lastRefreshTime); - ASSERT(health.degradedPeers.contains(badPeer2)); + ASSERT(health.degradedPeers.find(badPeer2) != health.degradedPeers.end()); ASSERT_EQ(health.degradedPeers[badPeer2].startTime, health.degradedPeers[badPeer2].lastRefreshTime); ASSERT_EQ(health.disconnectedPeers.size(), 2); - ASSERT(health.disconnectedPeers.contains(badPeer1)); + ASSERT(health.disconnectedPeers.find(badPeer1) != health.disconnectedPeers.end()); ASSERT_EQ(health.disconnectedPeers[badPeer1].startTime, health.disconnectedPeers[badPeer1].lastRefreshTime); - ASSERT(health.disconnectedPeers.contains(badPeer2)); + ASSERT(health.disconnectedPeers.find(badPeer2) != health.disconnectedPeers.end()); ASSERT_EQ(health.disconnectedPeers[badPeer2].startTime, health.disconnectedPeers[badPeer2].lastRefreshTime); } @@ -4482,23 +4484,23 @@ TEST_CASE("/fdbserver/clustercontroller/updateWorkerHealth") { req.disconnectedPeers.push_back(badPeer1); req.disconnectedPeers.push_back(badPeer3); data.updateWorkerHealth(req); - ASSERT(data.workerHealth.contains(workerAddress)); + ASSERT(data.workerHealth.find(workerAddress) != data.workerHealth.end()); auto& health = data.workerHealth[workerAddress]; ASSERT_EQ(health.degradedPeers.size(), 3); - ASSERT(health.degradedPeers.contains(badPeer1)); + ASSERT(health.degradedPeers.find(badPeer1) != health.degradedPeers.end()); ASSERT_LT(health.degradedPeers[badPeer1].startTime, health.degradedPeers[badPeer1].lastRefreshTime); - ASSERT(health.degradedPeers.contains(badPeer2)); + ASSERT(health.degradedPeers.find(badPeer2) != health.degradedPeers.end()); ASSERT_EQ(health.degradedPeers[badPeer2].startTime, health.degradedPeers[badPeer2].lastRefreshTime); ASSERT_EQ(health.degradedPeers[badPeer2].startTime, health.degradedPeers[badPeer1].startTime); - ASSERT(health.degradedPeers.contains(badPeer3)); + ASSERT(health.degradedPeers.find(badPeer3) != health.degradedPeers.end()); ASSERT_EQ(health.degradedPeers[badPeer3].startTime, health.degradedPeers[badPeer3].lastRefreshTime); ASSERT_EQ(health.disconnectedPeers.size(), 3); - ASSERT(health.disconnectedPeers.contains(badPeer1)); + ASSERT(health.disconnectedPeers.find(badPeer1) != health.disconnectedPeers.end()); ASSERT_LT(health.disconnectedPeers[badPeer1].startTime, health.disconnectedPeers[badPeer1].lastRefreshTime); - ASSERT(health.disconnectedPeers.contains(badPeer2)); + ASSERT(health.disconnectedPeers.find(badPeer2) != health.disconnectedPeers.end()); ASSERT_EQ(health.disconnectedPeers[badPeer2].startTime, health.disconnectedPeers[badPeer2].lastRefreshTime); ASSERT_EQ(health.disconnectedPeers[badPeer2].startTime, health.disconnectedPeers[badPeer1].startTime); - ASSERT(health.disconnectedPeers.contains(badPeer3)); + ASSERT(health.disconnectedPeers.find(badPeer3) != health.disconnectedPeers.end()); ASSERT_EQ(health.disconnectedPeers[badPeer3].startTime, health.disconnectedPeers[badPeer3].lastRefreshTime); previousStartTime = health.degradedPeers[badPeer3].startTime; @@ -4512,14 +4514,14 @@ TEST_CASE("/fdbserver/clustercontroller/updateWorkerHealth") { UpdateWorkerHealthRequest req; req.address = workerAddress; data.updateWorkerHealth(req); - ASSERT(data.workerHealth.contains(workerAddress)); + ASSERT(data.workerHealth.find(workerAddress) != data.workerHealth.end()); auto& health = data.workerHealth[workerAddress]; ASSERT_EQ(health.degradedPeers.size(), 3); - ASSERT(health.degradedPeers.contains(badPeer3)); + ASSERT(health.degradedPeers.find(badPeer3) != health.degradedPeers.end()); ASSERT_EQ(health.degradedPeers[badPeer3].startTime, previousStartTime); ASSERT_EQ(health.degradedPeers[badPeer3].lastRefreshTime, previousRefreshTime); ASSERT_EQ(health.disconnectedPeers.size(), 3); - ASSERT(health.disconnectedPeers.contains(badPeer3)); + ASSERT(health.disconnectedPeers.find(badPeer3) != health.disconnectedPeers.end()); ASSERT_EQ(health.disconnectedPeers[badPeer3].startTime, previousStartTime); ASSERT_EQ(health.disconnectedPeers[badPeer3].lastRefreshTime, previousRefreshTime); } @@ -4532,8 +4534,8 @@ TEST_CASE("/fdbserver/clustercontroller/updateWorkerHealth") { req.recoveredPeers.push_back(badPeer1); data.updateWorkerHealth(req); auto& health = data.workerHealth[workerAddress]; - ASSERT(!health.degradedPeers.contains(badPeer1)); - ASSERT(!health.disconnectedPeers.contains(badPeer1)); + ASSERT(health.degradedPeers.find(badPeer1) == health.degradedPeers.end()); + ASSERT(health.disconnectedPeers.find(badPeer1) == health.disconnectedPeers.end()); } } @@ -4577,11 +4579,12 @@ TEST_CASE("/fdbserver/clustercontroller/updateRecoveredWorkers") { data.updateRecoveredWorkers(); ASSERT_EQ(data.workerHealth.size(), 1); - ASSERT(data.workerHealth.contains(worker1)); - ASSERT(data.workerHealth[worker1].degradedPeers.contains(badPeer1)); - ASSERT(!data.workerHealth[worker1].degradedPeers.contains(badPeer2)); - ASSERT(data.workerHealth[worker1].degradedPeers.contains(disconnectedPeer3)); - ASSERT(!data.workerHealth.contains(worker2)); + ASSERT(data.workerHealth.find(worker1) != data.workerHealth.end()); + ASSERT(data.workerHealth[worker1].degradedPeers.find(badPeer1) != data.workerHealth[worker1].degradedPeers.end()); + ASSERT(data.workerHealth[worker1].degradedPeers.find(badPeer2) == data.workerHealth[worker1].degradedPeers.end()); + ASSERT(data.workerHealth[worker1].degradedPeers.find(disconnectedPeer3) != + data.workerHealth[worker1].degradedPeers.end()); + ASSERT(data.workerHealth.find(worker2) == data.workerHealth.end()); return Void(); } @@ -4619,7 +4622,7 @@ TEST_CASE("/fdbserver/clustercontroller/getDegradationInfo") { now() }; auto degradationInfo = data.getDegradationInfo(); ASSERT(degradationInfo.degradedServers.size() == 1); - ASSERT(degradationInfo.degradedServers.contains(badPeer1)); + ASSERT(degradationInfo.degradedServers.find(badPeer1) != degradationInfo.degradedServers.end()); ASSERT(degradationInfo.disconnectedServers.empty()); data.workerHealth.clear(); } @@ -4631,7 +4634,7 @@ TEST_CASE("/fdbserver/clustercontroller/getDegradationInfo") { now() }; auto degradationInfo = data.getDegradationInfo(); ASSERT(degradationInfo.disconnectedServers.size() == 1); - ASSERT(degradationInfo.disconnectedServers.contains(badPeer1)); + ASSERT(degradationInfo.disconnectedServers.find(badPeer1) != degradationInfo.disconnectedServers.end()); ASSERT(degradationInfo.degradedServers.empty()); data.workerHealth.clear(); } @@ -4649,10 +4652,11 @@ TEST_CASE("/fdbserver/clustercontroller/getDegradationInfo") { now() }; auto degradationInfo = data.getDegradationInfo(); ASSERT(degradationInfo.degradedServers.size() == 1); - ASSERT(degradationInfo.degradedServers.contains(worker) || degradationInfo.degradedServers.contains(badPeer1)); + ASSERT(degradationInfo.degradedServers.find(worker) != degradationInfo.degradedServers.end() || + degradationInfo.degradedServers.find(badPeer1) != degradationInfo.degradedServers.end()); ASSERT(degradationInfo.disconnectedServers.size() == 1); - ASSERT(degradationInfo.disconnectedServers.contains(worker) || - degradationInfo.disconnectedServers.contains(badPeer2)); + ASSERT(degradationInfo.disconnectedServers.find(worker) != degradationInfo.disconnectedServers.end() || + degradationInfo.disconnectedServers.find(badPeer2) != degradationInfo.disconnectedServers.end()); data.workerHealth.clear(); } @@ -4679,9 +4683,9 @@ TEST_CASE("/fdbserver/clustercontroller/getDegradationInfo") { now() }; auto degradationInfo = data.getDegradationInfo(); ASSERT(degradationInfo.degradedServers.size() == 1); - ASSERT(degradationInfo.degradedServers.contains(worker)); + ASSERT(degradationInfo.degradedServers.find(worker) != degradationInfo.degradedServers.end()); ASSERT(degradationInfo.disconnectedServers.size() == 1); - ASSERT(degradationInfo.disconnectedServers.contains(worker)); + ASSERT(degradationInfo.disconnectedServers.find(worker) != degradationInfo.disconnectedServers.end()); data.workerHealth.clear(); } diff --git a/fdbserver/core/MoveKeys.cpp b/fdbserver/core/MoveKeys.cpp index 71746314e20..d13da293091 100644 --- a/fdbserver/core/MoveKeys.cpp +++ b/fdbserver/core/MoveKeys.cpp @@ -257,7 +257,7 @@ Future deleteCheckpoints(Transaction* tr, std::set checkpointIds, UID continue; } CheckpointMetaData checkpoint = decodeCheckpointValue(value.get()); - ASSERT(checkpointIds.contains(checkpoint.checkpointID)); + ASSERT(checkpointIds.find(checkpoint.checkpointID) != checkpointIds.end()); const Key key = checkpointKeyFor(checkpoint.checkpointID); checkpoint.setState(CheckpointMetaData::Deleting); tr->set(key, checkpointValue(checkpoint)); diff --git a/fdbserver/kvstore/VersionedBTree.cpp b/fdbserver/kvstore/VersionedBTree.cpp index 178035270af..1a819a732d3 100644 --- a/fdbserver/kvstore/VersionedBTree.cpp +++ b/fdbserver/kvstore/VersionedBTree.cpp @@ -11201,8 +11201,6 @@ struct KVSource { for (auto& p : prefixes) { prefixesSorted.push_back(&p); } - // The comparator orders by prefix bytes, independent of pointer values. - // NOLINTNEXTLINE(bugprone-nondeterministic-pointer-iteration-order) std::sort(prefixesSorted.begin(), prefixesSorted.end(), [](const Prefix* a, const Prefix* b) { return KeyRef((uint8_t*)a->begin(), a->size()) < KeyRef((uint8_t*)b->begin(), b->size()); }); diff --git a/fdbserver/tester/TesterServer.cpp b/fdbserver/tester/TesterServer.cpp index 52ea491c6c0..1e2b9409e9f 100644 --- a/fdbserver/tester/TesterServer.cpp +++ b/fdbserver/tester/TesterServer.cpp @@ -141,8 +141,6 @@ void printSimulatedTopology() { return; } auto processes = g_simulator->getAllProcesses(); - // The comparator orders by locality and network address, independent of pointer values. - // NOLINTNEXTLINE(bugprone-nondeterministic-pointer-iteration-order) std::sort(processes.begin(), processes.end(), [](ISimulator::ProcessInfo* lhs, ISimulator::ProcessInfo* rhs) { auto l = lhs->locality; auto r = rhs->locality; diff --git a/fdbserver/worker/worker.cpp b/fdbserver/worker/worker.cpp index f47596402e9..e2aa4a3420e 100644 --- a/fdbserver/worker/worker.cpp +++ b/fdbserver/worker/worker.cpp @@ -1237,7 +1237,7 @@ UpdateWorkerHealthRequest doPeerHealthCheck(const WorkerInterface& interf, // Note that we don't need to calculate recovered peer in this case since all the recently closed peers are // considered permanently closed peers. for (const auto& address : FlowTransport::transport().getRecentClosedPeers()) { - if (allPeers.contains(address)) { + if (allPeers.find(address) != allPeers.end()) { // We have checked this peer in the above for loop. continue; } diff --git a/fdbserver/workloads/BulkDumping.cpp b/fdbserver/workloads/BulkDumping.cpp index 57ffe588e15..05a298028a8 100644 --- a/fdbserver/workloads/BulkDumping.cpp +++ b/fdbserver/workloads/BulkDumping.cpp @@ -436,7 +436,7 @@ struct BulkDumping : TestWorkload { } for (const auto& [key, value] : newKvs) { // newKvs should not contain keys outside the bulkDumpJobRange - ASSERT(!keyOutsideDumpData.contains(key) && bulkDumpJobRange.contains(key)); + ASSERT(keyOutsideDumpData.find(key) == keyOutsideDumpData.end() && bulkDumpJobRange.contains(key)); if (self->keyContainedInRanges(key, ignoreRanges)) { continue; } diff --git a/fdbserver/workloads/RangeLock.cpp b/fdbserver/workloads/RangeLock.cpp index e27af4827ca..24a43173b45 100644 --- a/fdbserver/workloads/RangeLock.cpp +++ b/fdbserver/workloads/RangeLock.cpp @@ -578,7 +578,7 @@ struct RangeLocking : TestWorkload { std::map currentKvsInDB; currentKvsInDB = co_await self->getKVSFromDB(self, cx); for (const auto& [key, value] : currentKvsInDB) { - if (!self->kvs.contains(key)) { + if (self->kvs.find(key) == self->kvs.end()) { TraceEvent(SevError, "RangeLockWorkLoadHistory") .detail("Ops", "CheckDBUniqueKey") .detail("Key", key) @@ -596,7 +596,7 @@ struct RangeLocking : TestWorkload { } } for (const auto& [key, value] : self->kvs) { - if (!currentKvsInDB.contains(key)) { + if (currentKvsInDB.find(key) == currentKvsInDB.end()) { TraceEvent(SevError, "RangeLockWorkLoadHistory") .detail("Ops", "CheckMemoryUniqueKey") .detail("Key", key) diff --git a/fdbserver/workloads/SaveAndKill.cpp b/fdbserver/workloads/SaveAndKill.cpp index f1148344f93..e26f4fb44b9 100644 --- a/fdbserver/workloads/SaveAndKill.cpp +++ b/fdbserver/workloads/SaveAndKill.cpp @@ -84,12 +84,12 @@ struct SaveAndKillWorkload : TestWorkload { g_simulator->currentlyRebootingProcesses; std::map allProcessesMap; for (const auto& [_, process] : rebootingProcesses) { - if (!allProcessesMap.contains(process->dataFolder) && !process->isSpawnedKVProcess()) { + if (allProcessesMap.find(process->dataFolder) == allProcessesMap.end() && !process->isSpawnedKVProcess()) { allProcessesMap[process->dataFolder] = process; } } for (const auto& process : processes) { - if (!allProcessesMap.contains(process->dataFolder) && !process->isSpawnedKVProcess()) { + if (allProcessesMap.find(process->dataFolder) == allProcessesMap.end() && !process->isSpawnedKVProcess()) { allProcessesMap[process->dataFolder] = process; } } @@ -101,7 +101,7 @@ struct SaveAndKillWorkload : TestWorkload { std::string machineId = printable(process->locality.machineId()); const char* machineIdString = machineId.c_str(); if (!process->excludeFromRestarts) { - if (!machines.contains(machineId)) { + if (machines.find(machineId) == machines.end()) { machines.insert(std::pair(machineId, 1)); ini.SetValue("META", format("%d", j).c_str(), machineIdString); ini.SetValue( diff --git a/fdbserver/workloads/UnitTests.cpp b/fdbserver/workloads/UnitTests.cpp index 5d5ebb7bdb9..1b3a90ee4fa 100644 --- a/fdbserver/workloads/UnitTests.cpp +++ b/fdbserver/workloads/UnitTests.cpp @@ -188,8 +188,6 @@ struct UnitTestWorkload : TestWorkload { } } - // The comparator orders by test names, independent of pointer values. - // NOLINTNEXTLINE(bugprone-nondeterministic-pointer-iteration-order) std::sort(tests.begin(), tests.end(), [](auto lhs, auto rhs) { return std::string_view(lhs->name) < std::string_view(rhs->name); }); diff --git a/flow/UnitTestRunner.cpp b/flow/UnitTestRunner.cpp index 42448812780..73fa86fc50a 100644 --- a/flow/UnitTestRunner.cpp +++ b/flow/UnitTestRunner.cpp @@ -284,8 +284,6 @@ std::vector collectTests(const UnitTestRunnerOptions& options, const } } - // The comparator orders tests by their names, independent of their addresses. - // NOLINTNEXTLINE(bugprone-nondeterministic-pointer-iteration-order) std::sort(tests.begin(), tests.end(), [](auto lhs, auto rhs) { return std::string_view(lhs->name) < std::string_view(rhs->name); }); diff --git a/flow/include/flow/IDispatched.h b/flow/include/flow/IDispatched.h index 9ea9263c9b5..2ca06bdf26e 100644 --- a/flow/include/flow/IDispatched.h +++ b/flow/include/flow/IDispatched.h @@ -44,7 +44,7 @@ struct IDispatched { #define REGISTER_DISPATCHED(Type, Instance, Key, Func) \ struct Type##Instance { \ Type##Instance() { \ - ASSERT(!Type::dispatches().contains(Key)); \ + ASSERT(Type::dispatches().find(Key) == Type::dispatches().end()); \ Type::dispatches()[Key] = Func; \ } \ }; \ @@ -52,8 +52,8 @@ struct IDispatched { #define REGISTER_DISPATCHED_ALIAS(Type, Instance, Target, Alias) \ struct Type##Instance { \ Type##Instance() { \ - ASSERT(!Type::dispatches().contains(Alias)); \ - ASSERT(Type::dispatches().contains(Target)); \ + ASSERT(Type::dispatches().find(Alias) == Type::dispatches().end()); \ + ASSERT(Type::dispatches().find(Target) != Type::dispatches().end()); \ Type::dispatches()[Alias] = Type::dispatches()[Target]; \ } \ }; \ From e297e8801824557736c4da5b9175af8a1c760db2 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Fri, 18 Sep 2026 15:07:09 -0700 Subject: [PATCH 117/170] Drop coroutine-reference clang-tidy rollout --- .clang-tidy | 1 - bindings/flow/tester/Tester.cpp | 10 ++- documentation/sphinx/source/clang-tidy.rst | 9 +- fdbcli/ConsistencyCheckCommand.cpp | 2 - fdbcli/ConsistencyScanCommand.cpp | 2 - fdbcli/FileConfigureCommand.cpp | 2 - fdbcli/ForceRecoveryWithDataLossCommand.cpp | 2 - fdbcli/HotRangeCommand.cpp | 4 +- fdbcli/IdempotencyIdsCommand.cpp | 2 - fdbcli/LockCommand.cpp | 2 - fdbcli/SnapshotCommand.cpp | 2 - fdbcli/SuspendCommand.cpp | 2 - fdbcli/include/fdbcli/fdbcli.h | 2 +- fdbclient/S3BlobStore.cpp | 16 ++-- fdbclient/include/fdbclient/IBlobStore.h | 2 +- .../RangePartitionedBackupWorker.cpp | 6 -- fdbserver/cdcproxy/CDCProxy.cpp | 22 ++--- .../clustercontroller/ClusterController.cpp | 5 -- .../ClusterHealthIFactor.cpp | 4 +- .../ClusterHealthMonitor.cpp | 4 - .../ClusterHealthMonitorTesting.cpp | 10 --- fdbserver/core/MoveKeys.cpp | 84 ++++++------------- fdbserver/datadistributor/DDTxnProcessor.cpp | 2 - fdbserver/networktest.cpp | 2 - .../tester/CustomShardConfigWorkload.cpp | 7 +- fdbserver/tester/DatabaseMaintenance.cpp | 6 +- fdbserver/tester/TesterServer.cpp | 28 ++++--- fdbserver/tester/test.cpp | 68 +++++++-------- fdbserver/worker/worker.cpp | 2 +- fdbserver/workloads/BackgroundSelectors.cpp | 4 +- fdbserver/workloads/BackupToDBAbort.cpp | 4 +- fdbserver/workloads/BulkDumping.cpp | 4 +- fdbserver/workloads/BulkLoading.cpp | 4 +- fdbserver/workloads/ConfigureDatabase.cpp | 12 +-- fdbserver/workloads/ConflictRange.cpp | 4 +- fdbserver/workloads/CpuProfiler.cpp | 8 +- fdbserver/workloads/DDBalance.cpp | 4 +- fdbserver/workloads/DDMetricsExclude.cpp | 4 +- .../workloads/DataDistributionMetrics.cpp | 4 +- .../workloads/DifferentClustersSameRV.cpp | 11 ++- fdbserver/workloads/FileSystem.cpp | 4 +- fdbserver/workloads/HealthMetricsApi.cpp | 10 +-- .../HighContentionPrefixAllocatorWorkload.cpp | 4 +- fdbserver/workloads/MemoryLifetime.cpp | 8 +- fdbserver/workloads/MetricLogging.cpp | 4 +- fdbserver/workloads/ProtocolVersion.cpp | 4 +- fdbserver/workloads/QueuePush.cpp | 4 +- fdbserver/workloads/RandomMoveKeys.cpp | 4 +- fdbserver/workloads/RandomRangeLock.cpp | 4 +- fdbserver/workloads/RangeLock.cpp | 4 +- fdbserver/workloads/ReadAfterWrite.cpp | 4 +- fdbserver/workloads/ReadHotDetection.cpp | 4 +- fdbserver/workloads/ReadWrite.cpp | 2 - fdbserver/workloads/S3ClientWorkload.cpp | 4 +- fdbserver/workloads/SaveAndKill.cpp | 4 +- fdbserver/workloads/SkewedReadWrite.cpp | 4 +- .../workloads/SpecialKeySpaceCorrectness.cpp | 4 +- .../workloads/SpecialKeySpaceRobustness.cpp | 4 +- fdbserver/workloads/StreamingRangeRead.cpp | 4 +- fdbserver/workloads/TaskBucketCorrectness.cpp | 8 +- fdbserver/workloads/ThreadSafety.cpp | 4 +- fdbserver/workloads/Throttling.cpp | 4 +- fdbserver/workloads/TimeKeeperCorrectness.cpp | 8 +- fdbserver/workloads/TransactionCost.cpp | 8 -- fdbserver/workloads/WatchAndWait.cpp | 4 +- fdbserver/workloads/Watches.cpp | 4 +- fdbserver/workloads/WorkerErrors.cpp | 4 +- fdbserver/workloads/WriteBandwidth.cpp | 8 +- flow/CoroTests.cpp | 16 ---- flow/IThreadPoolTest.cpp | 2 +- flow/Net2.cpp | 18 ++-- flow/UnitTest.cpp | 25 ------ flow/UnitTestRunner.cpp | 10 ++- flow/bench/BenchAsyncResult.cpp | 4 - flow/include/flow/UnitTest.h | 12 +-- 75 files changed, 174 insertions(+), 422 deletions(-) diff --git a/.clang-tidy b/.clang-tidy index 6d116142af0..1ee007891b7 100644 --- a/.clang-tidy +++ b/.clang-tidy @@ -39,7 +39,6 @@ Checks: > bugprone-use-after-move, bugprone-virtual-near-miss, cppcoreguidelines-avoid-capturing-lambda-coroutines, - cppcoreguidelines-avoid-reference-coroutine-parameters, misc-coroutine-hostile-raii, misc-redundant-expression, modernize-use-auto, diff --git a/bindings/flow/tester/Tester.cpp b/bindings/flow/tester/Tester.cpp index f7788645e8d..331f4f3fc09 100644 --- a/bindings/flow/tester/Tester.cpp +++ b/bindings/flow/tester/Tester.cpp @@ -43,7 +43,9 @@ std::map, Reference> trMap; const int ITERATION_PROGRESSION[] = { 256, 1000, 4096, 6144, 9216, 13824, 20736, 31104, 46656, 69984, 80000 }; const int MAX_ITERATION = sizeof(ITERATION_PROGRESSION) / sizeof(int); -static Future runTest(Reference data, Reference db, Standalone prefix); +static Future runTest(Reference const& data, + Reference const& db, + StringRef const& prefix); THREAD_FUNC networkThread(void* api) { // This is the fdb_flow network we're running on a thread @@ -1567,7 +1569,7 @@ struct UnitTestsFunc : InstructionFunc { const uint64_t locationCacheSize = 100001; const uint64_t maxWatches = 10001; - const uint64_t timeout = 60ULL * 1000; + const uint64_t timeout = 60 * 1000; const uint64_t noTimeout = 0; const uint64_t retryLimit = 50; const uint64_t noRetryLimit = -1; @@ -1715,7 +1717,9 @@ static Future doInstructions(Reference data) { // printf("Total num instructions:%d\n", data->instructions.size()); } -static Future runTest(Reference data, Reference db, Standalone prefix) { +static Future runTest(Reference const& data, + Reference const& db, + StringRef const& prefix) { ASSERT(data); try { data->db = db; diff --git a/documentation/sphinx/source/clang-tidy.rst b/documentation/sphinx/source/clang-tidy.rst index d1ee69557a9..162fbec994a 100644 --- a/documentation/sphinx/source/clang-tidy.rst +++ b/documentation/sphinx/source/clang-tidy.rst @@ -10,12 +10,12 @@ This guide explains how to run ``clang-tidy`` locally so you can fix issues befo What clang-tidy checks ====================== -FoundationDB configures 57 named checks in the ``.clang-tidy`` file at the repository root. The +FoundationDB configures 56 named checks in the ``.clang-tidy`` file at the repository root. The active set depends on the clang-tidy version and can be inspected with ``clang-tidy --list-checks``. The intent is to enable more as we go forward. Here are some example rules: * **37 Bugprone rules** -- catch potential runtime errors, including unsafe self-assignment, forwarding constructors that hide copy or move constructors, narrow accumulation initializers, mismatched argument comments, obvious infinite loops, chained comparisons, swapped arguments, integer division in floating-point calculations, missed base-class copy construction, repeated macro argument evaluation, near-miss virtual overrides, dangling returned references, incorrect erase/remove calls, ignored return values, and incorrect POSIX error checks -* **2 C++ Core Guidelines rules** -- catch unsafe captures in coroutine lambdas and borrowed coroutine parameters (``cppcoreguidelines-avoid-capturing-lambda-coroutines`` and ``cppcoreguidelines-avoid-reference-coroutine-parameters``) +* **1 C++ Core Guidelines rule** -- catch unsafe captures in coroutine lambdas (``cppcoreguidelines-avoid-capturing-lambda-coroutines``) * **2 Misc rules** -- catch redundant expressions and RAII objects held across coroutine suspension points * **4 Modernize rules** -- encourage modern C++ practices (e.g., ``modernize-use-auto``, ``modernize-use-override``) * **5 Performance rules** -- avoid unnecessary copies, hidden range-loop conversions, repeated vector growth in simple loops, pointless moves, and move constructors that copy movable members (``performance-for-range-copy``, ``performance-implicit-conversion-in-loop``, ``performance-inefficient-vector-operation``, ``performance-move-const-arg``, ``performance-move-constructor-init``) @@ -35,11 +35,6 @@ reference before releasing the old one use documented, check-specific also checks ``pthread_create``. It does not diagnose every discarded Flow future: intentional uncancellable helpers need different treatment from cancellable work. -``cppcoreguidelines-avoid-reference-coroutine-parameters`` encourages owning -arguments in coroutine frames. A synchronous forwarding wrapper can preserve an -interface that accepts references while its coroutine implementation takes values. -Copying a ``StringRef`` or ``KeyRef`` still does not retain the referenced bytes. - Basic examples of ``clang-tidy`` style and performance improvement changes: .. code-block:: cpp diff --git a/fdbcli/ConsistencyCheckCommand.cpp b/fdbcli/ConsistencyCheckCommand.cpp index d9e510e2191..213bc71a63c 100644 --- a/fdbcli/ConsistencyCheckCommand.cpp +++ b/fdbcli/ConsistencyCheckCommand.cpp @@ -30,9 +30,7 @@ namespace fdb_cli { const KeyRef consistencyCheckSpecialKey = "\xff\xff/management/consistency_check_suspended"_sr; -// The CLI retains the tokens and their backing line until this command finishes or is cancelled. Future consistencyCheckCommandActor(Reference tr, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::vector const& tokens, bool intrans) { // Here we do not proceed in a try-catch loop since the transaction is always supposed to succeed. diff --git a/fdbcli/ConsistencyScanCommand.cpp b/fdbcli/ConsistencyScanCommand.cpp index 4d8a4b9db97..0d56b7a0c35 100644 --- a/fdbcli/ConsistencyScanCommand.cpp +++ b/fdbcli/ConsistencyScanCommand.cpp @@ -41,8 +41,6 @@ Future dumpStats(ConsistencyScanState* cs, Reference consistencyScanCommandActor(Database db, std::vector const& tokens) { // Skip the command token so start at begin+1 std::list args(tokens.begin() + 1, tokens.end()); diff --git a/fdbcli/FileConfigureCommand.cpp b/fdbcli/FileConfigureCommand.cpp index 9f405980116..6cf827947b3 100644 --- a/fdbcli/FileConfigureCommand.cpp +++ b/fdbcli/FileConfigureCommand.cpp @@ -33,9 +33,7 @@ #include "flow/ThreadHelper.h" namespace fdb_cli { -// The configuration file is read into owned storage before suspension. Future fileConfigureCommandActor(Reference db, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::string const& filePath, bool isNewDatabase, bool force) { diff --git a/fdbcli/ForceRecoveryWithDataLossCommand.cpp b/fdbcli/ForceRecoveryWithDataLossCommand.cpp index f74fb6671f6..cca286dc795 100644 --- a/fdbcli/ForceRecoveryWithDataLossCommand.cpp +++ b/fdbcli/ForceRecoveryWithDataLossCommand.cpp @@ -27,8 +27,6 @@ #include "flow/ThreadHelper.h" namespace fdb_cli { -// The CLI retains the tokens and their backing line until this command finishes or is cancelled. -// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future forceRecoveryWithDataLossCommandActor(Reference db, std::vector const& tokens) { if (tokens.size() != 2) { printUsage(tokens[0]); diff --git a/fdbcli/HotRangeCommand.cpp b/fdbcli/HotRangeCommand.cpp index a48b616aa96..ca184bd1386 100644 --- a/fdbcli/HotRangeCommand.cpp +++ b/fdbcli/HotRangeCommand.cpp @@ -51,12 +51,10 @@ ReadHotSubRangeRequest::SplitType parseSplitType(const std::string& typeStr) { namespace fdb_cli { -// The CLI retains the tokens and their backing line until this command finishes or is cancelled. Future hotRangeCommandActor(Database localdb, Reference db, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::vector const& tokens, - std::map* storage_interface) { + std::map* const& storage_interface) { if (tokens.size() == 1) { // initialize storage interfaces diff --git a/fdbcli/IdempotencyIdsCommand.cpp b/fdbcli/IdempotencyIdsCommand.cpp index 56547997970..1984b769cbc 100644 --- a/fdbcli/IdempotencyIdsCommand.cpp +++ b/fdbcli/IdempotencyIdsCommand.cpp @@ -36,8 +36,6 @@ Optional parseAgeValue(StringRef token) { namespace fdb_cli { -// The CLI retains the tokens and their backing line until this command finishes or is cancelled. -// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future idempotencyIdsCommandActor(Database db, std::vector const& tokens) { if (tokens.size() < 2 || tokens.size() > 3) { printUsage(tokens[0]); diff --git a/fdbcli/LockCommand.cpp b/fdbcli/LockCommand.cpp index 7e937272aae..e46950be033 100644 --- a/fdbcli/LockCommand.cpp +++ b/fdbcli/LockCommand.cpp @@ -60,8 +60,6 @@ namespace fdb_cli { const KeyRef lockSpecialKey = "\xff\xff/management/db_locked"_sr; -// The CLI retains the tokens and their backing line until this command finishes or is cancelled. -// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future lockCommandActor(Reference db, std::vector const& tokens) { if (tokens.size() != 1) { printUsage(tokens[0]); diff --git a/fdbcli/SnapshotCommand.cpp b/fdbcli/SnapshotCommand.cpp index afd35973136..ed8288a4dd0 100644 --- a/fdbcli/SnapshotCommand.cpp +++ b/fdbcli/SnapshotCommand.cpp @@ -27,8 +27,6 @@ #include "flow/ThreadHelper.h" namespace fdb_cli { -// The CLI retains the tokens and their backing line until this command finishes or is cancelled. -// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future snapshotCommandActor(Reference db, std::vector const& tokens) { bool result = true; if (tokens.size() < 2) { diff --git a/fdbcli/SuspendCommand.cpp b/fdbcli/SuspendCommand.cpp index 958ccaf2b88..3f6c7a6f811 100644 --- a/fdbcli/SuspendCommand.cpp +++ b/fdbcli/SuspendCommand.cpp @@ -31,10 +31,8 @@ #include "flow/ThreadHelper.h" namespace fdb_cli { -// The CLI retains the tokens and their backing line until this command finishes or is cancelled. Future suspendCommandActor(Reference db, Reference tr, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::vector const& tokens, std::map>* address_interface) { ASSERT(!tokens.empty()); diff --git a/fdbcli/include/fdbcli/fdbcli.h b/fdbcli/include/fdbcli/fdbcli.h index 820fa31bdf4..513bf8534ff 100644 --- a/fdbcli/include/fdbcli/fdbcli.h +++ b/fdbcli/include/fdbcli/fdbcli.h @@ -301,7 +301,7 @@ Future unlockDatabaseActor(Reference db, UID uid); Future hotRangeCommandActor(Database localDb, Reference db, std::vector const& tokens, - std::map* storage_interface); + std::map* const& storage_interface); // maintenance command Future setHealthyZone(Reference db, StringRef zoneId, double seconds, bool printWarning = false); diff --git a/fdbclient/S3BlobStore.cpp b/fdbclient/S3BlobStore.cpp index 0d2255923ad..10dc842d152 100644 --- a/fdbclient/S3BlobStore.cpp +++ b/fdbclient/S3BlobStore.cpp @@ -545,17 +545,11 @@ void S3BlobStoreEndpoint::processRequestFailure(Reference S3BlobStoreEndpoint::preRetryCheck( - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) - std::string const& verb, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) - std::string const& resource, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) - ReusableConnection& rconn, - int requestTimeout, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) - bool& retryExtended) { +Future S3BlobStoreEndpoint::preRetryCheck(std::string const& verb, + std::string const& resource, + ReusableConnection& rconn, + int requestTimeout, + bool& retryExtended) { if (!isWriteRequest(verb) || !CLIENT_KNOBS->BACKUP_ALLOW_DRYRUN) { co_return true; } diff --git a/fdbclient/include/fdbclient/IBlobStore.h b/fdbclient/include/fdbclient/IBlobStore.h index 80e731b900e..949f084efb3 100644 --- a/fdbclient/include/fdbclient/IBlobStore.h +++ b/fdbclient/include/fdbclient/IBlobStore.h @@ -451,7 +451,7 @@ class IBlobStoreEndpoint { ReusableConnection& rconn, int requestTimeout, bool& retryExtended) { - return true; + co_return true; } // Do an HTTP request to the blob store, read the response. Handles connection, retry, and authentication. diff --git a/fdbserver/backupworker/RangePartitionedBackupWorker.cpp b/fdbserver/backupworker/RangePartitionedBackupWorker.cpp index 991ed4e79bf..e28c14b86ce 100644 --- a/fdbserver/backupworker/RangePartitionedBackupWorker.cpp +++ b/fdbserver/backupworker/RangePartitionedBackupWorker.cpp @@ -391,10 +391,8 @@ Future pullPartitionMapFromTLog(RangePartitionedBackupData* self, Parti // Persist the (epoch, version) -> PartitionMap row to SS so older epoch backup workers can read it during // recovery. Multiple workers may call this concurrently for the same (epoch, version) but only one succeed in writing // to SS. -// The partition map is serialized into owned storage before suspension. Future persistPartitionMapToSS(RangePartitionedBackupData* self, Version partitionMapVersion, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) PartitionMap const& partitionMap) { auto tr = makeReference(self->cx); Key key = backupPartitionMapHistoryKeyFor(self->backupEpoch, partitionMapVersion); @@ -528,20 +526,16 @@ Future uploadPartitionList(RangePartitionedBackupData* self, PartitionMap // Persists partitionMap to SS history (so that catch-up backup workers can find it during recovery) and writes the // partitionId_keyRange_Map file for every active backup container. -// The partition-map owner awaits both persistence and upload before releasing it. Future persistAndUploadPartitionMap(RangePartitionedBackupData* self, Version pmVersion, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) PartitionMap const& partitionMap) { co_await persistPartitionMapToSS(self, pmVersion, partitionMap); co_await uploadPartitionList(self, partitionMap); } // Updates local routing state to use the new partition map. -// The partition map is read only before suspension. Future setActivePartitionMap(RangePartitionedBackupData* self, Version pmVersion, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) PartitionMap const& partitionMap) { self->logFolderBaseVersion = pmVersion; ASSERT(partitionMap.contains(self->tag)); diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index b2b75123498..92cf4ec3531 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -1289,19 +1289,15 @@ Future CDCProxy::rotateContendedPeek() { co_await delay(SERVER_KNOBS->BLOCKING_PEEK_TIMEOUT); } -// The awaited buffer pass retains its selection and mutable permit reservation. -Future CDCProxy::materializeBufferSelection( - Reference tag, - Reference cursor, - Version throughVersion, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) - CDCBufferSelection const& selection, - int64_t rawPeekReservation, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) - FlowLock::Releaser& reservation, - int64_t bufferLimit, - Prefetch prefetch, - Future invalidated) { +Future CDCProxy::materializeBufferSelection(Reference tag, + Reference cursor, + Version throughVersion, + CDCBufferSelection const& selection, + int64_t rawPeekReservation, + FlowLock::Releaser& reservation, + int64_t bufferLimit, + Prefetch prefetch, + Future invalidated) { const int64_t materializationReservation = reservation.remaining - rawPeekReservation; ASSERT_GE(materializationReservation, 0); if (selection.selectedBytes <= materializationReservation) { diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index fb8bd4a64bf..8d0e9695d09 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -579,8 +579,6 @@ Future monitorAndRecruitLogRouters(ClusterControllerData* self) { } } -// Proxy endpoints are copied into owned failure futures before suspension. -// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future> monitorCDCProxies(std::vector const& cdcProxies) { std::vector> failures; failures.reserve(cdcProxies.size()); @@ -616,12 +614,9 @@ bool containsCDCProxy(std::vector const& proxies, UID proxyId proxies.begin(), proxies.end(), [proxyId](CDCProxyInterface const& proxy) { return proxy.id() == proxyId; }); } -// The recruitment loop retains both snapshots until this awaited replacement pass finishes. Future recruitFailedCDCProxies(ClusterControllerData* self, uint64_t recoveryCount, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::vector const& monitoredProxies, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::vector const& failedIndexes) { if (!self->db.recoveryData.isValid() || self->db.recoveryData->cstate.myDBState.recoveryCount != recoveryCount) { co_return; diff --git a/fdbserver/clustercontroller/ClusterHealthIFactor.cpp b/fdbserver/clustercontroller/ClusterHealthIFactor.cpp index a4279eb5467..de593105438 100644 --- a/fdbserver/clustercontroller/ClusterHealthIFactor.cpp +++ b/fdbserver/clustercontroller/ClusterHealthIFactor.cpp @@ -42,8 +42,8 @@ WorkerEvents filterEmptyEvents(WorkerEvents const& events) { } Future fetchSpaceLevel(LatestWorkerEvents eventsAndErrors, - std::string availableBytesField, - std::string totalBytesField, + std::string const& availableBytesField, + std::string const& totalBytesField, double interventionThreshold, double criticalInterventionThreshold, char const* failureTraceEventName) { diff --git a/fdbserver/clustercontroller/ClusterHealthMonitor.cpp b/fdbserver/clustercontroller/ClusterHealthMonitor.cpp index 57baddf2a84..c3e8fa7cd8d 100644 --- a/fdbserver/clustercontroller/ClusterHealthMonitor.cpp +++ b/fdbserver/clustercontroller/ClusterHealthMonitor.cpp @@ -227,8 +227,6 @@ AsyncResult WorkerEventProvider::getLatestEvents(std::string return latestEventOnWorkers(workers, eventName); } -// The event name is copied into the request helper before suspension. -// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) AsyncResult WorkerEventProvider::getLatestRatekeeperEvents(std::string const& eventName) const { if (!ratekeeperWorker.present()) { co_return LatestWorkerEvents(); @@ -236,9 +234,7 @@ AsyncResult WorkerEventProvider::getLatestRatekeeperEvents(s co_return co_await latestEventOnWorker(ratekeeperWorker.get(), eventName); } -// The event name is copied into the request helper before suspension. AsyncResult WorkerEventProvider::getLatestDataDistributorEvents( - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::string const& eventName) const { if (!dataDistributorWorker.present()) { co_return LatestWorkerEvents(); diff --git a/fdbserver/clustercontroller/ClusterHealthMonitorTesting.cpp b/fdbserver/clustercontroller/ClusterHealthMonitorTesting.cpp index 10c8d0c1c55..296de920996 100644 --- a/fdbserver/clustercontroller/ClusterHealthMonitorTesting.cpp +++ b/fdbserver/clustercontroller/ClusterHealthMonitorTesting.cpp @@ -74,8 +74,6 @@ class FakeWorkerEventProvider final : public IWorkerEventProvider, public Refere latestTLogEventsByName[std::move(eventName)] = std::move(latestEvents); } - // The fake provider reads the name and produces its result without suspending. - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) AsyncResult getLatestEvents(std::string const& eventName) const override { auto it = latestEventsByName.find(eventName); if (it == latestEventsByName.end()) { @@ -90,8 +88,6 @@ class FakeWorkerEventProvider final : public IWorkerEventProvider, public Refere AsyncResult> areAllCoordinatorsReachable() const override { co_return allCoordinatorsReachable; } - // The fake provider reads the name and produces its result without suspending. - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) AsyncResult getLatestRatekeeperEvents(std::string const& eventName) const override { auto it = latestRatekeeperEventsByName.find(eventName); if (it != latestRatekeeperEventsByName.end()) { @@ -100,8 +96,6 @@ class FakeWorkerEventProvider final : public IWorkerEventProvider, public Refere co_return co_await getLatestEvents(eventName); } - // The fake provider reads the name and produces its result without suspending. - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) AsyncResult getLatestDataDistributorEvents(std::string const& eventName) const override { auto it = latestDataDistributorEventsByName.find(eventName); if (it != latestDataDistributorEventsByName.end()) { @@ -110,8 +104,6 @@ class FakeWorkerEventProvider final : public IWorkerEventProvider, public Refere co_return co_await getLatestEvents(eventName); } - // The fake provider reads the name and produces its result without suspending. - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) AsyncResult getLatestStorageServerEvents(std::string const& eventName) const override { auto it = latestStorageServerEventsByName.find(eventName); if (it == latestStorageServerEventsByName.end()) { @@ -120,8 +112,6 @@ class FakeWorkerEventProvider final : public IWorkerEventProvider, public Refere co_return it->second; } - // The fake provider reads the name and produces its result without suspending. - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) AsyncResult getLatestTLogEvents(std::string const& eventName) const override { auto it = latestTLogEventsByName.find(eventName); if (it == latestTLogEventsByName.end()) { diff --git a/fdbserver/core/MoveKeys.cpp b/fdbserver/core/MoveKeys.cpp index d13da293091..029d0bf44a4 100644 --- a/fdbserver/core/MoveKeys.cpp +++ b/fdbserver/core/MoveKeys.cpp @@ -1610,12 +1610,9 @@ static Optional decodeKeyServersState(RangeResult const& // owns the FlowLock slot). Returns the interfaces plus the read version at // which they were fetched — the read version is what waitForShardReady() // needs (see finishMoveKeys where we save it before dropping the txn). -// Both server lists are consumed into owned containers before suspension. static Future, Version>> buildKeysDestServerInterfaces( Transaction* tr, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::vector const& dest, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::vector const& completeSrc, bool hasRemote) { std::set completeSrcSet(completeSrc.begin(), completeSrc.end()); @@ -1651,14 +1648,10 @@ static Future, Version>> buildKeys // populated) when only TSS is slow, so subsequent iterations can skip it. // Returns `destSize - (SSes not ready)`; the caller retries if not equal to // `destSize`. `keys` is only used for tracing. -// The pending move retains these snapshots until this awaited phase finishes. -// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) static Future waitForKeysDestServers(std::vector const& storageServerInterfaces, int destSize, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) KeyRange const& keys, Version readVersion, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::map const& tssMapping, int* waitForTSSCounter, std::unordered_set* tssToIgnore, @@ -1753,17 +1746,13 @@ static Future waitForKeysDestServers(std::vector co // change during the wait; the caller retries via retryAfterPostWaitChange(). // `currentKeys` and `endKey` are in/out because a KRM boundary re-truncation // can shorten them. -// The pending move retains these snapshots until this awaited phase finishes. static Future reverifyKeysDestAndCommit(Transaction* tr, MoveKeysLock lock, const DDEnabledState* ddEnabledState, KeyRange* currentKeys, Key* endKey, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::vector const& dest, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::set const& allServers, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) KeyRange const& keys, UID relocationIntervalId, FinishMoveRetryBudget* retryBudget, @@ -2515,23 +2504,17 @@ struct DecodedShardsKeyServers { // running the per-sub-range AUDIT_DATAMOVE_PRE_CHECK when enabled. On a // stamp mismatch, sets *cancelDataMove=true and throws retry() so the outer // loop enters the cancel path. `dataMove` is only used for tracing. -// The pending move retains its metadata and read arenas throughout this awaited phase. -static Future decodeAndPreCheckShards( - Database occ, - Transaction* tr, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) - RangeResult const& UIDtoTagMap, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) - RangeResult const& keyServers, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) - std::vector const& destServers, - UID dataMoveId, - bool runPreCheck, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) - DataMoveMetaData const& dataMove, - UID relocationIntervalId, - Severity sevDm, - bool* cancelDataMove) { +static Future decodeAndPreCheckShards(Database occ, + Transaction* tr, + RangeResult const& UIDtoTagMap, + RangeResult const& keyServers, + std::vector const& destServers, + UID dataMoveId, + bool runPreCheck, + DataMoveMetaData const& dataMove, + UID relocationIntervalId, + Severity sevDm, + bool* cancelDataMove) { std::vector completeSrc; std::unordered_set allServers; @@ -2593,12 +2576,9 @@ static Future decodeAndPreCheckShards( // finishMoveShards analog of buildKeysDestServerInterfaces. Only difference: // a missing serverList entry throws retry() rather than asserting — shards // tolerates the SS-removed race by re-reading dataMove and starting over. -// Both server lists are consumed into owned containers before suspension. static Future, Version>> buildShardsDestServerInterfaces( Transaction* tr, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::vector const& destServers, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::vector const& completeSrc, bool hasRemote) { std::set completeSrcSet(completeSrc.begin(), completeSrc.end()); @@ -2642,15 +2622,10 @@ static Future, Version>> buildShar // count for the caller's ready-versus-target comparison; also fills // `readyServers_out` and `tssCount_out` for the caller's post-wait // tracing. -// The pending move retains these snapshots until this awaited phase finishes. -// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) static Future waitForShardsDestServers(std::vector const& storageServerInterfaces, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::vector const& newDestinationIds, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) KeyRange const& range, Version readVersion, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) std::map const& tssMapping, bool* skipTss, double* ssReadyTime, @@ -2659,7 +2634,6 @@ static Future waitForShardsDestServers(std::vector UID dataMoveId, UID relocationIntervalId, Severity sevDm, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) DataMoveMetaData const& dataMove) { std::vector> serverReady; // only for count below std::vector> tssReady; // for waiting in parallel with tss @@ -2739,26 +2713,22 @@ enum class ReverifyShardsResult { RetryLoop, PartialCommitted, FullyCommitted }; // the outer loop skips the per-sub-range AUDIT precheck on the next attempt. // `cancelDataMove` is set on bulk-load-outdated so the outer catch enters // the cancel path. -// The pending move retains these snapshots until this awaited phase finishes. -static Future reverifyShardsAndCommit( - Transaction* tr, - Database occ, - MoveKeysLock lock, - const DDEnabledState* ddEnabledState, - UID dataMoveId, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) - std::vector const& destServers, - KeyRange* range, - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) - std::unordered_set const& allServers, - Optional bulkLoadTaskState, - DataMoveMetaData* postWaitDataMove_out, - UID relocationIntervalId, - Severity sevDm, - FinishMoveRetryBudget* retryBudget, - bool* runPreCheck, - bool* cancelDataMove, - TxnCounters* counters) { +static Future reverifyShardsAndCommit(Transaction* tr, + Database occ, + MoveKeysLock lock, + const DDEnabledState* ddEnabledState, + UID dataMoveId, + std::vector const& destServers, + KeyRange* range, + std::unordered_set const& allServers, + Optional bulkLoadTaskState, + DataMoveMetaData* postWaitDataMove_out, + UID relocationIntervalId, + Severity sevDm, + FinishMoveRetryBudget* retryBudget, + bool* runPreCheck, + bool* cancelDataMove, + TxnCounters* counters) { tr->trState->taskID = TaskPriority::MoveKeys; tr->setOption(FDBTransactionOptions::PRIORITY_SYSTEM_IMMEDIATE); tr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); diff --git a/fdbserver/datadistributor/DDTxnProcessor.cpp b/fdbserver/datadistributor/DDTxnProcessor.cpp index 3d3852ab567..2fba953ddb9 100644 --- a/fdbserver/datadistributor/DDTxnProcessor.cpp +++ b/fdbserver/datadistributor/DDTxnProcessor.cpp @@ -327,8 +327,6 @@ class DDTxnProcessorImpl { // // serverKeys entries are left in place — they drain naturally as DD // moves shards using the old path. - // The initialization loop retains this transaction while awaiting metadata rewrites. - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) static Future rewriteShardEncodedMetadata(Transaction& tr, UID distributorId) { TraceEvent(SevInfo, "DDInitShardEncodeOff", distributorId) .detail("KnobValue", SERVER_KNOBS->SHARD_ENCODE_LOCATION_METADATA); diff --git a/fdbserver/networktest.cpp b/fdbserver/networktest.cpp index 4122777ae40..d8f6473cbc6 100644 --- a/fdbserver/networktest.cpp +++ b/fdbserver/networktest.cpp @@ -219,8 +219,6 @@ static void networkTestnanosleep() { return; } -// The server list is parsed into owned addresses before suspension. -// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future networkTestClient(std::string const& testServers) { if (testServers == "nanosleep") { networkTestnanosleep(); diff --git a/fdbserver/tester/CustomShardConfigWorkload.cpp b/fdbserver/tester/CustomShardConfigWorkload.cpp index fba7171835f..ff0feb112d6 100644 --- a/fdbserver/tester/CustomShardConfigWorkload.cpp +++ b/fdbserver/tester/CustomShardConfigWorkload.cpp @@ -25,7 +25,8 @@ #include "fdbclient/DataDistributionConfig.h" #include "fdbserver/tester/tester.h" -static Future customShardConfigWorkloadImpl(Database cx) { +Future customShardConfigWorkload(Database const& cxUnsafe) { + auto cx = cxUnsafe; ReadYourWritesTransaction tr(cx); bool verbose = (KEYBACKEDTYPES_DEBUG != 0); @@ -131,7 +132,3 @@ static Future customShardConfigWorkloadImpl(Database cx) { co_await tr.onError(err); } } - -Future customShardConfigWorkload(Database const& cxUnsafe) { - return customShardConfigWorkloadImpl(cxUnsafe); -} diff --git a/fdbserver/tester/DatabaseMaintenance.cpp b/fdbserver/tester/DatabaseMaintenance.cpp index dd65b99626b..fe957260870 100644 --- a/fdbserver/tester/DatabaseMaintenance.cpp +++ b/fdbserver/tester/DatabaseMaintenance.cpp @@ -153,7 +153,7 @@ std::string toHTML(const StringRef& binaryString) { } // namespace -static Future dumpDatabaseImpl(Database cx, std::string outputFilename, KeyRange range) { +Future dumpDatabase(Database const& cx, std::string const& outputFilename, KeyRange const& range) { try { Transaction tr(cx); while (true) { @@ -196,10 +196,6 @@ static Future dumpDatabaseImpl(Database cx, std::string outputFilename, Ke } } -Future dumpDatabase(Database const& cx, std::string const& outputFilename, KeyRange const& range) { - return dumpDatabaseImpl(cx, outputFilename, range); -} - std::vector aggregateMetrics(std::vector> metrics) { std::map> metricMap; for (int i = 0; i < metrics.size(); i++) { diff --git a/fdbserver/tester/TesterServer.cpp b/fdbserver/tester/TesterServer.cpp index 1e2b9409e9f..86f812a67f7 100644 --- a/fdbserver/tester/TesterServer.cpp +++ b/fdbserver/tester/TesterServer.cpp @@ -527,11 +527,21 @@ Future testerServerWorkload(WorkloadRequest work, } // namespace -static Future testerServerCoreImpl(TesterInterface interfCopy, - Reference ccrCopy, - Reference const> dbInfoCopy, - LocalityData localityCopy, - Optional expectedWorkLoadCopy) { +Future testerServerCore(TesterInterface const& interf, + Reference const& ccr, + Reference const> const& dbInfo, + LocalityData const& locality, + Optional const& expectedWorkLoad) { + // C++20 coroutine safety: const& parameters only store the reference in the coroutine frame, + // not the object. The referred-to object may be destroyed after the coroutine suspends + // (e.g. local variables in a caller's if-block, or temporaries from default arguments). + // Copy all const& parameters to ensure they survive across suspend points. + TesterInterface interfCopy = interf; + Reference ccrCopy = ccr; + Reference const> dbInfoCopy = dbInfo; + LocalityData localityCopy = locality; + Optional expectedWorkLoadCopy = expectedWorkLoad; + PromiseStream> addWorkload; Future workerFatalError = actorCollection(addWorkload.getFuture()); @@ -616,11 +626,3 @@ static Future testerServerCoreImpl(TesterInterface interfCopy, } co_return; } - -Future testerServerCore(TesterInterface const& interf, - Reference const& ccr, - Reference const> const& dbInfo, - LocalityData const& locality, - Optional const& expectedWorkLoad) { - return testerServerCoreImpl(interf, ccr, dbInfo, locality, expectedWorkLoad); -} diff --git a/fdbserver/tester/test.cpp b/fdbserver/tester/test.cpp index 8753d42dc6b..1bf71946483 100644 --- a/fdbserver/tester/test.cpp +++ b/fdbserver/tester/test.cpp @@ -65,9 +65,13 @@ void throwIfError(const std::vector>>& futures, std::string er } } -static Future runWorkloadImpl(Database cxCopy, - std::vector testersCopy, - TestSpec specCopy) { +Future runWorkload(Database const& cx, + std::vector const& testers, + TestSpec const& spec) { + // C++20 coroutine safety: copy const& params to survive across suspend points + Database cxCopy = cx; + std::vector testersCopy = testers; + TestSpec specCopy = spec; std::string name = printable(specCopy.title); TraceEvent("TestRunning") @@ -183,12 +187,6 @@ static Future runWorkloadImpl(Database cxCopy, co_return DistributedTestResults(aggregateMetrics(metricsResults), success, failure); } -Future runWorkload(Database const& cx, - std::vector const& testers, - TestSpec const& spec) { - return runWorkloadImpl(cx, testers, spec); -} - // Sets the database configuration by running the ChangeConfig workload Future changeConfiguration(Database cx, std::vector testers, StringRef configMode) { TestSpec spec; @@ -877,15 +875,29 @@ Future runTests8(Reference runTestsImpl(Reference connRecord, - test_type_t whatToRun, - test_location_t at, - int minTestersExpected, - std::string fileName, - Standalone startingConfiguration, - LocalityData locality, - UnitTestParameters testOptions, - bool restartingTest) { +Future runTests(Reference const& connRecordUnsafe, + test_type_t const& whatToRunUnsafe, + test_location_t const& atUnsafe, + int const& minTestersExpectedUnsafe, + std::string const& fileNameUnsafe, + StringRef const& startingConfigurationUnsafe, + LocalityData const& localityUnsafe, + UnitTestParameters const& testOptionsUnsafe, + bool const& restartingTestUnsafe) { + // C++20 coroutine safety: copy parameters that might bind to temporaries (default args). + // const& parameters only store the reference in the coroutine frame; temporaries are + // destroyed after the first suspend point, leaving dangling references. + // Just do this for all parameters, including ones that could be passed by value. + Reference connRecord = connRecordUnsafe; + test_type_t whatToRun = whatToRunUnsafe; + test_location_t at = atUnsafe; + int minTestersExpected = minTestersExpectedUnsafe; + std::string fileName = fileNameUnsafe; + StringRef startingConfiguration = startingConfigurationUnsafe; + LocalityData locality = localityUnsafe; + UnitTestParameters testOptions = testOptionsUnsafe; + bool restartingTest = restartingTestUnsafe; + TestSet testSet; std::unique_ptr knobProtectiveGroup(nullptr); auto cc = makeReference>>(); @@ -1003,26 +1015,6 @@ static Future runTestsImpl(Reference connRecord, .run(); } -Future runTests(Reference const& connRecordUnsafe, - test_type_t const& whatToRunUnsafe, - test_location_t const& atUnsafe, - int const& minTestersExpectedUnsafe, - std::string const& fileNameUnsafe, - StringRef const& startingConfigurationUnsafe, - LocalityData const& localityUnsafe, - UnitTestParameters const& testOptionsUnsafe, - bool const& restartingTestUnsafe) { - return runTestsImpl(connRecordUnsafe, - whatToRunUnsafe, - atUnsafe, - minTestersExpectedUnsafe, - fileNameUnsafe, - Standalone(startingConfigurationUnsafe), - localityUnsafe, - testOptionsUnsafe, - restartingTestUnsafe); -} - namespace { Future testExpectedErrorImpl(Future test, const char* testDescr, diff --git a/fdbserver/worker/worker.cpp b/fdbserver/worker/worker.cpp index e2aa4a3420e..5dd0bf244a1 100644 --- a/fdbserver/worker/worker.cpp +++ b/fdbserver/worker/worker.cpp @@ -2890,7 +2890,7 @@ class WorkerServerCore { lastSnapReq(lastSnapReq), snapReqMap(snapReqMap), snapReqResultMap(snapReqResultMap), lastSnapTime(lastSnapTime) {} - Future run(Future handleErrors) { + Future run(Future const& handleErrors) { auto res = co_await race(interf.clientInterface.reboot.getFuture(), serveServerDBInfoUpdates(), serveFailureInjectionRequests(), diff --git a/fdbserver/workloads/BackgroundSelectors.cpp b/fdbserver/workloads/BackgroundSelectors.cpp index 7399616590f..c6992f97cb7 100644 --- a/fdbserver/workloads/BackgroundSelectors.cpp +++ b/fdbserver/workloads/BackgroundSelectors.cpp @@ -49,9 +49,7 @@ struct BackgroundSelectorWorkload : TestWorkload { Future setup(Database const& cx) override { return Void(); } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { for (int c = 0; c < actorsPerClient; c++) clients.push_back(timeout(backgroundSelectorWorker(cx, this), testDuration, Void())); co_await waitForAll(clients); diff --git a/fdbserver/workloads/BackupToDBAbort.cpp b/fdbserver/workloads/BackupToDBAbort.cpp index e5a14ace3d1..2293ac51388 100644 --- a/fdbserver/workloads/BackupToDBAbort.cpp +++ b/fdbserver/workloads/BackupToDBAbort.cpp @@ -92,9 +92,7 @@ struct BackupToDBAbort : TestWorkload { } } - Future check(const Database& cx) override { return checkImpl(cx); } - - Future checkImpl(Database cx) { + Future check(const Database& cx) override { TraceEvent("BDBA_UnlockPrimary").log(); // Too much of the tester framework expects the primary database to be unlocked, so we unlock it // once all of the workloads have finished. diff --git a/fdbserver/workloads/BulkDumping.cpp b/fdbserver/workloads/BulkDumping.cpp index 05a298028a8..c6f2ae64290 100644 --- a/fdbserver/workloads/BulkDumping.cpp +++ b/fdbserver/workloads/BulkDumping.cpp @@ -465,9 +465,7 @@ struct BulkDumping : TestWorkload { // (9) Validate the loaded data in DB is same as the data in DB before dumping within the bulkdump job range and // bulkload job range. Note that the bulkload job can be unretriable error. In this case, we ignore the error range; // (10) Validate the bulk load job history. - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { if (clientId != 0) { co_return; } diff --git a/fdbserver/workloads/BulkLoading.cpp b/fdbserver/workloads/BulkLoading.cpp index c9ed2359f88..3dd3d7fd28f 100644 --- a/fdbserver/workloads/BulkLoading.cpp +++ b/fdbserver/workloads/BulkLoading.cpp @@ -742,9 +742,7 @@ struct BulkLoading : TestWorkload { TraceEvent("BulkLoadingWorkLoadComplexTestComplete"); } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { if (clientId != 0) { co_return; } diff --git a/fdbserver/workloads/ConfigureDatabase.cpp b/fdbserver/workloads/ConfigureDatabase.cpp index 3574995a549..3e82f43919c 100644 --- a/fdbserver/workloads/ConfigureDatabase.cpp +++ b/fdbserver/workloads/ConfigureDatabase.cpp @@ -269,15 +269,11 @@ struct ConfigureDatabaseWorkload : TestWorkload { return ManagementAPI::changeConfig(cx.getReference(), config, force); } - Future setup(Database const& cx) override { return setupImpl(cx); } - - Future setupImpl(Database cx) { + Future setup(Database const& cx) override { co_await ManagementAPI::changeConfig(cx.getReference(), "single storage_migration_type=aggressive", true); } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { DatabaseConfiguration config = co_await getDatabaseConfiguration(cx); TraceEvent("ConfigureDatabase_Config").detail("Config", config.toString()); if (!SERVER_KNOBS->SHARD_ENCODE_LOCATION_METADATA) { @@ -318,9 +314,7 @@ struct ConfigureDatabaseWorkload : TestWorkload { co_return false; } - Future check(Database const& cx) override { return checkImpl(cx); } - - Future checkImpl(Database cx) { + Future check(Database const& cx) override { co_await delay(30.0); // only storage_migration_type=gradual && perpetual_storage_wiggle=1 need this check because in QuietDatabase // perpetual wiggle will be forced to close For other cases, later ConsistencyCheck will check KV store type diff --git a/fdbserver/workloads/ConflictRange.cpp b/fdbserver/workloads/ConflictRange.cpp index 22dd9b2d142..e835d8a3a0b 100644 --- a/fdbserver/workloads/ConflictRange.cpp +++ b/fdbserver/workloads/ConflictRange.cpp @@ -65,9 +65,7 @@ struct ConflictRangeWorkload : TestWorkload { m.push_back(retries.getMetric()); } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { if (clientId == 0) co_await timeout(conflictRangeClient(cx, this), testDuration, Void()); } diff --git a/fdbserver/workloads/CpuProfiler.cpp b/fdbserver/workloads/CpuProfiler.cpp index e1aeeda453d..9c7b2ec8bb7 100644 --- a/fdbserver/workloads/CpuProfiler.cpp +++ b/fdbserver/workloads/CpuProfiler.cpp @@ -96,9 +96,7 @@ struct CpuProfilerWorkload : TestWorkload { } } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { co_await delay(initialDelay); if (clientId == 0) TraceEvent("SignalProfilerOn").log(); @@ -113,9 +111,7 @@ struct CpuProfilerWorkload : TestWorkload { } } - Future check(Database const& cx) override { return checkImpl(cx); } - - Future checkImpl(Database cx) { + Future check(Database const& cx) override { // If no duration was given, then shut the profiler off now if (duration <= 0) { if (clientId == 0) diff --git a/fdbserver/workloads/DDBalance.cpp b/fdbserver/workloads/DDBalance.cpp index 9a4e3da24dd..b889e947f18 100644 --- a/fdbserver/workloads/DDBalance.cpp +++ b/fdbserver/workloads/DDBalance.cpp @@ -56,9 +56,7 @@ struct DDBalanceWorkload : TestWorkload { Future setup(Database const& cx) override { return ddbalanceSetup(cx, this); } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { for (int c = 0; c < moversPerClient; c++) clients.push_back(timeout(ddBalanceMover(cx, this, c), testDuration, Void())); co_await waitForAll(clients); diff --git a/fdbserver/workloads/DDMetricsExclude.cpp b/fdbserver/workloads/DDMetricsExclude.cpp index 7f60d23da2b..00b3854c952 100644 --- a/fdbserver/workloads/DDMetricsExclude.cpp +++ b/fdbserver/workloads/DDMetricsExclude.cpp @@ -69,9 +69,7 @@ struct DDMetricsExcludeWorkload : TestWorkload { co_return -1.0; } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { try { std::vector excluded; excluded.push_back(AddressExclusion(IPAddress::parse(excludeIp.toString()).get(), excludePort)); diff --git a/fdbserver/workloads/DataDistributionMetrics.cpp b/fdbserver/workloads/DataDistributionMetrics.cpp index a56fecb481d..4a45c87f96c 100644 --- a/fdbserver/workloads/DataDistributionMetrics.cpp +++ b/fdbserver/workloads/DataDistributionMetrics.cpp @@ -196,9 +196,7 @@ struct DataDistributionMetricsWorkload : KVWorkload { co_return true; } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { std::vector> clients; clients.push_back(resultConsistencyCheckClient(cx, this)); for (int i = 0; i < actorCount; ++i) diff --git a/fdbserver/workloads/DifferentClustersSameRV.cpp b/fdbserver/workloads/DifferentClustersSameRV.cpp index d9d23701d96..6c59f6c5951 100644 --- a/fdbserver/workloads/DifferentClustersSameRV.cpp +++ b/fdbserver/workloads/DifferentClustersSameRV.cpp @@ -107,7 +107,7 @@ struct DifferentClustersSameRVWorkload : TestWorkload { void getMetrics(std::vector& m) override {} - static Future>> doRead(Database cx, Value keyToRead) { + static Future>> doRead(Database cx, Value const& keyToRead) { Transaction tr(cx); while (true) { tr.setOption(FDBTransactionOptions::READ_LOCK_AWARE); @@ -233,7 +233,7 @@ struct DifferentClustersSameRVWorkload : TestWorkload { co_await unlockDatabase(originalDB, lockUid); // So quietDatabase can finish } - static Future writerClient(Database cx, Value keyToRead) { + static Future writerClient(Database cx, Value const& keyToRead) { Transaction tr(cx); while (true) { Error err; @@ -259,7 +259,10 @@ struct DifferentClustersSameRVWorkload : TestWorkload { } } - static Future> readAtVersion(Value keyToRead, const char* name, Transaction* tr, Version version) { + static Future> readAtVersion(Value const& keyToRead, + const char* name, + Transaction* tr, + Version version) { Optional res; try { tr->reset(); @@ -272,7 +275,7 @@ struct DifferentClustersSameRVWorkload : TestWorkload { } } - static Future readerClientSeparateDBs(Database cx, Database extraDB, Value keyToRead) { + static Future readerClientSeparateDBs(Database cx, Database extraDB, Value const& keyToRead) { Transaction tr1(cx); Transaction tr2(extraDB); Version rv1{ 0 }; diff --git a/fdbserver/workloads/FileSystem.cpp b/fdbserver/workloads/FileSystem.cpp index 6702892440b..371f28f345e 100644 --- a/fdbserver/workloads/FileSystem.cpp +++ b/fdbserver/workloads/FileSystem.cpp @@ -161,9 +161,7 @@ struct FileSystemWorkload : TestWorkload { .detail("FilesToSetUp", nodesToSetUp); } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { FileSystemOp* operation; if (operationName == "deletionQuery") operation = new ServerDeletionCountQuery(); diff --git a/fdbserver/workloads/HealthMetricsApi.cpp b/fdbserver/workloads/HealthMetricsApi.cpp index 3e8e251dac5..ad6e7c068e4 100644 --- a/fdbserver/workloads/HealthMetricsApi.cpp +++ b/fdbserver/workloads/HealthMetricsApi.cpp @@ -63,9 +63,7 @@ struct HealthMetricsApiWorkload : TestWorkload { maxAllowedStaleness = getOption(options, "maxAllowedStaleness"_sr, 60.0); } - Future setup(Database const& cx) override { return setupImpl(cx); } - - Future setupImpl(Database cx) { + Future setup(Database const& cx) override { if (!sendDetailedHealthMetrics) { // Internally cached health metrics time out after this knob. Wait // an extra second to avoid any off-by-1 ">" vs ">=" type issues. @@ -74,9 +72,9 @@ struct HealthMetricsApiWorkload : TestWorkload { cx->healthMetrics.tLogQueue.clear(); } } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { co_await timeout(healthMetricsChecker(cx), testDuration, Void()); } + Future start(Database const& cx) override { + co_await timeout(healthMetricsChecker(cx), testDuration, Void()); + } Future check(Database const& cx) override { if (!gotMetrics) { diff --git a/fdbserver/workloads/HighContentionPrefixAllocatorWorkload.cpp b/fdbserver/workloads/HighContentionPrefixAllocatorWorkload.cpp index 733cd05328a..fbbb3ed3f44 100644 --- a/fdbserver/workloads/HighContentionPrefixAllocatorWorkload.cpp +++ b/fdbserver/workloads/HighContentionPrefixAllocatorWorkload.cpp @@ -117,9 +117,7 @@ struct HighContentionPrefixAllocatorWorkload : TestWorkload { Future start(Database const& cx) override { return runTest(cx); } - Future check(Database const& cx) override { return checkImpl(cx); } - - Future checkImpl(Database cx) { + Future check(Database const& cx) override { if (expectedPrefixes != allocatedPrefixes.size()) { TraceEvent(SevError, "HighContentionAllocationWorkloadFailure") .detail("Reason", "Incorrect Number of Prefixes Allocated") diff --git a/fdbserver/workloads/MemoryLifetime.cpp b/fdbserver/workloads/MemoryLifetime.cpp index 8f785221e08..86256708f16 100644 --- a/fdbserver/workloads/MemoryLifetime.cpp +++ b/fdbserver/workloads/MemoryLifetime.cpp @@ -45,16 +45,12 @@ struct MemoryLifetime : KVWorkload { void getMetrics(std::vector& m) override {} - Future setup(Database const& cx) override { return setupImpl(cx); } - - Future setupImpl(Database cx) { + Future setup(Database const& cx) override { Promise loadTime; co_await bulkSetup(cx, this, nodeCount, loadTime); } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { double startTime = now(); ReadYourWritesTransaction tr(cx); Reverse reverse = Reverse::False; diff --git a/fdbserver/workloads/MetricLogging.cpp b/fdbserver/workloads/MetricLogging.cpp index b93202df8ab..94ffd7007bc 100644 --- a/fdbserver/workloads/MetricLogging.cpp +++ b/fdbserver/workloads/MetricLogging.cpp @@ -49,9 +49,7 @@ struct MetricLoggingWorkload : TestWorkload { } } - Future setup(Database const& cx) override { return setupImpl(); } - - Future setupImpl() { + Future setup(Database const& cx) override { co_await delay(2.0); for (int i = 0; i < metricCount; i++) { if (testBool) { diff --git a/fdbserver/workloads/ProtocolVersion.cpp b/fdbserver/workloads/ProtocolVersion.cpp index 4f4f577b264..0096b2da8b8 100644 --- a/fdbserver/workloads/ProtocolVersion.cpp +++ b/fdbserver/workloads/ProtocolVersion.cpp @@ -27,9 +27,7 @@ struct ProtocolVersionWorkload : TestWorkload { explicit ProtocolVersionWorkload(WorkloadContext const& wcx) : TestWorkload(wcx) {} - Future start(Database const& cx) override { return startImpl(); } - - Future startImpl() { + Future start(Database const& cx) override { std::vector allProcesses = g_simulator->getAllProcesses(); auto diffVersionProcess = find_if(allProcesses.begin(), allProcesses.end(), [](const ISimulator::ProcessInfo* p) { diff --git a/fdbserver/workloads/QueuePush.cpp b/fdbserver/workloads/QueuePush.cpp index 8a376477247..9fbe630030b 100644 --- a/fdbserver/workloads/QueuePush.cpp +++ b/fdbserver/workloads/QueuePush.cpp @@ -92,9 +92,7 @@ struct QueuePushWorkload : TestWorkload { } } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { for (int i = 0; i < actorCount; i++) { clients.push_back(writeClient(cx, this)); } diff --git a/fdbserver/workloads/RandomMoveKeys.cpp b/fdbserver/workloads/RandomMoveKeys.cpp index 4ec68f31aca..d815721da86 100644 --- a/fdbserver/workloads/RandomMoveKeys.cpp +++ b/fdbserver/workloads/RandomMoveKeys.cpp @@ -57,9 +57,7 @@ struct MoveKeysWorkload : FailureInjectionWorkload { return alreadyAdded < 1 && work.useDatabase && 0.1 / (1 + alreadyAdded) > random.random01(); } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { if (enabled) { // Get the database configuration so as to use proper team size Transaction tr(cx); diff --git a/fdbserver/workloads/RandomRangeLock.cpp b/fdbserver/workloads/RandomRangeLock.cpp index d14f8c5b70c..b060a93cc47 100644 --- a/fdbserver/workloads/RandomRangeLock.cpp +++ b/fdbserver/workloads/RandomRangeLock.cpp @@ -164,9 +164,7 @@ struct RandomRangeLockWorkload : FailureInjectionWorkload { } } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { if (enabled) { // Run lockActorCount number of actor concurrently. // Each actor conducts (1) locking a range for a while and (2) unlocking the range. diff --git a/fdbserver/workloads/RangeLock.cpp b/fdbserver/workloads/RangeLock.cpp index 24a43173b45..b9d9abd2bba 100644 --- a/fdbserver/workloads/RangeLock.cpp +++ b/fdbserver/workloads/RangeLock.cpp @@ -827,9 +827,7 @@ struct RangeLocking : TestWorkload { ASSERT(remainingLocks.empty()); } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { if (clientId != 0) { co_return; } diff --git a/fdbserver/workloads/ReadAfterWrite.cpp b/fdbserver/workloads/ReadAfterWrite.cpp index 6955c3233a6..194fdf0782b 100644 --- a/fdbserver/workloads/ReadAfterWrite.cpp +++ b/fdbserver/workloads/ReadAfterWrite.cpp @@ -105,9 +105,7 @@ struct ReadAfterWriteWorkload : KVWorkload { } } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { Future lifetime = benchmark(cx); co_await delay(testDuration); } diff --git a/fdbserver/workloads/ReadHotDetection.cpp b/fdbserver/workloads/ReadHotDetection.cpp index 716ae13142d..4eb07e8c5de 100644 --- a/fdbserver/workloads/ReadHotDetection.cpp +++ b/fdbserver/workloads/ReadHotDetection.cpp @@ -45,9 +45,7 @@ struct ReadHotDetectionWorkload : TestWorkload { readKey = StringRef(format("testkey%08x", deterministicRandom()->randomInt(0, keyCount))); } - Future setup(Database const& cx) override { return setupImpl(cx); } - - Future setupImpl(Database cx) { + Future setup(Database const& cx) override { Standalone largeValue; Standalone smallValue; largeValue = randomString(largeValue.arena(), 100000); diff --git a/fdbserver/workloads/ReadWrite.cpp b/fdbserver/workloads/ReadWrite.cpp index 5e452d091c8..19e726e75ce 100644 --- a/fdbserver/workloads/ReadWrite.cpp +++ b/fdbserver/workloads/ReadWrite.cpp @@ -104,8 +104,6 @@ struct ReadWriteCommonImpl { } } - // The workload owns setup's mutable metrics and outlives its returned future. - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) static Future setup(Database cx, ReadWriteCommon& self) { if (!self.doSetup) co_return; diff --git a/fdbserver/workloads/S3ClientWorkload.cpp b/fdbserver/workloads/S3ClientWorkload.cpp index ad969058aaf..d68af07c2cd 100644 --- a/fdbserver/workloads/S3ClientWorkload.cpp +++ b/fdbserver/workloads/S3ClientWorkload.cpp @@ -153,9 +153,7 @@ struct S3ClientWorkload : TestWorkload { } } - Future start(Database const& cx) override { return startImpl(); } - - Future startImpl() { + Future start(Database const& cx) override { if (clientId != 0) { // Our simulation test can trigger multiple same workloads at the same time // Only run one time workload in the simulation diff --git a/fdbserver/workloads/SaveAndKill.cpp b/fdbserver/workloads/SaveAndKill.cpp index e26f4fb44b9..b52b9e97efb 100644 --- a/fdbserver/workloads/SaveAndKill.cpp +++ b/fdbserver/workloads/SaveAndKill.cpp @@ -55,9 +55,7 @@ struct SaveAndKillWorkload : TestWorkload { g_simulator->disableSwapsToAll(); return Void(); } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { int i{ 0 }; co_await delay(deterministicRandom()->random01() * testDuration); DatabaseConfiguration config = co_await getDatabaseConfiguration(cx); diff --git a/fdbserver/workloads/SkewedReadWrite.cpp b/fdbserver/workloads/SkewedReadWrite.cpp index 2a02fe0c8ba..074898f2920 100644 --- a/fdbserver/workloads/SkewedReadWrite.cpp +++ b/fdbserver/workloads/SkewedReadWrite.cpp @@ -204,9 +204,7 @@ struct SkewedReadWriteWorkload : ReadWriteCommon { } } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { std::vector> clients; if (enableReadLatencyLogging) clients.push_back(tracePeriodically()); diff --git a/fdbserver/workloads/SpecialKeySpaceCorrectness.cpp b/fdbserver/workloads/SpecialKeySpaceCorrectness.cpp index c460c850368..2bb9022d542 100644 --- a/fdbserver/workloads/SpecialKeySpaceCorrectness.cpp +++ b/fdbserver/workloads/SpecialKeySpaceCorrectness.cpp @@ -116,9 +116,7 @@ struct SpecialKeySpaceCorrectnessWorkload : TestWorkload { return Void(); } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { testRywLifetime(cx); co_await timeout(testSpecialKeySpaceErrors(cx, this) && getRangeCallActor(cx, this) && testConflictRanges(cx, /*read*/ true) && testConflictRanges(cx, /*read*/ false) && diff --git a/fdbserver/workloads/SpecialKeySpaceRobustness.cpp b/fdbserver/workloads/SpecialKeySpaceRobustness.cpp index 1282e2f100e..958c64253a1 100644 --- a/fdbserver/workloads/SpecialKeySpaceRobustness.cpp +++ b/fdbserver/workloads/SpecialKeySpaceRobustness.cpp @@ -39,9 +39,7 @@ struct SpecialKeySpaceRobustnessWorkload : TestWorkload { Future _setup(Database cx, SpecialKeySpaceRobustnessWorkload* self) { return Void(); } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { // Only use one client to avoid potential conflicts on changing cluster configuration if (clientId == 0) co_await managementApiCorrectnessActor(cx, this); diff --git a/fdbserver/workloads/StreamingRangeRead.cpp b/fdbserver/workloads/StreamingRangeRead.cpp index 9347b996858..34dd5362d89 100644 --- a/fdbserver/workloads/StreamingRangeRead.cpp +++ b/fdbserver/workloads/StreamingRangeRead.cpp @@ -92,9 +92,7 @@ struct StreamingRangeReadWorkload : KVWorkload { return delay(testDuration); } - Future check(Database const& cx) override { return checkImpl(cx); } - - Future checkImpl(Database cx) { + Future check(Database const& cx) override { client = Void(); co_await checkSelectorBoundaries(cx->clone()); co_return true; diff --git a/fdbserver/workloads/TaskBucketCorrectness.cpp b/fdbserver/workloads/TaskBucketCorrectness.cpp index f51cd58f6f0..50b0539fb62 100644 --- a/fdbserver/workloads/TaskBucketCorrectness.cpp +++ b/fdbserver/workloads/TaskBucketCorrectness.cpp @@ -253,9 +253,7 @@ struct TaskBucketCorrectnessWorkload : TestWorkload { co_await allDone->onSetAddTask(tr, taskBucket, taskDone); } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { auto tr = makeReference(cx); Subspace taskSubspace("backup-agent"_sr); auto taskBucket = makeReference(taskSubspace.get("tasks"_sr)); @@ -327,9 +325,7 @@ struct TaskBucketCorrectnessWorkload : TestWorkload { } } - Future check(Database const& cx) override { return checkImpl(cx); } - - Future checkImpl(Database cx) { + Future check(Database const& cx) override { bool ret = co_await runRYWTransaction( cx, [=](Reference tr) { return checkSayHello(tr, subtaskCount); }); co_return ret; diff --git a/fdbserver/workloads/ThreadSafety.cpp b/fdbserver/workloads/ThreadSafety.cpp index ca57a1964f6..03e74536a3d 100644 --- a/fdbserver/workloads/ThreadSafety.cpp +++ b/fdbserver/workloads/ThreadSafety.cpp @@ -137,9 +137,7 @@ struct ThreadSafetyWorkload : TestWorkload { Future setup(Database const& cx) override { return Void(); } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { std::vector threadInfo; Reference dbRef = diff --git a/fdbserver/workloads/Throttling.cpp b/fdbserver/workloads/Throttling.cpp index 1504d434f80..549a4fc066c 100644 --- a/fdbserver/workloads/Throttling.cpp +++ b/fdbserver/workloads/Throttling.cpp @@ -182,9 +182,7 @@ struct ThrottlingWorkload : KVWorkload { } } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { std::vector> clientActors; clientActors.reserve(static_cast(actorsPerClient > 0 ? actorsPerClient : 0) + 2); for (int actorId = 0; actorId < actorsPerClient; ++actorId) { diff --git a/fdbserver/workloads/TimeKeeperCorrectness.cpp b/fdbserver/workloads/TimeKeeperCorrectness.cpp index 637a63ea531..4c907f7892a 100644 --- a/fdbserver/workloads/TimeKeeperCorrectness.cpp +++ b/fdbserver/workloads/TimeKeeperCorrectness.cpp @@ -37,9 +37,7 @@ struct TimeKeeperCorrectnessWorkload : TestWorkload { void getMetrics(std::vector& m) override {} - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { TraceEvent(SevInfo, "TKCorrectness_Start").log(); double start = now(); @@ -66,9 +64,7 @@ struct TimeKeeperCorrectnessWorkload : TestWorkload { TraceEvent(SevInfo, "TKCorrectness_Completed").log(); } - Future check(Database const& cx) override { return checkImpl(cx); } - - Future checkImpl(Database cx) { + Future check(Database const& cx) override { KeyBackedMap dbTimeKeeper = KeyBackedMap(timeKeeperPrefixRange.begin); auto tr = makeReference(cx); diff --git a/fdbserver/workloads/TransactionCost.cpp b/fdbserver/workloads/TransactionCost.cpp index c3762ee70f2..53a90e59ca8 100644 --- a/fdbserver/workloads/TransactionCost.cpp +++ b/fdbserver/workloads/TransactionCost.cpp @@ -62,8 +62,6 @@ class TransactionCostWorkload : public TestWorkload { }; class ReadLargeValueTest : public ITest { - // The enclosing workload retains this test and outlives its setup future. - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future setupImpl(TransactionCostWorkload const& workload, Database cx) { Transaction tr(cx); while (true) { @@ -150,8 +148,6 @@ class TransactionCostWorkload : public TestWorkload { }; class ReadRangeTest : public ITest { - // The enclosing workload retains this test and outlives its setup future. - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future setupImpl(TransactionCostWorkload const& workload, Database cx) { Transaction tr(cx); while (true) { @@ -185,8 +181,6 @@ class TransactionCostWorkload : public TestWorkload { }; class ReadMultipleValuesTest : public ITest { - // The enclosing workload retains this test and outlives its setup future. - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future setupImpl(TransactionCostWorkload const& workload, Database cx) { Transaction tr(cx); while (true) { @@ -224,8 +218,6 @@ class TransactionCostWorkload : public TestWorkload { }; class LargeReadRangeTest : public ITest { - // The enclosing workload retains this test and outlives its setup future. - // NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future setupImpl(TransactionCostWorkload const& workload, Database cx) { Transaction tr(cx); while (true) { diff --git a/fdbserver/workloads/WatchAndWait.cpp b/fdbserver/workloads/WatchAndWait.cpp index da6d89d9d8d..1ed822edaa2 100644 --- a/fdbserver/workloads/WatchAndWait.cpp +++ b/fdbserver/workloads/WatchAndWait.cpp @@ -84,9 +84,7 @@ struct WatchAndWaitWorkload : TestWorkload { m.push_back(retries.getMetric()); } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { std::vector> watches; uint64_t endNode = (nodeCount * (clientId + 1)) / clientCount; uint64_t startNode = (nodeCount * clientId) / clientCount; diff --git a/fdbserver/workloads/Watches.cpp b/fdbserver/workloads/Watches.cpp index e73430f26a8..81f591df66a 100644 --- a/fdbserver/workloads/Watches.cpp +++ b/fdbserver/workloads/Watches.cpp @@ -55,9 +55,7 @@ struct WatchesWorkload : TestWorkload { out.insert("RandomRangeLock"); } - Future setup(Database const& cx) override { return setupImpl(cx); } - - Future setupImpl(Database cx) { + Future setup(Database const& cx) override { // return _setup(cx, this); std::vector> setupActors; for (int i = 0; i < nodes; i++) { diff --git a/fdbserver/workloads/WorkerErrors.cpp b/fdbserver/workloads/WorkerErrors.cpp index ccc8b504ab8..e1cf5216898 100644 --- a/fdbserver/workloads/WorkerErrors.cpp +++ b/fdbserver/workloads/WorkerErrors.cpp @@ -52,9 +52,7 @@ struct WorkerErrorsWorkload : TestWorkload { co_return results; } - Future start(Database const& cx) override { return startImpl(); } - - Future startImpl() { + Future start(Database const& cx) override { std::vector workers = co_await getWorkers(dbInfo); std::vector errors = co_await latestEventOnWorkers(workers); for (const auto& e : errors) { diff --git a/fdbserver/workloads/WriteBandwidth.cpp b/fdbserver/workloads/WriteBandwidth.cpp index 4525a158456..af58ca464fa 100644 --- a/fdbserver/workloads/WriteBandwidth.cpp +++ b/fdbserver/workloads/WriteBandwidth.cpp @@ -76,9 +76,7 @@ struct WriteBandwidthWorkload : KVWorkload { Standalone operator()(uint64_t n) { return KeyValueRef(keyForIndex(n, false), randomValue()); } - Future setup(Database const& cx) override { return setupImpl(cx); } - - Future setupImpl(Database cx) { + Future setup(Database const& cx) override { Promise loadTime; Promise>> ratesAtKeyCounts; @@ -86,9 +84,7 @@ struct WriteBandwidthWorkload : KVWorkload { this->loadTime = loadTime.getFuture().get(); } - Future start(Database const& cx) override { return startImpl(cx); } - - Future startImpl(Database cx) { + Future start(Database const& cx) override { for (int i = 0; i < actorCount; i++) { clients.push_back(writeClient(cx, this)); } diff --git a/flow/CoroTests.cpp b/flow/CoroTests.cpp index bf38e4e25ee..2a07f680ad8 100644 --- a/flow/CoroTests.cpp +++ b/flow/CoroTests.cpp @@ -2126,8 +2126,6 @@ void assertNoThrowOnCancelDestroyedAfterFirstWait(NoThrowOnCancelRecorder const& } template -// The test log outlives the awaited or explicitly cancelled coroutine. -// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future simple_await_test(std::stringstream& ss, Future f) { ss << "start. "; LifetimeLogger ll(ss, 0); @@ -2140,8 +2138,6 @@ Future simple_await_test(std::stringstream& ss, Future f) { ss << "after co_return. "; } -// The test log outlives the awaited or explicitly cancelled coroutine. -// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future actor_cancel_test(std::stringstream& ss) { ss << "start. "; @@ -2165,8 +2161,6 @@ Future actor_cancel_test(std::stringstream& ss) { ss << "after co_return. "; } -// Test futures finish or cancel before their event recorder is destroyed. -// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future noThrowOnCancelTest(NoThrowOnCancelRecorder& recorder, Future signal, NoThrowOnCancel = {}) { recorder.record(NoThrowOnCancelEvent::Start); @@ -2195,8 +2189,6 @@ Future noThrowOnCancelReentrantCancelTest(Future* result, co_await signal; } -// Test futures finish or cancel before their event recorder is destroyed. -// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future noThrowOnCancelValueTest(NoThrowOnCancelRecorder& recorder, Future signal, NoThrowOnCancel = {}) { recorder.record(NoThrowOnCancelEvent::Start); @@ -2208,8 +2200,6 @@ Future noThrowOnCancelValueTest(NoThrowOnCancelRecorder& recorder, Future noThrowOnCancelSequentialAwaitsTest(NoThrowOnCancelRecorder& recorder, Future firstSignal, Future secondSignal, @@ -2224,8 +2214,6 @@ Future noThrowOnCancelSequentialAwaitsTest(NoThrowOnCancelRecorder& record recorder.record(NoThrowOnCancelEvent::AfterWait); } -// Test futures finish or cancel before their event recorder is destroyed. -// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future noThrowOnCancelFutureStreamTest(NoThrowOnCancelRecorder& recorder, FutureStream stream, NoThrowOnCancel = {}) { @@ -2239,8 +2227,6 @@ Future noThrowOnCancelFutureStreamTest(NoThrowOnCancelRecorder& recorder, co_return value; } -// Test futures finish or cancel before their event recorder is destroyed. -// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future noThrowOnCancelThreadFutureStreamTest(NoThrowOnCancelRecorder& recorder, ThreadFutureStream stream, NoThrowOnCancel = {}) { @@ -2254,8 +2240,6 @@ Future noThrowOnCancelThreadFutureStreamTest(NoThrowOnCancelRecorder& recor co_return value; } -// The test log outlives the awaited or explicitly cancelled coroutine. -// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future actor_throw_test(std::stringstream& ss) { ss << "start. "; diff --git a/flow/IThreadPoolTest.cpp b/flow/IThreadPoolTest.cpp index 1169d8a2a80..0f7a1a3ea34 100644 --- a/flow/IThreadPoolTest.cpp +++ b/flow/IThreadPoolTest.cpp @@ -64,7 +64,7 @@ Future getThreadName(Reference pool) { return fut; } -Future waitForThreadName(Reference pool, std::string expectedName) { +Future waitForThreadName(Reference pool, std::string const& expectedName) { // startThread() sets the pthread name from the creating thread after pthread_create(), so the worker may // briefly report its default name before the requested name is visible. Some environments also report // ENOENT from pthread_setname_np(), which startThread() treats as non-fatal. diff --git a/flow/Net2.cpp b/flow/Net2.cpp index ed8f327d5f8..371995b9ee6 100644 --- a/flow/Net2.cpp +++ b/flow/Net2.cpp @@ -2080,24 +2080,18 @@ static Future coordinatorDNSCacheRefresh(Net2* self) { } } -static Future> resolveTCPEndpointWithDNSCacheImpl(Net2* self, - std::string host, - std::string service) { +Future> Net2::resolveTCPEndpointWithDNSCache(const std::string& host, + const std::string& service) { if (FLOW_KNOBS->ENABLE_COORDINATOR_DNS_CACHE) { - Optional> cache = self->dnsCache.find(host, service); + Optional> cache = dnsCache.find(host, service); if (cache.present()) { co_return cache.get(); } - std::vector addresses = co_await resolveTCPEndpoint_impl(self, host, service); - self->dnsCache.add(host, service, addresses); + std::vector addresses = co_await resolveTCPEndpoint_impl(this, host, service); + dnsCache.add(host, service, addresses); co_return addresses; } - co_return co_await resolveTCPEndpoint_impl(self, host, service); -} - -Future> Net2::resolveTCPEndpointWithDNSCache(const std::string& host, - const std::string& service) { - return resolveTCPEndpointWithDNSCacheImpl(this, host, service); + co_return co_await resolveTCPEndpoint_impl(this, host, service); } std::vector Net2::resolveTCPEndpointBlocking(const std::string& host, const std::string& service) { diff --git a/flow/UnitTest.cpp b/flow/UnitTest.cpp index e13f0aeed5b..9d3cd67ad5f 100644 --- a/flow/UnitTest.cpp +++ b/flow/UnitTest.cpp @@ -71,28 +71,3 @@ std::string UnitTestParameters::getDataDir() const { void UnitTestParameters::setDataDir(std::string const& dataDir) { this->dataDir = dataDir; } - -TEST_CASE("/flow/UnitTestParameters/coroutineOwnership") { - const std::string marker = "unitTestParameterOwnershipProbe"; - if (params.get(marker).present()) { - co_await delay(0.001); - ASSERT_EQ(params.get(marker).get(), std::string("original")); - co_return; - } - - UnitTest* registered = g_unittests.tests; - while (registered != nullptr && StringRef(registered->name) != "/flow/UnitTestParameters/coroutineOwnership"_sr) { - registered = registered->next; - } - ASSERT(registered != nullptr); - - Future pending; - { - UnitTestParameters callerParams; - callerParams.set(marker, std::string("original")); - pending = registered->func(callerParams); - ASSERT(!pending.isReady()); - callerParams.set(marker, std::string("changed")); - } - co_await pending; -} diff --git a/flow/UnitTestRunner.cpp b/flow/UnitTestRunner.cpp index 73fa86fc50a..a93e724e146 100644 --- a/flow/UnitTestRunner.cpp +++ b/flow/UnitTestRunner.cpp @@ -290,7 +290,9 @@ std::vector collectTests(const UnitTestRunnerOptions& options, const return tests; } -Future runTests(UnitTestRunnerOptions options, UnitTestRunnerConfig config, UnitTestRunnerResult* result) { +Future runTests(const UnitTestRunnerOptions& options, + const UnitTestRunnerConfig& config, + UnitTestRunnerResult* result) { std::vector tests = collectTests(options, config); result->testsAvailable = tests.size(); @@ -362,11 +364,11 @@ Future runTests(UnitTestRunnerOptions options, UnitTestRunnerConfig config } Future runTestsAfterInitialization(Future initialization, - UnitTestRunnerOptions options, - UnitTestRunnerConfig config, + const UnitTestRunnerOptions& options, + const UnitTestRunnerConfig& config, UnitTestRunnerResult* result) { co_await initialization; - co_await runTests(std::move(options), std::move(config), result); + co_await runTests(options, config, result); } Future stopNetworkAfter(Future what, std::string_view traceName, int* exitCode) { diff --git a/flow/bench/BenchAsyncResult.cpp b/flow/bench/BenchAsyncResult.cpp index 4e95c71e783..7585e394eac 100644 --- a/flow/bench/BenchAsyncResult.cpp +++ b/flow/bench/BenchAsyncResult.cpp @@ -43,14 +43,10 @@ struct ExpensivePayload { } }; -// This eager coroutine copies the payload into its result without suspending. -// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) Future returnFuturePayload(ExpensivePayload const& payload) { co_return payload; } -// This eager coroutine copies the payload into its result without suspending. -// NOLINTNEXTLINE(cppcoreguidelines-avoid-reference-coroutine-parameters) AsyncResult returnAsyncResultPayload(ExpensivePayload const& payload) { co_return payload; } diff --git a/flow/include/flow/UnitTest.h b/flow/include/flow/UnitTest.h index ec2b1161bb0..e786f4fb5d2 100644 --- a/flow/include/flow/UnitTest.h +++ b/flow/include/flow/UnitTest.h @@ -47,8 +47,7 @@ * } * * The body of a TEST_CASE returns a Future. It may be an ordinary function - * or a C++ coroutine using `co_await` and `co_return`. Each test body receives its - * own const copy of the parameters; a coroutine keeps that copy in its frame. + * or a C++ coroutine using `co_await` and `co_return`. * * Our tools for actually executing tests are external to flow (and use g_unittests to find test cases). * See the `UnitTestWorkload` class. @@ -121,19 +120,16 @@ extern bool noUnseed; #ifdef FLOW_DISABLE_UNIT_TESTS -#define TEST_CASE(name) static Future FILE_UNIQUE_NAME(disabled_testcase_func)(const UnitTestParameters params) +#define TEST_CASE(name) static Future FILE_UNIQUE_NAME(disabled_testcase_func)(const UnitTestParameters& params) #else #define TEST_CASE(name) \ - static Future FILE_UNIQUE_NAME(testcase_impl)(const UnitTestParameters params); \ - static Future FILE_UNIQUE_NAME(testcase_func)(const UnitTestParameters& params) { \ - return FILE_UNIQUE_NAME(testcase_impl)(params); \ - } \ + static Future FILE_UNIQUE_NAME(testcase_func)(const UnitTestParameters& params); \ namespace { \ static UnitTest FILE_UNIQUE_NAME(testcase)(name, __FILE__, __LINE__, &FILE_UNIQUE_NAME(testcase_func)); \ } \ - static Future FILE_UNIQUE_NAME(testcase_impl)(const UnitTestParameters params) + static Future FILE_UNIQUE_NAME(testcase_func)(const UnitTestParameters& params) #endif From 3ecdf61d776b9fd6ca924282c6fb6cccf797434d Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sat, 19 Sep 2026 17:17:22 -0700 Subject: [PATCH 118/170] Drop unused-return-value clang-tidy check --- .clang-tidy | 3 --- bindings/java/JavaWorkload.cpp | 3 +-- documentation/sphinx/source/clang-tidy.rst | 8 ++------ fdbrpc/JsonWebKeySet.cpp | 10 +--------- fdbrpc/SimExternalConnection.cpp | 6 ------ flow/Net2.cpp | 12 ------------ 6 files changed, 4 insertions(+), 38 deletions(-) diff --git a/.clang-tidy b/.clang-tidy index 1ee007891b7..3ff20bd1f31 100644 --- a/.clang-tidy +++ b/.clang-tidy @@ -35,7 +35,6 @@ Checks: > bugprone-undefined-memory-manipulation, bugprone-unhandled-self-assignment, bugprone-unique-ptr-array-mismatch, - bugprone-unused-return-value, bugprone-use-after-move, bugprone-virtual-near-miss, cppcoreguidelines-avoid-capturing-lambda-coroutines, @@ -60,8 +59,6 @@ Checks: > CheckOptions: - key: bugprone-dangling-handle.HandleClasses value: 'std::basic_string_view;std::experimental::basic_string_view;std::span;StringRef' - - key: bugprone-unused-return-value.CheckedFunctions - value: '^::std::async$;^::std::launder$;^::std::remove$;^::std::remove_if$;^::std::unique$;^::std::unique_ptr::release$;^::std::basic_string::empty$;^::std::vector::empty$;^::std::back_inserter$;^::std::distance$;^::std::find$;^::std::find_if$;^::std::inserter$;^::std::lower_bound$;^::std::make_pair$;^::std::map::count$;^::std::map::find$;^::std::map::lower_bound$;^::std::multimap::equal_range$;^::std::multimap::upper_bound$;^::std::set::count$;^::std::set::find$;^::std::setfill$;^::std::setprecision$;^::std::setw$;^::std::upper_bound$;^::std::vector::at$;^::bsearch$;^::ferror$;^::feof$;^::isalnum$;^::isalpha$;^::isblank$;^::iscntrl$;^::isdigit$;^::isgraph$;^::islower$;^::isprint$;^::ispunct$;^::isspace$;^::isupper$;^::iswalnum$;^::iswprint$;^::iswspace$;^::isxdigit$;^::memchr$;^::memcmp$;^::strcmp$;^::strcoll$;^::strncmp$;^::strpbrk$;^::strrchr$;^::strspn$;^::strstr$;^::wcscmp$;^::access$;^::bind$;^::connect$;^::difftime$;^::dlsym$;^::fnmatch$;^::getaddrinfo$;^::getopt$;^::htonl$;^::htons$;^::iconv_open$;^::inet_addr$;^::isascii$;^::isatty$;^::mmap$;^::newlocale$;^::openat$;^::pathconf$;^::pthread_equal$;^::pthread_getspecific$;^::pthread_mutex_trylock$;^::readdir$;^::readlink$;^::recvmsg$;^::regexec$;^::scandir$;^::semget$;^::setjmp$;^::shm_open$;^::shmget$;^::sigismember$;^::strcasecmp$;^::strsignal$;^::ttyname$;^::pthread_create$' - key: misc-coroutine-hostile-raii.RAIITypesList value: 'std::lock_guard;std::scoped_lock;MutexHolder;ThreadSpinLockHolder' - key: modernize-use-auto.MinTypeNameLength diff --git a/bindings/java/JavaWorkload.cpp b/bindings/java/JavaWorkload.cpp index 8d8f61ef8fe..332bd6d55cc 100644 --- a/bindings/java/JavaWorkload.cpp +++ b/bindings/java/JavaWorkload.cpp @@ -423,8 +423,7 @@ struct JVM { auto clazz = getClass("com/apple/foundationdb/testing/Promise"); auto res = env->NewObject(clazz, getMethod(clazz, "", "(J)V"), reinterpret_cast(p.get())); checkException(); - // The Java promise owns the native state until JavaPromise::send deletes it. - p.release(); // NOLINT(bugprone-unused-return-value) + p.release(); return res; } diff --git a/documentation/sphinx/source/clang-tidy.rst b/documentation/sphinx/source/clang-tidy.rst index 162fbec994a..92ab54a26fa 100644 --- a/documentation/sphinx/source/clang-tidy.rst +++ b/documentation/sphinx/source/clang-tidy.rst @@ -10,11 +10,11 @@ This guide explains how to run ``clang-tidy`` locally so you can fix issues befo What clang-tidy checks ====================== -FoundationDB configures 56 named checks in the ``.clang-tidy`` file at the repository root. The +FoundationDB configures 55 named checks in the ``.clang-tidy`` file at the repository root. The active set depends on the clang-tidy version and can be inspected with ``clang-tidy --list-checks``. The intent is to enable more as we go forward. Here are some example rules: -* **37 Bugprone rules** -- catch potential runtime errors, including unsafe self-assignment, forwarding constructors that hide copy or move constructors, narrow accumulation initializers, mismatched argument comments, obvious infinite loops, chained comparisons, swapped arguments, integer division in floating-point calculations, missed base-class copy construction, repeated macro argument evaluation, near-miss virtual overrides, dangling returned references, incorrect erase/remove calls, ignored return values, and incorrect POSIX error checks +* **36 Bugprone rules** -- catch potential runtime errors, including unsafe self-assignment, forwarding constructors that hide copy or move constructors, narrow accumulation initializers, mismatched argument comments, obvious infinite loops, chained comparisons, swapped arguments, integer division in floating-point calculations, missed base-class copy construction, repeated macro argument evaluation, near-miss virtual overrides, dangling returned references, incorrect erase/remove calls, and incorrect POSIX error checks * **1 C++ Core Guidelines rule** -- catch unsafe captures in coroutine lambdas (``cppcoreguidelines-avoid-capturing-lambda-coroutines``) * **2 Misc rules** -- catch redundant expressions and RAII objects held across coroutine suspension points * **4 Modernize rules** -- encourage modern C++ practices (e.g., ``modernize-use-auto``, ``modernize-use-override``) @@ -31,10 +31,6 @@ with suspicious fields. Reference-counted assignments that acquire the incoming reference before releasing the old one use documented, check-specific ``NOLINTNEXTLINE`` annotations where the checker cannot recognize their safety. -``bugprone-unused-return-value`` retains the standard checked-function list and -also checks ``pthread_create``. It does not diagnose every discarded Flow future: -intentional uncancellable helpers need different treatment from cancellable work. - Basic examples of ``clang-tidy`` style and performance improvement changes: .. code-block:: cpp diff --git a/fdbrpc/JsonWebKeySet.cpp b/fdbrpc/JsonWebKeySet.cpp index 8b61d4e5ba6..62e5c3f5653 100644 --- a/fdbrpc/JsonWebKeySet.cpp +++ b/fdbrpc/JsonWebKeySet.cpp @@ -356,32 +356,24 @@ Optional parseRsaKey(StringRef b64n, JWK_PARSE_ERROR_OSSL("RSA_set0_key()"); return {}; } - // Successful RSA_set0_key transfers ownership to rsa. - // NOLINTBEGIN(bugprone-unused-return-value) + // set0 == ownership taken by rsa, no need to free n.release(); e.release(); d.release(); - // NOLINTEND(bugprone-unused-return-value) if (!isPublic) { if (1 != ::RSA_set0_factors(rsa, p, q)) { JWK_PARSE_ERROR_OSSL("RSA_set0_factors()"); return {}; } - // Successful RSA_set0_factors transfers ownership to rsa. - // NOLINTBEGIN(bugprone-unused-return-value) p.release(); q.release(); - // NOLINTEND(bugprone-unused-return-value) if (1 != ::RSA_set0_crt_params(rsa, dp, dq, qi)) { JWK_PARSE_ERROR_OSSL("RSA_set0_crt_params()"); return {}; } - // Successful RSA_set0_crt_params transfers ownership to rsa. - // NOLINTBEGIN(bugprone-unused-return-value) dp.release(); dq.release(); qi.release(); - // NOLINTEND(bugprone-unused-return-value) } auto pkey = AutoCPointer(::EVP_PKEY_new(), &::EVP_PKEY_free); if (!pkey) { diff --git a/fdbrpc/SimExternalConnection.cpp b/fdbrpc/SimExternalConnection.cpp index 66233496d9f..4594350a5ea 100644 --- a/fdbrpc/SimExternalConnection.cpp +++ b/fdbrpc/SimExternalConnection.cpp @@ -51,8 +51,6 @@ class SimExternalConnectionImpl { const bool wasNonBlocking = self->socket.non_blocking(); boost::system::error_code err; if (!wasNonBlocking) { - // The error is checked through the output parameter. - // NOLINTNEXTLINE(bugprone-unused-return-value) self->socket.non_blocking(true, err); if (err) { throw connection_failed(); @@ -64,8 +62,6 @@ class SimExternalConnectionImpl { boost::system::error_code restoreErr; if (!wasNonBlocking) { - // The error is checked through the output parameter. - // NOLINTNEXTLINE(bugprone-unused-return-value) self->socket.non_blocking(false, restoreErr); } if (restoreErr || (err && err != boost::asio::error::would_block && err != boost::asio::error::try_again)) { @@ -98,8 +94,6 @@ class SimExternalConnectionImpl { address = boost::asio::ip::address_v4(ip.toV4()); } boost::system::error_code err; - // The error is checked through the output parameter. - // NOLINTNEXTLINE(bugprone-unused-return-value) socket.connect(ip::tcp::endpoint(address, toAddr.port), err); if (err) { co_return Reference(); diff --git a/flow/Net2.cpp b/flow/Net2.cpp index 371995b9ee6..4a6cccf8668 100644 --- a/flow/Net2.cpp +++ b/flow/Net2.cpp @@ -579,8 +579,6 @@ class Connection final : public IConnection, ReferenceCounted { void closeSocket() { boost::system::error_code error; - // The same error is returned through the output parameter and checked below. - // NOLINTNEXTLINE(bugprone-unused-return-value) socket.close(error); if (error) { TraceEvent(SevWarn, "N2_CloseError", id) @@ -728,8 +726,6 @@ class UDPSocket : public IUDPSocket, ReferenceCounted { void bind(NetworkAddress const& addr) override { boost::system::error_code ec; - // The same error is returned through the output parameter and checked below. - // NOLINTNEXTLINE(bugprone-unused-return-value) socket.bind(udpEndpoint(addr), ec); if (ec) { Error x; @@ -763,8 +759,6 @@ class UDPSocket : public IUDPSocket, ReferenceCounted { void closeSocket() { boost::system::error_code error; - // The same error is returned through the output parameter and checked below. - // NOLINTNEXTLINE(bugprone-unused-return-value) socket.close(error); if (error) { TraceEvent(SevWarn, "N2_CloseError", id) @@ -871,8 +865,6 @@ struct SSLHandshakerThread final : IThreadPoolReceiver { void action(Handshake& h) { try { - // Each operation returns the same error through h.err, which gates the next operation. - // NOLINTBEGIN(bugprone-unused-return-value) h.socket.next_layer().non_blocking(false, h.err); if (!h.err.failed()) { h.socket.handshake(h.type, h.err); @@ -880,7 +872,6 @@ struct SSLHandshakerThread final : IThreadPoolReceiver { if (!h.err.failed()) { h.socket.next_layer().non_blocking(true, h.err); } - // NOLINTEND(bugprone-unused-return-value) if (h.err.failed()) { TraceEvent(SevWarn, h.type == ssl_socket::handshake_type::client ? "N2_ConnectHandshakeError"_audit @@ -1289,15 +1280,12 @@ class SSLConnection final : public IConnection, ReferenceCounted } void closeSocket() { - // Teardown is best effort; errors cannot leave the connection usable. - // NOLINTBEGIN(bugprone-unused-return-value) boost::system::error_code cancelError; socket.cancel(cancelError); boost::system::error_code closeError; socket.close(closeError); boost::system::error_code shutdownError; ssl_sock.shutdown(shutdownError); - // NOLINTEND(bugprone-unused-return-value) } void onReadError(const boost::system::error_code& error) { From 6ada8c05e11d79749bcf46457031841e7a90759f Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sat, 19 Sep 2026 22:11:10 -0700 Subject: [PATCH 119/170] Fix trace metadata and assertion findings --- bindings/java/JavaWorkload.cpp | 2 +- fdbclient/ActorLineageProfiler.cpp | 2 +- fdbclient/S3Client.cpp | 4 +-- fdbclient/Tracing.cpp | 2 +- fdbmonitor/fdbmonitor_tests.cpp | 27 ++++++++++--------- fdbrpc/FlowGrpc.cpp | 2 +- fdbserver/SimulatedCluster.cpp | 2 +- fdbserver/commitproxy/CommitProxyServer.cpp | 4 +-- .../datadistributor/DDTeamCollection.cpp | 6 ++--- fdbserver/kvstore/VFSAsync.cpp | 21 ++++++++------- fdbserver/mocks3/MockS3Server.cpp | 1 - fdbserver/mocks3/MockS3ServerChaos.cpp | 1 - fdbserver/tester/DatabaseMaintenance.cpp | 4 +-- fdbserver/workloads/UnitTests.cpp | 2 +- flow/MkCertCli.cpp | 10 +++++-- flow/UnitTestRunner.cpp | 2 +- flow/flow.cpp | 2 +- flow/include/flow/TDMetric.h | 2 +- flow/include/flow/flow.h | 2 +- flow/include/flow/swift_stream_support.h | 2 +- 20 files changed, 53 insertions(+), 47 deletions(-) diff --git a/bindings/java/JavaWorkload.cpp b/bindings/java/JavaWorkload.cpp index 332bd6d55cc..27b17377135 100644 --- a/bindings/java/JavaWorkload.cpp +++ b/bindings/java/JavaWorkload.cpp @@ -27,6 +27,7 @@ #include "com_apple_foundationdb_testing_WorkloadContext.h" #include +#include #include #include #include @@ -80,7 +81,6 @@ void printTrace(JNIEnv* env, jclass, jlong logger, jint severity, jstring messag } else if (severity < 40) { sev = FDBSeverity::WarnAlways; } else { - assert(false); std::abort(); } log->trace(sev, msg, detailsMap); diff --git a/fdbclient/ActorLineageProfiler.cpp b/fdbclient/ActorLineageProfiler.cpp index 22c13ec1bf2..88e19f14f61 100644 --- a/fdbclient/ActorLineageProfiler.cpp +++ b/fdbclient/ActorLineageProfiler.cpp @@ -74,7 +74,7 @@ class Packer : public msgpack::packer { void visit(const std::any& val, Packer& packer) { auto iter = visitorMap.find(val.type()); if (iter == visitorMap.end()) { - TraceEvent(SevError, "PackerTypeNotFound").detail("Type", val.type().name()); + TraceEvent(SevError, "PackerTypeNotFound").detail("ValueType", val.type().name()); } else { iter->second(val, packer); } diff --git a/fdbclient/S3Client.cpp b/fdbclient/S3Client.cpp index f4392aecc44..81205a235ec 100644 --- a/fdbclient/S3Client.cpp +++ b/fdbclient/S3Client.cpp @@ -166,7 +166,7 @@ Reference getEndpoint(const std::string& s3url, return endpoint; } catch (Error& e) { - TraceEvent(SevError, "S3ClientGetEndpointFailed").detail("URL", StringRef(s3url)).detail("Error", e.what()); + TraceEvent(SevError, "S3ClientGetEndpointFailed").error(e).detail("URL", StringRef(s3url)); throw; } } @@ -1112,7 +1112,7 @@ Future listFiles(std::string s3url, int maxDepth) { } } } catch (Error& e) { - TraceEvent(SevError, "S3ClientListFilesError").detail("URL", s3url).detail("Error", e.what()); + TraceEvent(SevError, "S3ClientListFilesError").error(e).detail("URL", s3url); if (e.code() == error_code_backup_invalid_url) { std::cerr << "ERROR: Invalid blobstore URL: " << s3url << std::endl; } else if (e.code() == error_code_backup_auth_missing) { diff --git a/fdbclient/Tracing.cpp b/fdbclient/Tracing.cpp index 47b5ca3f353..f7ad960a6db 100644 --- a/fdbclient/Tracing.cpp +++ b/fdbclient/Tracing.cpp @@ -69,7 +69,7 @@ struct LogfileTracer : ITracer { for (const auto& event : span.events) { TraceEvent(SevInfo, "TracingSpanEvent", span.context.traceID) .detail("Name", event.name) - .detail("Time", event.time); + .detail("SpanEventTime", event.time); for (const auto& [key, value] : event.attributes) { TraceEvent(SevInfo, "TracingSpanEventAttribute", span.context.traceID) .detail("Key", key) diff --git a/fdbmonitor/fdbmonitor_tests.cpp b/fdbmonitor/fdbmonitor_tests.cpp index 01d22a52227..0dc2dd145aa 100644 --- a/fdbmonitor/fdbmonitor_tests.cpp +++ b/fdbmonitor/fdbmonitor_tests.cpp @@ -1,6 +1,6 @@ #include "fdbmonitor.h" -#include +#include #include #include @@ -125,27 +125,30 @@ void testPathOps() { testPathFunction2("parentDirectory", parentDirectory, "foo/./../foo2/./bar//", true, joinPath(cwd, "foo2/")); printf("%d errors.\n", errors); - assert(errors == 0); + assert_msg(errors == 0, "Path operation tests failed"); } void testEnvVarUtils() { // Ensure key-value extraction works const std::pair keyValuePair1{ "FOO", "BAR" }; - assert(keyValuePair1 == EnvVarUtils::extractKeyAndValue("FOO=BAR")); + assert_msg(keyValuePair1 == EnvVarUtils::extractKeyAndValue("FOO=BAR"), "Failed to extract FOO=BAR"); const std::pair keyValuePair2{ "x", "y" }; - assert(keyValuePair2 == EnvVarUtils::extractKeyAndValue("x=y")); + assert_msg(keyValuePair2 == EnvVarUtils::extractKeyAndValue("x=y"), "Failed to extract x=y"); const std::pair keyValuePair3{ "MALLOC_CONF", "prof:true,lg_prof_interval:30,prof_prefix:jeprof.out" }; - assert(keyValuePair3 == - EnvVarUtils::extractKeyAndValue("MALLOC_CONF=prof:true,lg_prof_interval:30,prof_prefix:jeprof.out")); + assert_msg(keyValuePair3 == + EnvVarUtils::extractKeyAndValue("MALLOC_CONF=prof:true,lg_prof_interval:30,prof_prefix:jeprof.out"), + "Failed to extract MALLOC_CONF"); // Ensure key-value validation passes for good inputs - assert(EnvVarUtils::keyValueValid("FOO=BAR", "FOO=BAR")); - assert(EnvVarUtils::keyValueValid("x=y", "FOO=BAR x=y")); - assert(EnvVarUtils::keyValueValid("MALLOC_CONF=prof:true,lg_prof_interval:30,prof_prefix:jeprof.out", - "MALLOC_CONF=prof:true,lg_prof_interval:30,prof_prefix:jeprof.out")); - assert(EnvVarUtils::keyValueValid("MALLOC_CONF=prof:true,lg_prof_interval:30,prof_prefix:jeprof.out", - "MALLOC_CONF=prof:true,lg_prof_interval:30,prof_prefix:jeprof.out FOO=BAR")); + assert_msg(EnvVarUtils::keyValueValid("FOO=BAR", "FOO=BAR"), "FOO=BAR should be valid"); + assert_msg(EnvVarUtils::keyValueValid("x=y", "FOO=BAR x=y"), "x=y should be valid in a list"); + assert_msg(EnvVarUtils::keyValueValid("MALLOC_CONF=prof:true,lg_prof_interval:30,prof_prefix:jeprof.out", + "MALLOC_CONF=prof:true,lg_prof_interval:30,prof_prefix:jeprof.out"), + "MALLOC_CONF should be valid"); + assert_msg(EnvVarUtils::keyValueValid("MALLOC_CONF=prof:true,lg_prof_interval:30,prof_prefix:jeprof.out", + "MALLOC_CONF=prof:true,lg_prof_interval:30,prof_prefix:jeprof.out FOO=BAR"), + "MALLOC_CONF should be valid in a list"); // Ensure key-value validation fails for bad inputs assert_msg(!EnvVarUtils::keyValueValid("", "FOO=BAR ="), "Key-Value can not be empty"); diff --git a/fdbrpc/FlowGrpc.cpp b/fdbrpc/FlowGrpc.cpp index 186f0a68a12..5e66ddc6c4e 100644 --- a/fdbrpc/FlowGrpc.cpp +++ b/fdbrpc/FlowGrpc.cpp @@ -68,7 +68,7 @@ Future GrpcServer::run() { co_await run_actor_; } catch (Error& err) { if (err.code() != error_code_operation_cancelled) { - TraceEvent(SevError, "GrpcServerRunError").detail("Endpoint", address_).detail("Error", err.name()); + TraceEvent(SevError, "GrpcServerRunError").error(err).detail("Endpoint", address_); throw; } } diff --git a/fdbserver/SimulatedCluster.cpp b/fdbserver/SimulatedCluster.cpp index 090b41777bb..42073d6de31 100644 --- a/fdbserver/SimulatedCluster.cpp +++ b/fdbserver/SimulatedCluster.cpp @@ -874,7 +874,7 @@ Future simulatedFDBDRebooter(ReferenceisSimulated() && e.code() != error_code_io_timeout && (bool)g_network->global(INetwork::enASIOTimedOut)) { TraceEvent(SevError, "IOTimeoutErrorSuppressed") - .detail("ErrorCode", e.code()) + .detail("ObservedErrorCode", e.code()) .detail("RandomId", randomId) .backtrace(); } diff --git a/fdbserver/commitproxy/CommitProxyServer.cpp b/fdbserver/commitproxy/CommitProxyServer.cpp index 5b2ecc821eb..12779b2b510 100644 --- a/fdbserver/commitproxy/CommitProxyServer.cpp +++ b/fdbserver/commitproxy/CommitProxyServer.cpp @@ -2209,9 +2209,7 @@ Future commitBatch(ProxyCommitData* pCommitData, if (err.code() == error_code_actor_cancelled) { throw; } - TraceEvent(SevInfo, "CommitBatchFailed", pCommitData->dbgid) - .detail("Stage", context.stage) - .detail("ErrorCode", err.code()); + TraceEvent(SevInfo, "CommitBatchFailed", pCommitData->dbgid).error(err).detail("Stage", context.stage); throw failed_to_progress(); } } diff --git a/fdbserver/datadistributor/DDTeamCollection.cpp b/fdbserver/datadistributor/DDTeamCollection.cpp index b93baab5d29..25211170fc7 100644 --- a/fdbserver/datadistributor/DDTeamCollection.cpp +++ b/fdbserver/datadistributor/DDTeamCollection.cpp @@ -153,7 +153,7 @@ class DDTeamCollectionImpl { start = now(); } } catch (Error& e) { - TraceEvent("CheckAndRemoveInvalidLocalityAddrRetry", self->distributorId).detail("Error", e.what()); + TraceEvent("CheckAndRemoveInvalidLocalityAddrRetry", self->distributorId).error(e); } } } @@ -5337,7 +5337,7 @@ void DDTeamCollection::rebuildMachineLocalityMap() { for (auto& [_, machine] : machine_info) { if (machine->serversOnMachine.empty()) { TraceEvent(SevWarn, "RebuildMachineLocalityMapError") - .detail("Machine", machine->machineID.toString()) + .detail("MachineID", machine->machineID.toString()) .detail("NumServersOnMachine", 0); continue; } @@ -5348,7 +5348,7 @@ void DDTeamCollection::rebuildMachineLocalityMap() { auto& locality = representativeServer->getLastKnownInterface().locality; if (!isValidLocality(configuration.storagePolicy, locality)) { TraceEvent(SevWarn, "RebuildMachineLocalityMapError") - .detail("Machine", machine->machineID.toString()) + .detail("MachineID", machine->machineID.toString()) .detail("InvalidLocality", locality.toString()); continue; } diff --git a/fdbserver/kvstore/VFSAsync.cpp b/fdbserver/kvstore/VFSAsync.cpp index 5f7f5508bb9..103dfed5d7b 100644 --- a/fdbserver/kvstore/VFSAsync.cpp +++ b/fdbserver/kvstore/VFSAsync.cpp @@ -250,7 +250,7 @@ static int asyncLock(sqlite3_file* pFile, int eLock) { return eLock == EXCLUSIVE_LOCK ? SQLITE_BUSY : SQLITE_OK; } static int asyncUnlock(sqlite3_file* pFile, int eLock) { - assert(eLock <= SHARED_LOCK); + ASSERT_ABORT(eLock <= SHARED_LOCK); return SQLITE_OK; } @@ -347,7 +347,7 @@ static int asyncShmMap(sqlite3_file* fd, /* Handle open on database file */ ++memInfo->refcount; // printf("Shared memory for: '%s' (%d refs)\n", filename.c_str(), memInfo->refcount); } else { - assert(memInfo->regionSize == szRegion); + ASSERT_ABORT(memInfo->regionSize == szRegion); } if (iRegion >= memInfo->regions.size()) { @@ -380,11 +380,12 @@ static int asyncShmLock(sqlite3_file* fd, /* Database file holding the shared me int n, /* Number of locks to acquire or release */ int flags /* What to do with the lock */ ) { - assert(ofst >= 0 && ofst + n <= SQLITE_SHM_NLOCK); - assert(n >= 1); - assert(flags == (SQLITE_SHM_LOCK | SQLITE_SHM_SHARED) || flags == (SQLITE_SHM_LOCK | SQLITE_SHM_EXCLUSIVE) || - flags == (SQLITE_SHM_UNLOCK | SQLITE_SHM_SHARED) || flags == (SQLITE_SHM_UNLOCK | SQLITE_SHM_EXCLUSIVE)); - assert(n == 1 || (flags & SQLITE_SHM_EXCLUSIVE) != 0); + ASSERT_ABORT(ofst >= 0 && ofst + n <= SQLITE_SHM_NLOCK); + ASSERT_ABORT(n >= 1); + ASSERT_ABORT(flags == (SQLITE_SHM_LOCK | SQLITE_SHM_SHARED) || flags == (SQLITE_SHM_LOCK | SQLITE_SHM_EXCLUSIVE) || + flags == (SQLITE_SHM_UNLOCK | SQLITE_SHM_SHARED) || + flags == (SQLITE_SHM_UNLOCK | SQLITE_SHM_EXCLUSIVE)); + ASSERT_ABORT(n == 1 || (flags & SQLITE_SHM_EXCLUSIVE) != 0); MutexHolder hold(SharedMemoryInfo::mutex); @@ -617,9 +618,9 @@ static int asyncAccess(sqlite3_vfs* pVfs, const char* zPath, int flags, int* pRe int rc; /* access() return code */ int eAccess = F_OK; /* Second argument to access() */ - assert(flags == SQLITE_ACCESS_EXISTS /* access(zPath, F_OK) */ - || flags == SQLITE_ACCESS_READ /* access(zPath, R_OK) */ - || flags == SQLITE_ACCESS_READWRITE /* access(zPath, R_OK|W_OK) */ + ASSERT_ABORT(flags == SQLITE_ACCESS_EXISTS /* access(zPath, F_OK) */ + || flags == SQLITE_ACCESS_READ /* access(zPath, R_OK) */ + || flags == SQLITE_ACCESS_READWRITE /* access(zPath, R_OK|W_OK) */ ); if (flags == SQLITE_ACCESS_READWRITE) diff --git a/fdbserver/mocks3/MockS3Server.cpp b/fdbserver/mocks3/MockS3Server.cpp index bc1b30c25f1..9e559e3133b 100644 --- a/fdbserver/mocks3/MockS3Server.cpp +++ b/fdbserver/mocks3/MockS3Server.cpp @@ -1593,7 +1593,6 @@ Future registerMockS3Server_impl(std::string ip, std::string port) { TraceEvent(SevError, "MockS3ServerRegistrationFailed") .error(e) .detail("Address", serverKey) - .detail("ErrorCode", e.code()) .detail("ErrorName", e.name()); throw; } diff --git a/fdbserver/mocks3/MockS3ServerChaos.cpp b/fdbserver/mocks3/MockS3ServerChaos.cpp index 1eaac2c86d3..64b07e6d0ab 100644 --- a/fdbserver/mocks3/MockS3ServerChaos.cpp +++ b/fdbserver/mocks3/MockS3ServerChaos.cpp @@ -345,7 +345,6 @@ Future registerMockS3ChaosServer(std::string ip, std::string port) { TraceEvent(SevError, "MockS3ChaosServerRegistrationFailed") .error(e) .detail("Address", serverKey) - .detail("ErrorCode", e.code()) .detail("ErrorName", e.name()); throw; } diff --git a/fdbserver/tester/DatabaseMaintenance.cpp b/fdbserver/tester/DatabaseMaintenance.cpp index fe957260870..357081eaf6b 100644 --- a/fdbserver/tester/DatabaseMaintenance.cpp +++ b/fdbserver/tester/DatabaseMaintenance.cpp @@ -67,7 +67,7 @@ Future clearData(Database cx) { break; } catch (Error& e) { TraceEvent(SevWarn, "TesterClearingDatabaseError", tr.trState->readOptions.get().debugID.get()).error(e); - TraceEvent("ClearData_Loop1_Catch").detail("Phase", "Loop1_Error").detail("ErrorCode", e.code()); + TraceEvent("ClearData_Loop1_Catch").error(e).detail("Phase", "Loop1_Error"); err = e; } @@ -109,7 +109,7 @@ Future clearData(Database cx) { } catch (Error& e) { TraceEvent(SevWarn, "TesterCheckDatabaseClearedError", tr.trState->readOptions.get().debugID.get()) .error(e); - TraceEvent("ClearData_Loop2_Catch").detail("Phase", "Loop2_Error").detail("ErrorCode", e.code()); + TraceEvent("ClearData_Loop2_Catch").error(e).detail("Phase", "Loop2_Error"); caughtError = e; needsErrorHandling = true; } diff --git a/fdbserver/workloads/UnitTests.cpp b/fdbserver/workloads/UnitTests.cpp index 1b3a90ee4fa..61fc5235097 100644 --- a/fdbserver/workloads/UnitTests.cpp +++ b/fdbserver/workloads/UnitTests.cpp @@ -215,7 +215,7 @@ struct UnitTestWorkload : TestWorkload { .detail("Name", test->name) .detail("File", test->file) .detail("Line", test->line) - .detail("Rand", deterministicRandom()->randomInt(0, 100001)); + .detail("Rand", debugRandom()->randomInt(0, 100001)); Error result = success(); double start_now = now(); diff --git a/flow/MkCertCli.cpp b/flow/MkCertCli.cpp index 02b3f615910..b006dd9d71b 100644 --- a/flow/MkCertCli.cpp +++ b/flow/MkCertCli.cpp @@ -229,7 +229,10 @@ int main(int argc, char** argv) { case OPT_SERVER_CHAIN_LEN: try { serverArgs.length = std::stoul(args.OptionArg()); - assert(serverArgs.length > 0); + if (serverArgs.length == 0) { + fmt::print(stderr, "ERROR: Certificate chain length must be positive\n"); + return FDB_EXIT_ERROR; + } } catch (std::exception const& ex) { fmt::print(stderr, "ERROR: Invalid chain length ({})\n", ex.what()); return FDB_EXIT_ERROR; @@ -238,7 +241,10 @@ int main(int argc, char** argv) { case OPT_CLIENT_CHAIN_LEN: try { clientArgs.length = std::stoul(args.OptionArg()); - assert(clientArgs.length > 0); + if (clientArgs.length == 0) { + fmt::print(stderr, "ERROR: Certificate chain length must be positive\n"); + return FDB_EXIT_ERROR; + } } catch (std::exception const& ex) { fmt::print(stderr, "ERROR: Invalid chain length ({})\n", ex.what()); return FDB_EXIT_ERROR; diff --git a/flow/UnitTestRunner.cpp b/flow/UnitTestRunner.cpp index a93e724e146..44ecaf2e879 100644 --- a/flow/UnitTestRunner.cpp +++ b/flow/UnitTestRunner.cpp @@ -328,7 +328,7 @@ Future runTests(const UnitTestRunnerOptions& options, .detail("Name", test->name) .detail("File", test->file) .detail("Line", test->line) - .detail("Rand", deterministicRandom()->randomInt(0, 100001)); + .detail("Rand", debugRandom()->randomInt(0, 100001)); Error resultCode = success(); double startNow = now(); diff --git a/flow/flow.cpp b/flow/flow.cpp index 4dc1e790608..605691bbef3 100644 --- a/flow/flow.cpp +++ b/flow/flow.cpp @@ -422,7 +422,7 @@ void bindDeterministicRandomToOpenssl() { } int nChooseK(int n, int k) { - assert(n >= k && k >= 0); + ASSERT(n >= k && k >= 0); if (k == 0) { return 1; } diff --git a/flow/include/flow/TDMetric.h b/flow/include/flow/TDMetric.h index a32a6da6985..8725d35003f 100644 --- a/flow/include/flow/TDMetric.h +++ b/flow/include/flow/TDMetric.h @@ -173,7 +173,7 @@ struct MetricBatch { MetricBatch() = default; explicit MetricBatch(FDBScope* in) { - assert(in != nullptr); + ASSERT(in != nullptr); scope.inserts = std::move(in->inserts); scope.appends = std::move(in->appends); scope.updates = std::move(in->updates); diff --git a/flow/include/flow/flow.h b/flow/include/flow/flow.h index 9dc08741f8f..e199deca5d3 100644 --- a/flow/include/flow/flow.h +++ b/flow/include/flow/flow.h @@ -937,7 +937,7 @@ class void set(const void* _Nonnull pointerToContinuationInstance, Future f, const void* _Nonnull thisPointer) { // Verify Swift did not make a copy of the `self` value for this method // call. - assert(this == thisPointer); + ASSERT_ABORT(this == thisPointer); // FIXME: Propagate `SwiftCC` to Swift using forward // interop, without relying on passing it via a `void *` diff --git a/flow/include/flow/swift_stream_support.h b/flow/include/flow/swift_stream_support.h index 6919a121bb7..db08d45df4c 100644 --- a/flow/include/flow/swift_stream_support.h +++ b/flow/include/flow/swift_stream_support.h @@ -46,7 +46,7 @@ class FlowSingleCallbackForSwiftContinuation : SingleCallback { void set(const void* _Nonnull pointerToContinuationInstance, FutureStream fs, const void* _Nonnull thisPointer) { // Verify Swift did not make a copy of the `self` value for this method // call. - assert(this == thisPointer); + ASSERT_ABORT(this == thisPointer); // FIXME: Propagate `SwiftCC` to Swift using forward // interop, without relying on passing it via a `void *` From fa97c08c5885d6e4a22d59ec96c2e7074dd4dd53 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sat, 19 Sep 2026 22:27:57 -0700 Subject: [PATCH 120/170] Document intentional blocking waits in coroutines --- fdbrpc/SimExternalConnection.cpp | 2 ++ fdbserver/kvstore/FDBExecHelper.cpp | 2 ++ fdbserver/networktest.cpp | 2 ++ fdbserver/worker/worker.cpp | 2 ++ 4 files changed, 8 insertions(+) diff --git a/fdbrpc/SimExternalConnection.cpp b/fdbrpc/SimExternalConnection.cpp index 4594350a5ea..66a6b44e8b3 100644 --- a/fdbrpc/SimExternalConnection.cpp +++ b/fdbrpc/SimExternalConnection.cpp @@ -77,6 +77,8 @@ class SimExternalConnectionImpl { while (self->readBuffer.empty()) { readAvailable(self); if (self->readBuffer.empty()) { + // Give the external peer wall-clock time; simulated delay alone cannot wait for it. + // ast-grep-ignore: fdb-no-blocking-sleep-in-coroutine threadSleep(0.01); co_await delayJittered(0.1); } diff --git a/fdbserver/kvstore/FDBExecHelper.cpp b/fdbserver/kvstore/FDBExecHelper.cpp index ab705241915..c045e528983 100644 --- a/fdbserver/kvstore/FDBExecHelper.cpp +++ b/fdbserver/kvstore/FDBExecHelper.cpp @@ -233,6 +233,8 @@ Future spawnProcess(std::string path, // child process has not completed yet if (isSync || g_network->isSimulated()) { // synchronously sleep + // Synchronous and simulated execution must wait without advancing other actors. + // ast-grep-ignore: fdb-no-blocking-sleep-in-coroutine threadSleep(0.1); } else { // yield for other actors to run diff --git a/fdbserver/networktest.cpp b/fdbserver/networktest.cpp index d8f6473cbc6..da087bc87ab 100644 --- a/fdbserver/networktest.cpp +++ b/fdbserver/networktest.cpp @@ -548,6 +548,8 @@ struct P2PNetworkTest { } catch (Error& e) { printf("Server: handshake error %s\n", e.what()); } + // Keep the one-shot TLS benchmark quiescent on this thread after its handshake. + // ast-grep-ignore: fdb-no-blocking-sleep-in-coroutine threadSleep(11.0); co_return; } diff --git a/fdbserver/worker/worker.cpp b/fdbserver/worker/worker.cpp index 5dd0bf244a1..60a8a58965b 100644 --- a/fdbserver/worker/worker.cpp +++ b/fdbserver/worker/worker.cpp @@ -2087,6 +2087,8 @@ class WorkerServerCore { flushTraceFileVoid(); setProfilingEnabled(0); g_network->stop(); + // The network has stopped, so process suspension cannot depend on its event loop. + // ast-grep-ignore: fdb-no-blocking-sleep-in-coroutine threadSleep(req.waitForDuration); } if (rebootReq.checkData) { From 55bc8d2dcdf741a481384978a5a214270078b80a Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sat, 19 Sep 2026 22:37:25 -0700 Subject: [PATCH 121/170] Enable disconnect monitoring in ClogRemoteTLog --- tests/rare/ClogRemoteTLog.toml | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tests/rare/ClogRemoteTLog.toml b/tests/rare/ClogRemoteTLog.toml index 0d8dfc91b47..89760125b04 100644 --- a/tests/rare/ClogRemoteTLog.toml +++ b/tests/rare/ClogRemoteTLog.toml @@ -12,6 +12,8 @@ worker_health_monitor_interval = 10 cc_enable_worker_health_monitor = true cc_health_trigger_recovery = true cc_enable_remote_tlog_degradation_monitoring = true +# Clogged links can be reported as disconnected instead of degraded. +cc_enable_remote_tlog_disconnect_monitoring = true gray_failure_allow_remote_ss_to_complain = true cc_worker_health_checking_interval = 45 cc_min_degradation_interval = 30 From f60931cc2130db4fcb4fe9023aa59962a0806443 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sat, 19 Sep 2026 22:41:47 -0700 Subject: [PATCH 122/170] Restore excluded files and test RNG behavior --- fdbmonitor/fdbmonitor_tests.cpp | 27 ++++++++++++--------------- fdbrpc/SimExternalConnection.cpp | 2 -- fdbserver/kvstore/FDBExecHelper.cpp | 2 -- fdbserver/networktest.cpp | 2 -- fdbserver/worker/worker.cpp | 2 -- fdbserver/workloads/UnitTests.cpp | 2 +- flow/UnitTestRunner.cpp | 2 +- 7 files changed, 14 insertions(+), 25 deletions(-) diff --git a/fdbmonitor/fdbmonitor_tests.cpp b/fdbmonitor/fdbmonitor_tests.cpp index 0dc2dd145aa..01d22a52227 100644 --- a/fdbmonitor/fdbmonitor_tests.cpp +++ b/fdbmonitor/fdbmonitor_tests.cpp @@ -1,6 +1,6 @@ #include "fdbmonitor.h" -#include +#include #include #include @@ -125,30 +125,27 @@ void testPathOps() { testPathFunction2("parentDirectory", parentDirectory, "foo/./../foo2/./bar//", true, joinPath(cwd, "foo2/")); printf("%d errors.\n", errors); - assert_msg(errors == 0, "Path operation tests failed"); + assert(errors == 0); } void testEnvVarUtils() { // Ensure key-value extraction works const std::pair keyValuePair1{ "FOO", "BAR" }; - assert_msg(keyValuePair1 == EnvVarUtils::extractKeyAndValue("FOO=BAR"), "Failed to extract FOO=BAR"); + assert(keyValuePair1 == EnvVarUtils::extractKeyAndValue("FOO=BAR")); const std::pair keyValuePair2{ "x", "y" }; - assert_msg(keyValuePair2 == EnvVarUtils::extractKeyAndValue("x=y"), "Failed to extract x=y"); + assert(keyValuePair2 == EnvVarUtils::extractKeyAndValue("x=y")); const std::pair keyValuePair3{ "MALLOC_CONF", "prof:true,lg_prof_interval:30,prof_prefix:jeprof.out" }; - assert_msg(keyValuePair3 == - EnvVarUtils::extractKeyAndValue("MALLOC_CONF=prof:true,lg_prof_interval:30,prof_prefix:jeprof.out"), - "Failed to extract MALLOC_CONF"); + assert(keyValuePair3 == + EnvVarUtils::extractKeyAndValue("MALLOC_CONF=prof:true,lg_prof_interval:30,prof_prefix:jeprof.out")); // Ensure key-value validation passes for good inputs - assert_msg(EnvVarUtils::keyValueValid("FOO=BAR", "FOO=BAR"), "FOO=BAR should be valid"); - assert_msg(EnvVarUtils::keyValueValid("x=y", "FOO=BAR x=y"), "x=y should be valid in a list"); - assert_msg(EnvVarUtils::keyValueValid("MALLOC_CONF=prof:true,lg_prof_interval:30,prof_prefix:jeprof.out", - "MALLOC_CONF=prof:true,lg_prof_interval:30,prof_prefix:jeprof.out"), - "MALLOC_CONF should be valid"); - assert_msg(EnvVarUtils::keyValueValid("MALLOC_CONF=prof:true,lg_prof_interval:30,prof_prefix:jeprof.out", - "MALLOC_CONF=prof:true,lg_prof_interval:30,prof_prefix:jeprof.out FOO=BAR"), - "MALLOC_CONF should be valid in a list"); + assert(EnvVarUtils::keyValueValid("FOO=BAR", "FOO=BAR")); + assert(EnvVarUtils::keyValueValid("x=y", "FOO=BAR x=y")); + assert(EnvVarUtils::keyValueValid("MALLOC_CONF=prof:true,lg_prof_interval:30,prof_prefix:jeprof.out", + "MALLOC_CONF=prof:true,lg_prof_interval:30,prof_prefix:jeprof.out")); + assert(EnvVarUtils::keyValueValid("MALLOC_CONF=prof:true,lg_prof_interval:30,prof_prefix:jeprof.out", + "MALLOC_CONF=prof:true,lg_prof_interval:30,prof_prefix:jeprof.out FOO=BAR")); // Ensure key-value validation fails for bad inputs assert_msg(!EnvVarUtils::keyValueValid("", "FOO=BAR ="), "Key-Value can not be empty"); diff --git a/fdbrpc/SimExternalConnection.cpp b/fdbrpc/SimExternalConnection.cpp index 66a6b44e8b3..4594350a5ea 100644 --- a/fdbrpc/SimExternalConnection.cpp +++ b/fdbrpc/SimExternalConnection.cpp @@ -77,8 +77,6 @@ class SimExternalConnectionImpl { while (self->readBuffer.empty()) { readAvailable(self); if (self->readBuffer.empty()) { - // Give the external peer wall-clock time; simulated delay alone cannot wait for it. - // ast-grep-ignore: fdb-no-blocking-sleep-in-coroutine threadSleep(0.01); co_await delayJittered(0.1); } diff --git a/fdbserver/kvstore/FDBExecHelper.cpp b/fdbserver/kvstore/FDBExecHelper.cpp index c045e528983..ab705241915 100644 --- a/fdbserver/kvstore/FDBExecHelper.cpp +++ b/fdbserver/kvstore/FDBExecHelper.cpp @@ -233,8 +233,6 @@ Future spawnProcess(std::string path, // child process has not completed yet if (isSync || g_network->isSimulated()) { // synchronously sleep - // Synchronous and simulated execution must wait without advancing other actors. - // ast-grep-ignore: fdb-no-blocking-sleep-in-coroutine threadSleep(0.1); } else { // yield for other actors to run diff --git a/fdbserver/networktest.cpp b/fdbserver/networktest.cpp index da087bc87ab..d8f6473cbc6 100644 --- a/fdbserver/networktest.cpp +++ b/fdbserver/networktest.cpp @@ -548,8 +548,6 @@ struct P2PNetworkTest { } catch (Error& e) { printf("Server: handshake error %s\n", e.what()); } - // Keep the one-shot TLS benchmark quiescent on this thread after its handshake. - // ast-grep-ignore: fdb-no-blocking-sleep-in-coroutine threadSleep(11.0); co_return; } diff --git a/fdbserver/worker/worker.cpp b/fdbserver/worker/worker.cpp index 60a8a58965b..5dd0bf244a1 100644 --- a/fdbserver/worker/worker.cpp +++ b/fdbserver/worker/worker.cpp @@ -2087,8 +2087,6 @@ class WorkerServerCore { flushTraceFileVoid(); setProfilingEnabled(0); g_network->stop(); - // The network has stopped, so process suspension cannot depend on its event loop. - // ast-grep-ignore: fdb-no-blocking-sleep-in-coroutine threadSleep(req.waitForDuration); } if (rebootReq.checkData) { diff --git a/fdbserver/workloads/UnitTests.cpp b/fdbserver/workloads/UnitTests.cpp index 61fc5235097..1b3a90ee4fa 100644 --- a/fdbserver/workloads/UnitTests.cpp +++ b/fdbserver/workloads/UnitTests.cpp @@ -215,7 +215,7 @@ struct UnitTestWorkload : TestWorkload { .detail("Name", test->name) .detail("File", test->file) .detail("Line", test->line) - .detail("Rand", debugRandom()->randomInt(0, 100001)); + .detail("Rand", deterministicRandom()->randomInt(0, 100001)); Error result = success(); double start_now = now(); diff --git a/flow/UnitTestRunner.cpp b/flow/UnitTestRunner.cpp index 44ecaf2e879..a93e724e146 100644 --- a/flow/UnitTestRunner.cpp +++ b/flow/UnitTestRunner.cpp @@ -328,7 +328,7 @@ Future runTests(const UnitTestRunnerOptions& options, .detail("Name", test->name) .detail("File", test->file) .detail("Line", test->line) - .detail("Rand", debugRandom()->randomInt(0, 100001)); + .detail("Rand", deterministicRandom()->randomInt(0, 100001)); Error resultCode = success(); double startNow = now(); From 010472bdc605cd1281a577cd8f7c16bc577dc0d3 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sat, 19 Sep 2026 22:51:44 -0700 Subject: [PATCH 123/170] Guard Swift stream support in builds without Swift --- flow/include/flow/swift_stream_support.h | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/flow/include/flow/swift_stream_support.h b/flow/include/flow/swift_stream_support.h index db08d45df4c..ee594d998e6 100644 --- a/flow/include/flow/swift_stream_support.h +++ b/flow/include/flow/swift_stream_support.h @@ -21,6 +21,8 @@ #ifndef SWIFT_STREAM_SUPPORT_H #define SWIFT_STREAM_SUPPORT_H +#ifdef WITH_SWIFT + #include "swift.h" #include "flow.h" #include "unsafe_swift_compat.h" @@ -153,4 +155,6 @@ struct UNSAFE_SWIFT_CXX_IMMORTAL_REF SwiftContinuationSingleCallbackCInt : Singl } }; +#endif /* WITH_SWIFT */ + #endif From a98362a2fc286d653d03d7a4e10ff0f874ffab15 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sat, 19 Sep 2026 22:58:21 -0700 Subject: [PATCH 124/170] Remove redundant ErrorName trace details --- fdbserver/mocks3/MockS3Server.cpp | 5 +---- fdbserver/mocks3/MockS3ServerChaos.cpp | 5 +---- 2 files changed, 2 insertions(+), 8 deletions(-) diff --git a/fdbserver/mocks3/MockS3Server.cpp b/fdbserver/mocks3/MockS3Server.cpp index 9e559e3133b..15a245b98c3 100644 --- a/fdbserver/mocks3/MockS3Server.cpp +++ b/fdbserver/mocks3/MockS3Server.cpp @@ -1590,10 +1590,7 @@ Future registerMockS3Server_impl(std::string ip, std::string port) { .detail("Address", serverKey) .detail("TotalRegistered", registeredServers.size()); } catch (Error& e) { - TraceEvent(SevError, "MockS3ServerRegistrationFailed") - .error(e) - .detail("Address", serverKey) - .detail("ErrorName", e.name()); + TraceEvent(SevError, "MockS3ServerRegistrationFailed").error(e).detail("Address", serverKey); throw; } } diff --git a/fdbserver/mocks3/MockS3ServerChaos.cpp b/fdbserver/mocks3/MockS3ServerChaos.cpp index 64b07e6d0ab..62826783bfd 100644 --- a/fdbserver/mocks3/MockS3ServerChaos.cpp +++ b/fdbserver/mocks3/MockS3ServerChaos.cpp @@ -342,10 +342,7 @@ Future registerMockS3ChaosServer(std::string ip, std::string port) { .detail("TotalRegistered", registeredMockS3ChaosServers().size()); } catch (Error& e) { - TraceEvent(SevError, "MockS3ChaosServerRegistrationFailed") - .error(e) - .detail("Address", serverKey) - .detail("ErrorName", e.name()); + TraceEvent(SevError, "MockS3ChaosServerRegistrationFailed").error(e).detail("Address", serverKey); throw; } } From 65dd5bf73ef180f735cf635e28d23f09875bb8f9 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sat, 19 Sep 2026 23:18:34 -0700 Subject: [PATCH 125/170] Preserve zero-length client certificate chains --- flow/MkCertCli.cpp | 4 -- tests/CMakeLists.txt | 5 ++ tests/argument_parsing/test_mkcert.py | 82 +++++++++++++++++++++++++++ 3 files changed, 87 insertions(+), 4 deletions(-) create mode 100644 tests/argument_parsing/test_mkcert.py diff --git a/flow/MkCertCli.cpp b/flow/MkCertCli.cpp index b006dd9d71b..193eaeaadee 100644 --- a/flow/MkCertCli.cpp +++ b/flow/MkCertCli.cpp @@ -241,10 +241,6 @@ int main(int argc, char** argv) { case OPT_CLIENT_CHAIN_LEN: try { clientArgs.length = std::stoul(args.OptionArg()); - if (clientArgs.length == 0) { - fmt::print(stderr, "ERROR: Certificate chain length must be positive\n"); - return FDB_EXIT_ERROR; - } } catch (std::exception const& ex) { fmt::print(stderr, "ERROR: Invalid chain length ({})\n", ex.what()); return FDB_EXIT_ERROR; diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 17a46f04af2..1f2646575f6 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -498,6 +498,11 @@ if(WITH_PYTHON) COMMAND ${Python3_EXECUTABLE} ${CMAKE_SOURCE_DIR}/tests/argument_parsing/test_argument_parsing.py ${CMAKE_BINARY_DIR} ) set_tests_properties(command_line_argument_test PROPERTIES ENVIRONMENT "FDB_CLUSTER_FILE=${CMAKE_BINARY_DIR}/fdb.cluster") + add_test( + NAME mkcert_cli_test + COMMAND ${Python3_EXECUTABLE} ${CMAKE_SOURCE_DIR}/tests/argument_parsing/test_mkcert.py $ + ) + set_tests_properties(mkcert_cli_test PROPERTIES TIMEOUT 60) endif() verify_testing() diff --git a/tests/argument_parsing/test_mkcert.py b/tests/argument_parsing/test_mkcert.py new file mode 100644 index 00000000000..e160eba489b --- /dev/null +++ b/tests/argument_parsing/test_mkcert.py @@ -0,0 +1,82 @@ +#!/usr/bin/env python3 +# +# test_mkcert.py +# +# This source file is part of the FoundationDB open source project +# +# Copyright 2013-2026 Apple Inc. and the FoundationDB project authors +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +import argparse +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest + + +class MkcertTest(unittest.TestCase): + def setUp(self): + directory = tempfile.TemporaryDirectory() + self.addCleanup(directory.cleanup) + self.directory = Path(directory.name) + + def run_mkcert(self, *args): + return subprocess.run( + [str(self.binary), *args], + cwd=self.directory, + capture_output=True, + text=True, + timeout=20, + ) + + def test_zero_client_chain_clears_existing_credentials(self): + result = self.run_mkcert( + "--server-chain-length", "1", "--client-chain-length", "1" + ) + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + for side in ("server", "client"): + for suffix in ("cert", "key", "ca"): + self.assertGreater( + (self.directory / f"{side}_{suffix}.pem").stat().st_size, 0 + ) + + result = self.run_mkcert( + "--server-chain-length", "1", "--client-chain-length", "0" + ) + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + for suffix in ("cert", "key", "ca"): + self.assertEqual( + (self.directory / f"client_{suffix}.pem").read_bytes(), b"" + ) + self.assertGreater( + (self.directory / f"server_{suffix}.pem").stat().st_size, 0 + ) + + def test_zero_server_chain_is_rejected(self): + result = self.run_mkcert("--server-chain-length", "0") + self.assertNotEqual(result.returncode, 0) + self.assertIn("Certificate chain length must be positive", result.stderr) + self.assertEqual(list(self.directory.iterdir()), []) + + +if __name__ == "__main__": + parser = argparse.ArgumentParser( + description="Test mkcert certificate chain options." + ) + parser.add_argument("binary", type=Path, help="Path to the mkcert executable") + args = parser.parse_args() + MkcertTest.binary = args.binary.resolve() + unittest.main(argv=[sys.argv[0]]) From a138d47a5bd47474416bbfc9f95d250b3103c4ef Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sat, 19 Sep 2026 23:22:42 -0700 Subject: [PATCH 126/170] Remove mkcert CLI regression test --- tests/CMakeLists.txt | 5 -- tests/argument_parsing/test_mkcert.py | 82 --------------------------- 2 files changed, 87 deletions(-) delete mode 100644 tests/argument_parsing/test_mkcert.py diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 1f2646575f6..17a46f04af2 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -498,11 +498,6 @@ if(WITH_PYTHON) COMMAND ${Python3_EXECUTABLE} ${CMAKE_SOURCE_DIR}/tests/argument_parsing/test_argument_parsing.py ${CMAKE_BINARY_DIR} ) set_tests_properties(command_line_argument_test PROPERTIES ENVIRONMENT "FDB_CLUSTER_FILE=${CMAKE_BINARY_DIR}/fdb.cluster") - add_test( - NAME mkcert_cli_test - COMMAND ${Python3_EXECUTABLE} ${CMAKE_SOURCE_DIR}/tests/argument_parsing/test_mkcert.py $ - ) - set_tests_properties(mkcert_cli_test PROPERTIES TIMEOUT 60) endif() verify_testing() diff --git a/tests/argument_parsing/test_mkcert.py b/tests/argument_parsing/test_mkcert.py deleted file mode 100644 index e160eba489b..00000000000 --- a/tests/argument_parsing/test_mkcert.py +++ /dev/null @@ -1,82 +0,0 @@ -#!/usr/bin/env python3 -# -# test_mkcert.py -# -# This source file is part of the FoundationDB open source project -# -# Copyright 2013-2026 Apple Inc. and the FoundationDB project authors -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -import argparse -from pathlib import Path -import subprocess -import sys -import tempfile -import unittest - - -class MkcertTest(unittest.TestCase): - def setUp(self): - directory = tempfile.TemporaryDirectory() - self.addCleanup(directory.cleanup) - self.directory = Path(directory.name) - - def run_mkcert(self, *args): - return subprocess.run( - [str(self.binary), *args], - cwd=self.directory, - capture_output=True, - text=True, - timeout=20, - ) - - def test_zero_client_chain_clears_existing_credentials(self): - result = self.run_mkcert( - "--server-chain-length", "1", "--client-chain-length", "1" - ) - self.assertEqual(result.returncode, 0, result.stdout + result.stderr) - for side in ("server", "client"): - for suffix in ("cert", "key", "ca"): - self.assertGreater( - (self.directory / f"{side}_{suffix}.pem").stat().st_size, 0 - ) - - result = self.run_mkcert( - "--server-chain-length", "1", "--client-chain-length", "0" - ) - self.assertEqual(result.returncode, 0, result.stdout + result.stderr) - for suffix in ("cert", "key", "ca"): - self.assertEqual( - (self.directory / f"client_{suffix}.pem").read_bytes(), b"" - ) - self.assertGreater( - (self.directory / f"server_{suffix}.pem").stat().st_size, 0 - ) - - def test_zero_server_chain_is_rejected(self): - result = self.run_mkcert("--server-chain-length", "0") - self.assertNotEqual(result.returncode, 0) - self.assertIn("Certificate chain length must be positive", result.stderr) - self.assertEqual(list(self.directory.iterdir()), []) - - -if __name__ == "__main__": - parser = argparse.ArgumentParser( - description="Test mkcert certificate chain options." - ) - parser.add_argument("binary", type=Path, help="Path to the mkcert executable") - args = parser.parse_args() - MkcertTest.binary = args.binary.resolve() - unittest.main(argv=[sys.argv[0]]) From c4e0d0eee2a6e26c5d3bbeaa0eaf82e20d818c98 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sat, 19 Sep 2026 23:46:28 -0700 Subject: [PATCH 127/170] Move successful TLS peer verification traces to debug --- flow/TLSConfig.cpp | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/flow/TLSConfig.cpp b/flow/TLSConfig.cpp index d8f1aed3a37..b904a4a8008 100644 --- a/flow/TLSConfig.cpp +++ b/flow/TLSConfig.cpp @@ -1007,8 +1007,7 @@ bool TLSPolicy::verify_peer(bool preverified, X509_STORE_CTX* store_ctx, const N .detail("Rule", rule.toString()); } } else { - TraceEvent(SevInfo, "TLSPolicySuccess") - .suppressFor(1.0) + TraceEvent(SevDebug, "TLSPolicySuccess") .detail("PeerAddress", peerAddress) .detail("Reason", verifier.getSuccessReason()); } From ec5bc59469aae212d1289ebabfab9fb446359d08 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 20 Sep 2026 14:03:26 -0700 Subject: [PATCH 128/170] Preserve committed CDC progress from quiet log replicas --- fdbserver/logsystem/LogSystemPeekCursor.cpp | 107 +++++++++++++++++++- 1 file changed, 105 insertions(+), 2 deletions(-) diff --git a/fdbserver/logsystem/LogSystemPeekCursor.cpp b/fdbserver/logsystem/LogSystemPeekCursor.cpp index e5db866d4cc..46dc58d5aaf 100644 --- a/fdbserver/logsystem/LogSystemPeekCursor.cpp +++ b/fdbserver/logsystem/LogSystemPeekCursor.cpp @@ -1016,7 +1016,12 @@ const LogMessageVersion& MergedPeekCursor::version() const { } Version MergedPeekCursor::getMinKnownCommittedVersion() const { - return serverCursors[currentCursor]->getMinKnownCommittedVersion(); + // A committed-version certificate is global, even when its replica has no next tagged message. + Version minKnownCommittedVersion = 0; + for (const auto& cursor : serverCursors) { + minKnownCommittedVersion = std::max(minKnownCommittedVersion, cursor->getMinKnownCommittedVersion()); + } + return minKnownCommittedVersion; } Version MergedPeekCursor::getMaxKnownVersion() const { @@ -1386,7 +1391,14 @@ const LogMessageVersion& SetPeekCursor::version() const { } Version SetPeekCursor::getMinKnownCommittedVersion() const { - return serverCursors[currentSet][currentCursor]->getMinKnownCommittedVersion(); + // Empty replies can certify progress independently of the current payload source. + Version minKnownCommittedVersion = 0; + for (const auto& cursors : serverCursors) { + for (const auto& cursor : cursors) { + minKnownCommittedVersion = std::max(minKnownCommittedVersion, cursor->getMinKnownCommittedVersion()); + } + } + return minKnownCommittedVersion; } Version SetPeekCursor::getMaxKnownVersion() const { @@ -1665,6 +1677,97 @@ TEST_CASE("/NativeCDC/ReplayPeekReplyAccounting") { return Void(); } +TEST_CASE("/NativeCDC/MergedPeekEmptyCommittedFrontier") { + auto makeServerCursor = []() { + return makeReference( + Reference>>(), Tag(tagLocalityCDC, 0), 0, 1000, false, false); + }; + std::vector> servers{ makeServerCursor(), makeServerCursor() }; + auto merged = makeReference( + servers, LogMessageVersion(0), 1, 2, Optional(), Reference(), 0); + ASSERT_EQ(merged->getMinKnownCommittedVersion(), 0); + + TLogPeekReply reply; + reply.end = 101; + reply.maxKnownVersion = 500; + reply.minKnownCommittedVersion = 100; + updateCursorWithReply(servers[1].getPtr(), reply); + merged->calcHasMessage(); + ASSERT(!merged->hasMessage()); + ASSERT_EQ(merged->version().version, 101); + ASSERT_EQ(merged->currentCursor, 0); + ASSERT_EQ(servers[0]->getMinKnownCommittedVersion(), 0); + ASSERT_EQ(merged->getMaxKnownVersion(), 500); + ASSERT_EQ(merged->getMinKnownCommittedVersion(), 100); + + // Another replica's stronger certificate does not advance the tagged read frontier. + reply.end = 151; + reply.maxKnownVersion = 800; + reply.minKnownCommittedVersion = 150; + updateCursorWithReply(servers[0].getPtr(), reply); + merged->calcHasMessage(); + ASSERT(!merged->hasMessage()); + ASSERT_EQ(merged->version().version, 101); + ASSERT_EQ(merged->currentCursor, 0); + ASSERT_EQ(merged->getMaxKnownVersion(), 800); + ASSERT_EQ(merged->getMinKnownCommittedVersion(), 150); + return Void(); +} + +TEST_CASE("/NativeCDC/SetPeekEmptyCommittedFrontier") { + std::vector> logSets; + std::vector>> servers(2); + for (int i = 0; i < servers.size(); ++i) { + auto logSet = makeReference(); + logSet->logServers.resize(2); + logSet->tLogReplicationFactor = 1; + logSet->tLogPolicy = makeReference(); + logSet->tLogLocalities.resize(2); + logSet->updateLocalitySet(logSet->tLogLocalities); + logSets.push_back(logSet); + for (int j = 0; j < 2; ++j) { + servers[i].push_back( + makeReference(Reference>>(), + Tag(tagLocalityCDC, 0), + 0, + 1000, + false, + false)); + } + } + auto merged = + makeReference(logSets, servers, LogMessageVersion(0), 0, 1, Optional(), true); + ASSERT_EQ(merged->getMinKnownCommittedVersion(), 0); + + TLogPeekReply reply; + reply.end = 101; + reply.maxKnownVersion = 500; + reply.minKnownCommittedVersion = 100; + updateCursorWithReply(servers[0][1].getPtr(), reply); + merged->calcHasMessage(); + ASSERT(!merged->hasMessage()); + ASSERT_EQ(merged->version().version, 101); + ASSERT_EQ(merged->currentSet, 0); + ASSERT_EQ(merged->currentCursor, 0); + ASSERT_EQ(servers[0][0]->getMinKnownCommittedVersion(), 0); + ASSERT_EQ(merged->getMaxKnownVersion(), 500); + ASSERT_EQ(merged->getMinKnownCommittedVersion(), 100); + + // A fallback set can know more is committed without changing the selected set's read frontier. + reply.end = 201; + reply.maxKnownVersion = 800; + reply.minKnownCommittedVersion = 200; + updateCursorWithReply(servers[1][1].getPtr(), reply); + merged->calcHasMessage(); + ASSERT(!merged->hasMessage()); + ASSERT_EQ(merged->version().version, 101); + ASSERT_EQ(merged->currentSet, 0); + ASSERT_EQ(merged->currentCursor, 0); + ASSERT_EQ(merged->getMaxKnownVersion(), 800); + ASSERT_EQ(merged->getMinKnownCommittedVersion(), 200); + return Void(); +} + TEST_CASE("/NativeCDC/ReplayPeekCommittedEpochBoundary") { auto makeServerCursor = [](Version committedVersion) { auto cursor = makeReference( From 8453fb935a33f94509505fae5d49c45ee89fed5c Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 20 Sep 2026 14:08:54 -0700 Subject: [PATCH 129/170] Release partial CDC buffer reservations before retrying --- design/cdc.md | 5 +- fdbserver/cdcproxy/CDCProxy.cpp | 33 ++--- .../include/fdbserver/cdcproxy/CDCProxyTest.h | 75 ++++++++++ fdbserver/workloads/CMakeLists.txt | 1 + fdbserver/workloads/NativeCdcEndToEnd.cpp | 128 ++++++++++++++++++ tests/CMakeLists.txt | 1 + tests/fast/NativeCdcBufferContention.toml | 38 ++++++ 7 files changed, 264 insertions(+), 17 deletions(-) create mode 100644 fdbserver/cdcproxy/include/fdbserver/cdcproxy/CDCProxyTest.h create mode 100644 tests/fast/NativeCdcBufferContention.toml diff --git a/design/cdc.md b/design/cdc.md index b98f3f8663d..d5c98f2aab8 100644 --- a/design/cdc.md +++ b/design/cdc.md @@ -533,7 +533,10 @@ and cap each reply at the smaller of `MAXIMUM_PEEK_BYTES` and budget plus one reply-sized materialization window before issuing a peek. It marks these delivery cursors with the same per-reply limit; recovery cursors remain uncapped so that transaction-system replay is not constrained -by a delivery memory knob. The pass retains the aggregate raw reservation +by a delivery memory knob. If materialization needs a larger window, the reader +releases its cursor and reservation before retrying with the full required +reservation. Competing readers cannot hold partial reservations while waiting +for each other to release capacity. The pass retains the aggregate raw reservation while filtering and copying, then releases it and transfers only accepted filtered bytes to the stream buffers. Acknowledgement or stream removal releases those retained permits. The usable retained-batch capacity is diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index 92cf4ec3531..0e905b0b1a7 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -32,6 +32,7 @@ #include "fdbclient/SystemData.h" #include "NativeCdcInternal.h" #include "fdbserver/cdcproxy/CDCProxy.h" +#include "fdbserver/cdcproxy/CDCProxyTest.h" #include "fdbserver/core/Knobs.h" #include "fdbserver/core/LogProtocolMessage.h" #include "fdbserver/core/OTELSpanContextMessage.h" @@ -178,6 +179,7 @@ struct CDCBufferedBatch { struct CDCBufferedTag : ReferenceCounted { Tag tag; bool active = true; + int64_t nextPassReservation = 0; std::set streamIds; AsyncTrigger refresh; AsyncTrigger stopped; @@ -1305,22 +1307,17 @@ Future CDCProxy::materializeBufferSelection(ReferenceonChange(), - tag->stopped.onTrigger(), - tag->refresh.onTrigger()); - if (exactCapacity.index() == 1 || exactCapacity.index() == 3) { - co_return CDCBufferTagPassResult::RETRY; - } - if (exactCapacity.index() == 2) { - co_return CDCBufferTagPassResult::STOP; + if (auto test = CDCProxyMaterializationTest::get()) { + // Poll shared simulation state without running callbacks in another simulated process. + while (test->holdExpansion(id, tag->tag, reservation.remaining)) { + co_await delay(0.01); + } } - reservation.remaining += additionalBytes; - recordBufferUsage(); + // Two readers can exhaust the budget with initial reservations and then both wait for an expansion. + // Drop this cursor and reservation before reacquiring the full amount in one request. + tag->nextPassReservation = rawPeekReservation + selection.selectedBytes; + ASSERT_LE(tag->nextPassReservation, bufferLimit); + co_return CDCBufferTagPassResult::RETRY; } if (!tag->active) { co_return CDCBufferTagPassResult::STOP; @@ -1358,6 +1355,7 @@ Future CDCProxy::materializeBufferSelection(ReferencenextPassReservation = 0; advanceTagBufferedThrough(tag, throughVersion, selection.selectedStreamIds); // Every raw cursor arena is covered by rawPeekReservation only for this pass. Reopen from the shared minimum // after releasing it so no cursor response remains live outside the proxy memory budget. @@ -1413,7 +1411,7 @@ Future CDCProxy::bufferTagCursor(ReferencenextPassReservation); if (prefetch && (bufferLock.waiters() != 0 || bufferLock.available() < passReservation)) { co_return CDCBufferTagPassResult::RETRY; } @@ -1939,6 +1937,9 @@ Future CDCProxy::consumeReply(Reference stre auto buffered = co_await race(waitForBufferedVersion(stream, begin), delay(SERVER_KNOBS->CDC_PROXY_CONSUME_POLL_TIMEOUT)); if (buffered.index() == 1) { + if (auto test = CDCProxyMaterializationTest::get()) { + test->recordLeaseExpiry(id); + } CODE_PROBE(true, "CDC proxy expires an idle consume lease"); CDCConsumeReply reply; reply.lastConsumedVersion = cursor.lastConsumedVersion; diff --git a/fdbserver/cdcproxy/include/fdbserver/cdcproxy/CDCProxyTest.h b/fdbserver/cdcproxy/include/fdbserver/cdcproxy/CDCProxyTest.h new file mode 100644 index 00000000000..1fc55f9d6f0 --- /dev/null +++ b/fdbserver/cdcproxy/include/fdbserver/cdcproxy/CDCProxyTest.h @@ -0,0 +1,75 @@ +/* + * CDCProxyTest.h + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#pragma once + +#include +#include "fdbclient/FDBTypes.h" +#include "flow/flow.h" + +// Simulation-only barrier at the point where two real tag readers need to expand their reservations. +class CDCProxyMaterializationTest : public ReferenceCounted { + UID owner; + std::map reservations; + bool released = false; + int expiredLeases = 0; + inline static Reference installed; + +public: + CDCProxyMaterializationTest(UID owner, Tag first, Tag second) : owner(owner) { + ASSERT_NE(first, second); + reservations.emplace(first, 0); + reservations.emplace(second, 0); + } + + static Reference get() { + return g_network->isSimulated() ? installed : Reference(); + } + static void install(Reference test) { + ASSERT(g_network->isSimulated()); + ASSERT(!installed); + installed = test; + } + static void uninstall() { + if (installed) { + installed->released = true; + installed.clear(); + } + } + bool holdExpansion(UID proxy, Tag tag, int64_t reserved) { + if (proxy != owner || released || !reservations.contains(tag)) { + return false; + } + reservations.at(tag) = reserved; + return true; + } + bool bothReadersHeld() const { return reservations.begin()->second > 0 && reservations.rbegin()->second > 0; } + int64_t heldBytes() const { return reservations.begin()->second + reservations.rbegin()->second; } + void release() { + ASSERT(bothReadersHeld()); + released = true; + } + void recordLeaseExpiry(UID proxy) { + if (proxy == owner) { + ++expiredLeases; + } + } + int leaseExpiries() const { return expiredLeases; } +}; diff --git a/fdbserver/workloads/CMakeLists.txt b/fdbserver/workloads/CMakeLists.txt index aadaf00b858..f88415501a2 100644 --- a/fdbserver/workloads/CMakeLists.txt +++ b/fdbserver/workloads/CMakeLists.txt @@ -12,6 +12,7 @@ configure_fdbserver_common_includes(fdbserver_workloads) target_include_directories(fdbserver_workloads PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}) target_link_libraries(fdbserver_workloads PRIVATE + fdbserver_cdcproxy fdbserver_consistencyscan fdbserver_core fdbserver_checkpoint diff --git a/fdbserver/workloads/NativeCdcEndToEnd.cpp b/fdbserver/workloads/NativeCdcEndToEnd.cpp index 8ada52bcf78..d78faed7f4f 100644 --- a/fdbserver/workloads/NativeCdcEndToEnd.cpp +++ b/fdbserver/workloads/NativeCdcEndToEnd.cpp @@ -30,11 +30,14 @@ #include "fdbclient/DatabaseContext.h" #include "fdbclient/NativeCdc.h" #include "fdbclient/SystemData.h" +#include "fdbserver/cdcproxy/CDCProxyTest.h" +#include "fdbserver/core/Knobs.h" #include "fdbserver/core/RecoveryState.h" #include "fdbserver/core/ServerDBInfo.h" #include "fdbserver/tester/workloads.h" #include "fdbrpc/simulator.h" #include "flow/DeterministicRandom.h" +#include "flow/ScopeExit.h" // Exercises native CDC by registering overlapping streams, writing mutations, consuming and acknowledging them, // and checking delivery, retention, assignment publication, failure recovery, and drain behavior. Test options @@ -72,6 +75,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { bool testTagOwnership; bool injectUndeliveredProxyHalt; bool testMemoryBound; + bool testBufferContention; bool testReplyChunking; bool testMultipleRanges; bool testOversizedPeek; @@ -247,6 +251,120 @@ class NativeCdcEndToEndWorkload : public TestWorkload { } } + Future initializeBufferContentionStreams(Database cx) { + for (int i = 0; i < 2; ++i) { + co_await addStream(cx, KeyRange(KeyRangeRef(keyForIndex(i), keyForIndex(i + 1)))); + } + } + + Future consumeContendedBuffer(int index, + Version through, + Reference> delivered, + Future releaseAcknowledgements) { + auto& stream = streams[index]; + while (stream.consumer->position().lastConsumedVersion < through) { + const Version previous = stream.consumer->position().lastConsumedVersion; + const CDCConsumeReply reply = co_await stream.consumer->consume(); + for (const auto& versioned : reply.mutations) { + ASSERT_GT(versioned.version, previous); + ASSERT_LE(versioned.version, reply.lastConsumedVersion); + ASSERT_EQ(versioned.mutations.size(), stream.expected.size()); + for (const auto& mutation : versioned.mutations) { + ASSERT_EQ(mutation.type, MutationRef::SetValue); + auto found = stream.expected.find(std::make_pair(Key(mutation.param1), Value(mutation.param2))); + ASSERT(found != stream.expected.end()); + ASSERT_LE(versioned.version, found->second.committedVersion); + ASSERT(found->second.observedVersions.insert(versioned.version).second); + } + } + } + for (const auto& [value, expected] : stream.expected) { + ASSERT(expected.observedVersions.contains(expected.committedVersion)); + } + delivered->set(delivered->get() + 1); + co_await releaseAcknowledgements; + co_await stream.consumer->acknowledge(); + } + + Future validateBufferContention(Database cx) { + const NativeCdcStatus metadata = co_await timeoutError(getNativeCdcStatus(cx), operationTimeout); + ASSERT(metadata.metadataComplete); + ASSERT_EQ(metadata.streams.size(), 2); + const auto& first = metadata.streams[0]; + const auto& second = metadata.streams[1]; + ASSERT(first.owner.present()); + ASSERT_EQ(first.owner, second.owner); + ASSERT_EQ(first.tags.size(), 1); + ASSERT_EQ(second.tags.size(), 1); + ASSERT_NE(first.tags.front(), second.tags.front()); + const UID owner = first.owner.get(); + std::vector committed; + for (int index = 0; index < 2; ++index) { + std::vector> values; + // These small mutations fit one raw reply but need more space after materialization. + for (int i = 0; i < 12; ++i) { + const Key key = streams[index].keys.begin.withSuffix(StringRef(format("/%02d", i))); + values.emplace_back(key, Value(StringRef(std::string(32, 'x')))); + } + committed.push_back(co_await writeValues(cx, values)); + recordExpectedWrites(values, committed.back()); + } + auto barrier = makeReference(owner, first.tags.front(), second.tags.front()); + CDCProxyMaterializationTest::install(barrier); + ScopeExit removeBarrier([] { CDCProxyMaterializationTest::uninstall(); }); + Promise releaseAcknowledgements; + auto delivered = makeReference>(0); + std::vector> consumers{ + consumeContendedBuffer(0, committed[0], delivered, releaseAcknowledgements.getFuture()), + consumeContendedBuffer(1, committed[1], delivered, releaseAcknowledgements.getFuture()) + }; + const double deadline = now() + operationTimeout; + while (!barrier->bothReadersHeld()) { + for (const auto& consumer : consumers) { + if (consumer.isReady()) { + consumer.get(); + ASSERT(false); + } + } + ASSERT_LT(now(), deadline); + co_await delay(0.01); + } + auto status = co_await timeoutError(getAssignedProxyStatus(cx, streams.front().consumer->position().streamId), + operationTimeout); + ASSERT_EQ(status.first.id(), owner); + ASSERT_EQ(status.second.activePermits, barrier->heldBytes()); + ASSERT_EQ(status.second.activePermits, status.second.bufferLimit); + ASSERT_EQ(status.second.bufferedBytes, 0); + ASSERT_EQ(delivered->get(), 0); + TraceEvent("NativeCdcBufferContendedReaders") + .detail("ActivePermits", status.second.activePermits) + .detail("BufferLimit", status.second.bufferLimit); + const double releasedAt = now(); + barrier->release(); + while (delivered->get() < 2) { + co_await timeoutError(delivered->onChange(), std::max(0.0, releasedAt + operationTimeout - now())); + } + ASSERT_EQ(barrier->leaseExpiries(), 0); + ASSERT_LT(now() - releasedAt, SERVER_KNOBS->CDC_PROXY_CONSUME_POLL_TIMEOUT); + status = co_await timeoutError(getAssignedProxyStatus(cx, streams.front().consumer->position().streamId), + operationTimeout); + ASSERT_EQ(status.first.id(), owner); + ASSERT_GT(status.second.bufferedBytes, 0); + ASSERT_LE(status.second.bufferedBytes, status.second.activePermits); + ASSERT_LE(status.second.activePermits, status.second.bufferLimit); + ASSERT_LE(status.second.peakActivePermits, status.second.bufferLimit); + releaseAcknowledgements.send(Void()); + co_await timeoutError(waitForAll(consumers), operationTimeout); + ASSERT_EQ(barrier->leaseExpiries(), 0); + CODE_PROBE(true, + "Native CDC expanded reservations progress on two tags without acknowledgements or lease expiry"); + for (const auto& stream : streams) { + co_await removeNativeCdcStreamClient(cx, stream.name); + } + streams.clear(); + co_await timeoutError(waitForRetiredTagCleanup(cx), operationTimeout); + } + Future initializeOversizedPeekStreams(Database cx) { ASSERT_GE(keyCount, 4); co_await addStream(cx, KeyRange(KeyRangeRef(keyForIndex(0), keyForIndex(2)))); @@ -2102,6 +2220,10 @@ class NativeCdcEndToEndWorkload : public TestWorkload { } Future run(Database cx) { + if (testBufferContention) { + co_await validateBufferContention(cx); + co_return; + } if (testMultipleRanges) { co_await validateMultipleRanges(cx); co_return; @@ -2204,6 +2326,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { testTagOwnership = getOption(options, "testTagOwnership"_sr, false); injectUndeliveredProxyHalt = getOption(options, "injectUndeliveredProxyHalt"_sr, false); testMemoryBound = getOption(options, "testMemoryBound"_sr, false); + testBufferContention = getOption(options, "testBufferContention"_sr, false); testReplyChunking = getOption(options, "testReplyChunking"_sr, false); testMultipleRanges = getOption(options, "testMultipleRanges"_sr, false); testOversizedPeek = getOption(options, "testOversizedPeek"_sr, false); @@ -2234,6 +2357,8 @@ class NativeCdcEndToEndWorkload : public TestWorkload { ASSERT(!(testOversizedPeek && testDurableAckScan)); ASSERT(!(testRetiredSharedTagSnapshot && testRetiredRecovery)); ASSERT(!blockRetiredPopWithLiveStream || testRetiredRecovery); + ASSERT(!testBufferContention || (initialStreamCount == 2 && !prepareRestartDrain && !drainAfterRestart && + !testMultipleRanges && !testRetiredSharedTagSnapshot && !testMemoryBound)); } // RandomRangeLock can outlive this bounded CDC workload and mask its progress check. @@ -2255,6 +2380,9 @@ class NativeCdcEndToEndWorkload : public TestWorkload { if (testRetiredSharedTagSnapshot) { return Void(); } + if (testBufferContention) { + return initializeBufferContentionStreams(cx); + } if (testOversizedPeek) { return initializeOversizedPeekStreams(cx); } diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 17a46f04af2..91a19bb42ed 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -230,6 +230,7 @@ if(WITH_PYTHON) add_fdb_test(TEST_FILES fast/NativeCdcSharedTag.toml) add_fdb_test(TEST_FILES fast/NativeCdcRetiredSharedTagSnapshot.toml) add_fdb_test(TEST_FILES fast/NativeCdcMemoryBound.toml) + add_fdb_test(TEST_FILES fast/NativeCdcBufferContention.toml) add_fdb_test(TEST_FILES fast/NativeCdcReplyChunking.toml) add_fdb_test(TEST_FILES fast/NativeCdcMultipleRanges.toml) add_fdb_test(TEST_FILES fast/NativeCdcLargeVersionBatching.toml) diff --git a/tests/fast/NativeCdcBufferContention.toml b/tests/fast/NativeCdcBufferContention.toml new file mode 100644 index 00000000000..dfea7f93ef2 --- /dev/null +++ b/tests/fast/NativeCdcBufferContention.toml @@ -0,0 +1,38 @@ +[configuration] +config = 'single logs=1 commit_proxies=1 grv_proxies=1' +singleRegion = true +datacenters = 1 +machineCount = 10 +statelessProcessClassesPerDC = 4 +buggify = false +faultInjection = false + +[[knobs]] +enable_native_cdc = true +native_cdc_tag_count = 2 +cdc_proxy_buffer_bytes = 4608 +maximum_peek_bytes = 1152 +cdc_proxy_consume_reply_bytes = 4096 +# Progress must precede any lease expiry or periodic pop that could free capacity. +cdc_proxy_consume_poll_timeout = 600.0 +cdc_proxy_pop_scan_interval = 600.0 + +[[test]] +testTitle = 'NativeCdcBufferContention' +useDB = true +runFailureWorkloads = false +runConsistencyCheck = false +waitForQuiescenceEnd = false +connectionFailuresDisableDuration = 1000000 +timeout = 600 + + [[test.workload]] + testName = 'NativeCdcEndToEnd' + initialStreamCount = 2 + minStreamCount = 2 + maxStreamCount = 2 + keyCount = 2 + writesPerRound = 1 + rounds = 0 + testBufferContention = true + operationTimeout = 20.0 From 62c4f4c774218ef8c8a588858231153ddf3b9032 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 20 Sep 2026 14:09:46 -0700 Subject: [PATCH 130/170] Place new native CDC streams using producer throughput --- design/cdc.md | 43 +- fdbclient/NativeCdc.cpp | 433 +------------ fdbclient/NativeCdcInternal.h | 13 +- fdbclient/SystemData.cpp | 38 ++ fdbclient/include/fdbclient/SystemData.h | 15 + fdbserver/cdcproxy/CDCProxy.cpp | 1 + .../clustercontroller/ClusterController.cpp | 1 + fdbserver/core/CMakeLists.txt | 1 + fdbserver/core/NativeCdcMetadata.cpp | 578 ++++++++++++++++++ fdbserver/core/ServerKnobs.cpp | 8 + fdbserver/core/include/fdbserver/core/Knobs.h | 8 + .../fdbserver/core/NativeCdcMetadata.h | 62 ++ fdbserver/datadistributor/CMakeLists.txt | 1 + .../datadistributor/DataDistribution.cpp | 3 + .../datadistributor/NativeCdcBalancer.cpp | 397 ++++++++++++ fdbserver/datadistributor/NativeCdcBalancer.h | 32 + .../workloads/NativeCdcInitialPlacement.cpp | 198 ++++++ fdbserver/workloads/UnitTests.cpp | 4 + tests/CMakeLists.txt | 1 + tests/fast/NativeCdcInitialPlacement.toml | 27 + 20 files changed, 1420 insertions(+), 444 deletions(-) create mode 100644 fdbserver/core/NativeCdcMetadata.cpp create mode 100644 fdbserver/core/include/fdbserver/core/NativeCdcMetadata.h create mode 100644 fdbserver/datadistributor/NativeCdcBalancer.cpp create mode 100644 fdbserver/datadistributor/NativeCdcBalancer.h create mode 100644 fdbserver/workloads/NativeCdcInitialPlacement.cpp create mode 100644 tests/fast/NativeCdcInitialPlacement.toml diff --git a/design/cdc.md b/design/cdc.md index d5c98f2aab8..86078d1cd03 100644 --- a/design/cdc.md +++ b/design/cdc.md @@ -374,6 +374,7 @@ than transaction state: | --- | --- | --- | | `\xff\x02/cdc/minVersion/` | `Version` | Earliest version that an active stream may still require. | | `\xff\x02/cdc/retiredTagPopVersion/` | `Version` | Final pop watermark required after a stream using a tag is removed. | +| `\xff\x02/cdc/tagLoad/` | assignment generation, sample version, expiry version, sampled write rate | Advisory producer-load estimate used for placement. | | `\xff\x02/cdc/tagOwner/` | `CDCStreamId` | Derived representative stream used to look up a current tag's proxy owner. | The initial `minVersion` is written with a versionstamp at stream @@ -397,9 +398,11 @@ Registration runs as a durable metadata transaction: same-name/same-range-set rule, even when admission is disabled. 3. For a new name, it validates the feature knob. 4. It allocates a new monotonically increasing `CDCStreamId`. -5. It selects a CDC tag using current active stream counts. The allocator uses - the least populated tag among `NATIVE_CDC_TAG_COUNT` tags (256 by default), - choosing the lowest tag ID on a tie. +5. It selects a CDC tag from `NATIVE_CDC_TAG_COUNT` tags (256 by default). + With fresh, complete write-load samples for the current assignment generation, + it chooses the least-loaded tag, breaking ties by stream count then tag ID. + Otherwise it uses the least populated tag, breaking ties by tag ID. An unused + tag has zero current load; an occupied tag without a sample has unknown load. 6. It records the stream name, canonical ranges, initial tag history entry, and versionstamped initial minimum version. 7. It records an available CDC proxy owner and signals assignment monitoring. @@ -428,6 +431,34 @@ validation also rejects stale representatives left by older metadata writers. The allocator's stream-count scan remains necessary, and ownership discovery still scans global metadata when the representative is absent or invalid. +### Throughput-aware initial placement + +When native CDC admission and `NATIVE_CDC_TAG_BALANCING_ENABLED` are enabled, +the data distributor samples producer writes using storage-server range metrics. +Both sampling and load-aware registration are advisory: existing streams retain +their original tags. The sampling knob defaults to enabled; native CDC admission +remains disabled by default. + +Registered ranges are divided into disjoint segments, and each tag is counted +once per segment so overlapping streams do not inflate a shared tag's load. +The model bounds projected coverage entries by +`2 * streamCount * totalRangeCount` before construction. The default +`NATIVE_CDC_TAG_MODEL_MAX_ENTRIES` budget is 2,000,000 entries; this is a +conservative entry bound, not an exact resident-memory bound. + +Sampling also limits streams, shards, request concurrency, and elapsed time. +The default interval is 30 seconds and sample lifetime is 90 seconds. Partial, +failed, expired, or previous-generation samples never imply zero load. +Publication revalidates both the assignment generation and data-distributor +lock. Registration, removal, and proxy ownership changes invalidate existing +comparisons. Registrations use stream counts until every occupied eligible tag +has a fresh sample from the current generation. + +`NativeCdcInitialPlacement` drives real producer writes, verifies a colder tag +wins despite having more streams, and checks delivery through the new stream. +Storage metrics remain write-cost estimates with their existing sampling and +range-clear attribution semantics; they do not measure exact tagged TLog bytes. + ### Metadata lifecycle example Assume a client registers stream name `orders` for range @@ -808,9 +839,9 @@ The design records tag history and proxy ownership in forms that support more complete load balancing, but the first implementation intentionally keeps policy simple. -* Tag selection is based on active stream counts, not observed byte or mutation - throughput. Data distribution could make equally counted tags very - different in cost. +* Producer-write metrics are estimates with the storage-metrics sampling window. + Missing or stale comparisons fall back to active stream counts. Initial + placement does not rebalance existing streams after their loads change. * Registration selects an available CDC proxy without balancing aggregate proxy throughput, buffer memory, lag, or number of active readers. * Assignment mutations use one coalescing change key that wakes a full durable diff --git a/fdbclient/NativeCdc.cpp b/fdbclient/NativeCdc.cpp index e14d7da6ca4..4d1d6598637 100644 --- a/fdbclient/NativeCdc.cpp +++ b/fdbclient/NativeCdc.cpp @@ -36,16 +36,6 @@ #include "flow/Trace.h" #include "flow/UnitTest.h" -namespace { - -using CDCTagId = uint16_t; - -constexpr uint32_t maxNativeCdcTagCount = static_cast(std::numeric_limits::max()) + 1; - -bool validNativeCdcTagCount(int tagCount) { - return tagCount > 0 && static_cast(tagCount) <= maxNativeCdcTagCount; -} - void validateNativeCdcEnabled(bool enabled) { if (!enabled) { CODE_PROBE(true, "Native CDC registration rejected while feature disabled"); @@ -53,49 +43,6 @@ void validateNativeCdcEnabled(bool enabled) { } } -class NativeCdcIdentifierAllocator { - bool sawStream = false; - CDCStreamId maxStreamId = 0; - std::unordered_map tagStreamCounts; - -public: - void observeStreamId(CDCStreamId streamId) { - sawStream = true; - maxStreamId = std::max(maxStreamId, streamId); - } - - void observeTag(Tag tag) { - ASSERT_WE_THINK(tag.locality == tagLocalityCDC); - ++tagStreamCounts[tag.id]; - } - - bool hasStreams(Tag tag) const { return tagStreamCounts.contains(tag.id); } - - std::pair allocate(int tagCount) const { - if (sawStream && maxStreamId == std::numeric_limits::max()) { - throw operation_failed(); - } - - const CDCStreamId streamId = sawStream ? maxStreamId + 1 : 1; - if (!validNativeCdcTagCount(tagCount)) { - throw invalid_option_value(); - } - uint32_t leastStreams = std::numeric_limits::max(); - CDCTagId selectedTagId = 0; - // TODO: Use data-distributor-observed per-tag write throughput to rebalance CDC tags, including - // migrating active streams with versioned tag-history assignments. - for (uint32_t tagId = 0; tagId < static_cast(tagCount); ++tagId) { - auto count = tagStreamCounts.find(static_cast(tagId)); - const uint32_t streamCount = count == tagStreamCounts.end() ? 0 : count->second; - if (streamCount < leastStreams) { - leastStreams = streamCount; - selectedTagId = static_cast(tagId); - } - } - return { streamId, Tag(tagLocalityCDC, selectedTagId) }; - } -}; - void normalizeNativeCdcStreamRanges(KeyRef const& name, std::vector& ranges) { if (name.empty() || ranges.empty() || ranges.size() > NATIVE_CDC_MAX_RANGES) { throw client_invalid_operation(); @@ -130,143 +77,11 @@ void normalizeNativeCdcStreamRanges(KeyRef const& name, std::vector& r } } -Future> getNativeCdcProxyAssignment(Transaction* tr, CDCStreamId streamId) { - RangeResult assignments = co_await tr->getRange(cdcProxyRangeFor(streamId), 2); - ASSERT_LE(assignments.size(), 1); - if (assignments.empty()) { - co_return Optional(); - } - const auto [assignedStreamId, proxyId] = decodeCDCProxyKey(assignments[0].key); - ASSERT_WE_THINK(assignedStreamId == streamId); - co_return proxyId; -} - -Future getNativeCdcCurrentTag(Transaction* tr, CDCStreamId streamId) { - // Tag-history keys sort by their big-endian assignment version, so the final - // key in this stream's prefix range contains its current tag. - RangeResult history = co_await tr->getRange(cdcTagHistoryRangeFor(streamId), 1, Snapshot::False, Reverse::True); - if (history.empty()) { - throw client_invalid_operation(); - } - co_return decodeCDCTagHistoryKey(history.front().key).tag; -} - -Future> getNativeCdcProxyAssignmentForTag(Transaction* tr, Tag targetTag) { - const Key ownerKey = cdcTagOwnerKeyFor(targetTag); - Optional indexedStream = co_await tr->get(ownerKey); - if (indexedStream.present()) { - const CDCStreamId streamId = decodeCDCTagOwnerValue(indexedStream.get()); - Future> activeStream = tr->get(cdcStreamKeyFor(streamId)); - RangeResult history = co_await tr->getRange(cdcTagHistoryRangeFor(streamId), 1, Snapshot::False, Reverse::True); - // Keep the await separate so GCC 13 does not evaluate history.front() before the short-circuit guards. - const Optional activeStreamValue = co_await activeStream; - // The index is derived: removal or retagging can invalidate its representative, and the per-stream - // assignment remains authoritative across proxy replacement, including by older metadata writers. - if (activeStreamValue.present() && !history.empty() && - decodeCDCTagHistoryKey(history.front().key).tag == targetTag) { - Optional proxyId = co_await getNativeCdcProxyAssignment(tr, streamId); - if (proxyId.present()) { - CODE_PROBE(true, "Native CDC resolves a shared tag owner from its persisted index"); - co_return proxyId; - } - } - CODE_PROBE(true, "Native CDC rebuilds a stale tag owner index"); - tr->clear(ownerKey); - } - - std::set activeStreamIds; - Key begin = cdcStreamKeys.begin; - while (begin < cdcStreamKeys.end) { - RangeResult streams = co_await tr->getRange(KeyRangeRef(begin, cdcStreamKeys.end), CLIENT_KNOBS->TOO_MANY); - for (const auto& stream : streams) { - activeStreamIds.insert(decodeCDCStreamKey(stream.key)); - } - if (!streams.more) { - break; - } - begin = keyAfter(streams.back().key); - } - - std::unordered_map currentTags; - begin = cdcTagHistoryKeys.begin; - while (begin < cdcTagHistoryKeys.end) { - RangeResult histories = - co_await tr->getRange(KeyRangeRef(begin, cdcTagHistoryKeys.end), CLIENT_KNOBS->TOO_MANY); - for (const auto& history : histories) { - const CDCTagHistoryEntry decoded = decodeCDCTagHistoryKey(history.key); - if (activeStreamIds.contains(decoded.streamId)) { - currentTags[decoded.streamId] = decoded.tag; - } - } - if (!histories.more) { - break; - } - begin = keyAfter(histories.back().key); - } - - for (const auto& [streamId, tag] : currentTags) { - if (tag == targetTag) { - Optional proxyId = co_await getNativeCdcProxyAssignment(tr, streamId); - if (proxyId.present()) { - tr->set(ownerKey, cdcTagOwnerValue(streamId)); - CODE_PROBE(true, "Native CDC reconstructs a missing tag owner index from active streams"); - co_return proxyId; - } - } - } - co_return Optional(); -} - -void signalNativeCdcProxyAssignmentChange(Transaction* tr) { - // Assignment updates are low-rate control-plane operations. A single - // coalescing signal lets the cluster controller rescan all durable owners. - tr->set(cdcProxyAssignmentChangeKey, - BinaryWriter::toValue(deterministicRandom()->randomUniqueID(), - IncludeVersion(ProtocolVersion::withNativeCdc()))); +bool nativeCdcNameMatchesStream(Optional const& currentId, CDCStreamId streamId) { + return currentId.present() && decodeCDCStreamNameValue(currentId.get()) == streamId; } -Future observeNativeCdcMetadata(Transaction* tr, NativeCdcIdentifierAllocator* allocator) { - Optional maxStreamId = co_await tr->get(cdcMaxStreamIdKey); - if (maxStreamId.present()) { - allocator->observeStreamId(decodeCDCMaxStreamIdValue(maxStreamId.get())); - } - - std::set activeStreamIds; - Key begin = cdcStreamKeys.begin; - while (begin < cdcStreamKeys.end) { - RangeResult streams = co_await tr->getRange(KeyRangeRef(begin, cdcStreamKeys.end), CLIENT_KNOBS->TOO_MANY); - for (const auto& kv : streams) { - const CDCStreamId streamId = decodeCDCStreamKey(kv.key); - activeStreamIds.insert(streamId); - allocator->observeStreamId(streamId); - } - if (!streams.more) { - break; - } - begin = keyAfter(streams.back().key); - } - - std::unordered_map currentTags; - begin = cdcTagHistoryKeys.begin; - while (begin < cdcTagHistoryKeys.end) { - RangeResult histories = - co_await tr->getRange(KeyRangeRef(begin, cdcTagHistoryKeys.end), CLIENT_KNOBS->TOO_MANY); - for (const auto& kv : histories) { - const CDCTagHistoryEntry history = decodeCDCTagHistoryKey(kv.key); - allocator->observeStreamId(history.streamId); - if (activeStreamIds.contains(history.streamId)) { - currentTags[history.streamId] = history.tag; - } - } - if (!histories.more) { - break; - } - begin = keyAfter(histories.back().key); - } - for (const auto& tagAssignment : currentTags) { - allocator->observeTag(tagAssignment.second); - } -} +namespace { bool retryNativeCdcProxyRequest(Error const& error) { return error.code() == error_code_wrong_shard_server || error.code() == error_code_broken_promise || @@ -375,10 +190,6 @@ Future getNativeCdcStreamProxy(Database cx, CDCStreamId strea } } -bool nativeCdcNameMatchesStream(Optional const& currentId, CDCStreamId streamId) { - return currentId.present() && decodeCDCStreamNameValue(currentId.get()) == streamId; -} - Future namedNativeCdcStreamStillExists(Database cx, Key name, CDCStreamId streamId) { Transaction tr(cx); while (true) { @@ -484,162 +295,6 @@ Future sampleNativeCdcProxy(CDCProxyInterface proxy, std:: } // namespace -Future registerNativeCdcStream(Database cx, Key name, std::vector ranges, UID proxyId) { - normalizeNativeCdcStreamRanges(name, ranges); - - Transaction tr(cx); - while (true) { - Error err; - try { - tr.setOption(FDBTransactionOptions::LOCK_AWARE); - tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); - - const Key nameKey = cdcStreamNameKeyFor(name); - Optional currentId = co_await tr.get(nameKey); - if (currentId.present()) { - const CDCStreamId streamId = decodeCDCStreamNameValue(currentId.get()); - Optional currentKeys = co_await tr.get(cdcStreamKeyFor(streamId)); - if (!currentKeys.present() || decodeCDCStreamKeysValue(currentKeys.get()) != ranges) { - throw client_invalid_operation(); - } - if (!(co_await getNativeCdcProxyAssignment(&tr, streamId)).present()) { - CODE_PROBE(true, "Native CDC registration restores missing stream owner", probe::decoration::rare); - const Tag tag = co_await getNativeCdcCurrentTag(&tr, streamId); - Optional sharedTagProxy = co_await getNativeCdcProxyAssignmentForTag(&tr, tag); - CODE_PROBE(sharedTagProxy.present(), - "Native CDC shared-tag streams use one owner", - probe::decoration::rare); - const UID selectedProxy = sharedTagProxy.present() ? sharedTagProxy.get() : proxyId; - tr.set(cdcProxyKeyFor(streamId, selectedProxy), Value()); - if (!sharedTagProxy.present()) { - tr.set(cdcTagOwnerKeyFor(tag), cdcTagOwnerValue(streamId)); - } - signalNativeCdcProxyAssignmentChange(&tr); - co_await tr.commit(); - } - co_return streamId; - } - - // Disabling CDC stops new admission, but existing registrations and - // owner repair must remain available so durable streams can drain. - const bool nativeCdcEnabled = cx->clientInfo->get().nativeCdcEnabled; - const int nativeCdcTagCount = cx->clientInfo->get().nativeCdcTagCount; - validateNativeCdcEnabled(nativeCdcEnabled); - NativeCdcIdentifierAllocator allocator; - co_await observeNativeCdcMetadata(&tr, &allocator); - const auto [streamId, tag] = allocator.allocate(nativeCdcTagCount); - // The read version is a conservative lower bound for tag routing. - // The versionstamped minimum below is the commit version, and stream - // initialization takes their maximum before exposing mutations. - const Version registrationVersion = co_await tr.getReadVersion(); - - tr.set(nameKey, cdcStreamNameValue(streamId)); - tr.set(cdcMaxStreamIdKey, cdcMaxStreamIdValue(streamId)); - tr.set(cdcStreamKeyFor(streamId), cdcStreamKeysValue(ranges)); - tr.set(cdcTagHistoryKeyFor(streamId, registrationVersion, tag), Value()); - tr.atomicOp( - cdcMinVersionKeyFor(streamId), cdcVersionstampedMinVersionValue(), MutationRef::SetVersionstampedValue); - Optional sharedTagProxy; - if (allocator.hasStreams(tag)) { - sharedTagProxy = co_await getNativeCdcProxyAssignmentForTag(&tr, tag); - } - const UID selectedProxy = sharedTagProxy.present() ? sharedTagProxy.get() : proxyId; - tr.set(cdcProxyKeyFor(streamId, selectedProxy), Value()); - if (!sharedTagProxy.present()) { - tr.set(cdcTagOwnerKeyFor(tag), cdcTagOwnerValue(streamId)); - } - signalNativeCdcProxyAssignmentChange(&tr); - co_await tr.commit(); - co_return streamId; - } catch (Error& e) { - err = e; - } - co_await tr.onError(err); - } -} - -Future removeNativeCdcStream(Database cx, Key name, CDCStreamId streamId, UID proxyId) { - if (name.empty() || streamId == 0) { - throw client_invalid_operation(); - } - - Transaction tr(cx); - while (true) { - Error err; - try { - tr.setOption(FDBTransactionOptions::LOCK_AWARE); - tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); - - const Key nameKey = cdcStreamNameKeyFor(name); - Optional currentId = co_await tr.get(nameKey); - if (!nativeCdcNameMatchesStream(currentId, streamId)) { - CODE_PROBE(currentId.present(), "Native CDC preserves a replacement stream during removal retry"); - if (currentId.present()) { - TraceEvent("NativeCdcRemovalPreservesReplacement") - .detail("RemovedStreamId", streamId) - .detail("ReplacementStreamId", decodeCDCStreamNameValue(currentId.get())); - } - co_return false; - } - - Optional assignedProxy = co_await getNativeCdcProxyAssignment(&tr, streamId); - if (!assignedProxy.present() || assignedProxy.get() != proxyId) { - CODE_PROBE(true, "Native CDC rejects removal through a stale owner"); - throw wrong_shard_server(); - } - - std::set removedTags; - const KeyRange historyRange = cdcTagHistoryRangeFor(streamId); - Key begin = historyRange.begin; - while (begin < historyRange.end) { - RangeResult history = - co_await tr.getRange(KeyRangeRef(begin, historyRange.end), CLIENT_KNOBS->TOO_MANY); - for (const auto& entry : history) { - removedTags.insert(decodeCDCTagHistoryKey(entry.key).tag); - } - if (!history.more) { - break; - } - begin = keyAfter(history.back().key); - } - - tr.clear(nameKey); - tr.clear(cdcStreamKeyFor(streamId)); - for (const Tag& tag : removedTags) { - const Key ownerKey = cdcTagOwnerKeyFor(tag); - Optional indexedStream = co_await tr.get(ownerKey); - if (indexedStream.present() && decodeCDCTagOwnerValue(indexedStream.get()) == streamId) { - tr.clear(ownerKey); - } - tr.set(cdcRetiredTagPopKeyFor(tag), Value()); - tr.atomicOp(cdcRetiredTagPopVersionKeyFor(tag), - cdcVersionstampedMinVersionValue(), - MutationRef::SetVersionstampedValue); - } - tr.clear(cdcTagHistoryRangeFor(streamId)); - tr.clear(cdcMinVersionKeyFor(streamId)); - tr.clear(cdcProxyRangeFor(streamId)); - if (assignedProxy.present()) { - signalNativeCdcProxyAssignmentChange(&tr); - } - co_await tr.commit(); - CODE_PROBE(!removedTags.empty(), "Native CDC removal records final tagged pop work"); - TraceEvent("NativeCdcStreamRemoved") - .detail("StreamId", streamId) - .detail("ProxyID", proxyId) - .detail("CommitVersion", tr.getCommittedVersion()) - .detail("RetiredTagCount", removedTags.size()); - co_return true; - } catch (Error& e) { - if (e.code() == error_code_wrong_shard_server) { - throw; - } - err = e; - } - co_await tr.onError(err); - } -} - Future> listNativeCdcStreams(Database cx) { Transaction tr(cx); while (true) { @@ -707,51 +362,6 @@ Future> listNativeCdcStreams(Database cx) { } } -Future reassignNativeCdcStreams(Database cx, UID oldProxyId, UID newProxyId) { - if (oldProxyId == newProxyId) { - co_return; - } - - Transaction tr(cx); - while (true) { - Error err; - try { - tr.setOption(FDBTransactionOptions::LOCK_AWARE); - tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); - tr.setOption(FDBTransactionOptions::PRIORITY_SYSTEM_IMMEDIATE); - - bool changed = false; - Key begin = cdcProxyKeys.begin; - while (begin < cdcProxyKeys.end) { - RangeResult assignments = - co_await tr.getRange(KeyRangeRef(begin, cdcProxyKeys.end), CLIENT_KNOBS->TOO_MANY); - for (const auto& assignment : assignments) { - const auto [streamId, proxyId] = decodeCDCProxyKey(assignment.key); - if (proxyId == oldProxyId) { - tr.clear(assignment.key); - tr.set(cdcProxyKeyFor(streamId, newProxyId), Value()); - changed = true; - } - } - if (!assignments.more) { - break; - } - begin = keyAfter(assignments.back().key); - } - - if (changed) { - CODE_PROBE(true, "Native CDC reassigns streams after proxy replacement"); - signalNativeCdcProxyAssignmentChange(&tr); - co_await tr.commit(); - } - co_return; - } catch (Error& e) { - err = e; - } - co_await tr.onError(err); - } -} - Future acknowledgeNativeCdcStream(Database cx, CDCStreamId streamId, Version consumedThrough, @@ -1203,43 +813,6 @@ TEST_CASE("/NativeCDC/InvalidRanges") { return Void(); } -TEST_CASE("/NativeCDC/LifecycleAllocation") { - ASSERT(!validNativeCdcTagCount(-1)); - ASSERT(!validNativeCdcTagCount(0)); - ASSERT(validNativeCdcTagCount(1)); - ASSERT(validNativeCdcTagCount(std::numeric_limits::max() + 1u)); - ASSERT(!validNativeCdcTagCount(std::numeric_limits::max() + 2u)); - - NativeCdcIdentifierAllocator allocator; - auto [initialId, initialTag] = allocator.allocate(CLIENT_KNOBS->NATIVE_CDC_TAG_COUNT); - ASSERT_EQ(initialId, 1); - ASSERT_EQ(initialTag, Tag(tagLocalityCDC, 0)); - - allocator.observeStreamId(9); - allocator.observeTag(initialTag); - allocator.observeTag(Tag(tagLocalityCDC, 2)); - auto [nextId, nextTag] = allocator.allocate(CLIENT_KNOBS->NATIVE_CDC_TAG_COUNT); - ASSERT_EQ(nextId, 10); - ASSERT_EQ(nextTag, Tag(tagLocalityCDC, 1)); - - NativeCdcIdentifierAllocator publishedPoolAllocator; - publishedPoolAllocator.observeTag(Tag(tagLocalityCDC, 0)); - // The cluster-controller-published pool is authoritative even when it differs from this process's knob. - auto [publishedPoolId, publishedPoolTag] = publishedPoolAllocator.allocate(1); - ASSERT_EQ(publishedPoolId, 1); - ASSERT_EQ(publishedPoolTag, Tag(tagLocalityCDC, 0)); - - NativeCdcIdentifierAllocator fullPoolAllocator; - for (uint32_t tagId = 0; tagId < static_cast(CLIENT_KNOBS->NATIVE_CDC_TAG_COUNT); ++tagId) { - fullPoolAllocator.observeTag(Tag(tagLocalityCDC, static_cast(tagId))); - } - auto [sharedId, sharedTag] = fullPoolAllocator.allocate(CLIENT_KNOBS->NATIVE_CDC_TAG_COUNT); - ASSERT_EQ(sharedId, 1); - ASSERT_EQ(sharedTag, Tag(tagLocalityCDC, 0)); - - return Void(); -} - TEST_CASE("/NativeCDC/ConsumerRewindsUnacknowledgedCursorOnProxyReplacement") { CDCCursor cursor(1, invalidVersion); Optional deliveryProxyId; diff --git a/fdbclient/NativeCdcInternal.h b/fdbclient/NativeCdcInternal.h index 3acd6b919f5..ec2a2c521de 100644 --- a/fdbclient/NativeCdcInternal.h +++ b/fdbclient/NativeCdcInternal.h @@ -24,15 +24,12 @@ #include "fdbclient/NativeCdc.h" -// Durable metadata operations used by CDC server roles. Registration is -// feature gated; drain and cleanup operations remain available for streams -// persisted before native CDC is disabled. -Future registerNativeCdcStream(Database cx, Key name, std::vector ranges, UID proxyId); -// Persists per-tag final-pop watermarks before removing stream metadata. -Future removeNativeCdcStream(Database cx, Key name, CDCStreamId streamId, UID proxyId); +// Shared admission and metadata identity checks for native CDC operations. +void validateNativeCdcEnabled(bool enabled); +void normalizeNativeCdcStreamRanges(KeyRef const& name, std::vector& ranges); +bool nativeCdcNameMatchesStream(Optional const& currentId, CDCStreamId streamId); + Future> listNativeCdcStreams(Database cx); -// Atomically moves any streams assigned to a failed proxy to its replacement. -Future reassignNativeCdcStreams(Database cx, UID oldProxyId, UID newProxyId); // Persists the exclusive unpopped watermark after consuming through a version. // knownAvailableThrough permits a consumer to acknowledge log data it has // already received before that version is visible at a transaction read version. diff --git a/fdbclient/SystemData.cpp b/fdbclient/SystemData.cpp index b838149bda5..6a1009db9e3 100644 --- a/fdbclient/SystemData.cpp +++ b/fdbclient/SystemData.cpp @@ -787,6 +787,7 @@ const KeyRangeRef cdcStreamNameKeys("\xff/cdc/name/"_sr, "\xff/cdc/name0"_sr); const KeyRef cdcMaxStreamIdKey = "\xff/cdc/maxStreamId"_sr; const KeyRangeRef cdcStreamKeys("\xff/cdc/keys/"_sr, "\xff/cdc/keys0"_sr); const KeyRangeRef cdcTagHistoryKeys("\xff/cdc/tagHistory/"_sr, "\xff/cdc/tagHistory0"_sr); +const KeyRangeRef cdcTagLoadKeys("\xff\x02/cdc/tagLoad/"_sr, "\xff\x02/cdc/tagLoad0"_sr); const KeyRangeRef cdcTagOwnerKeys("\xff\x02/cdc/tagOwner/"_sr, "\xff\x02/cdc/tagOwner0"_sr); const KeyRangeRef cdcMinVersionKeys("\xff\x02/cdc/minVersion/"_sr, "\xff\x02/cdc/minVersion0"_sr); const KeyRangeRef cdcRetiredTagPopKeys("\xff/cdc/retiredTagPop/"_sr, "\xff/cdc/retiredTagPop0"_sr); @@ -885,6 +886,34 @@ CDCTagHistoryEntry decodeCDCTagHistoryKey(KeyRef const& key) { return CDCTagHistoryEntry(streamId, bigEndian64(encodedVersion), tag); } +Key cdcTagLoadKeyFor(Tag tag) { + BinaryWriter wr(Unversioned()); + wr.serializeBytes(cdcTagLoadKeys.begin); + wr << tag; + return wr.toValue(); +} + +Tag decodeCDCTagLoadKey(KeyRef const& key) { + Tag tag; + BinaryReader reader(key.removePrefix(cdcTagLoadKeys.begin), Unversioned()); + reader >> tag; + return tag; +} + +Value cdcTagLoadValue(CDCTagLoadSample const& sample) { + BinaryWriter wr(IncludeVersion(ProtocolVersion::withNativeCdc())); + wr << sample.assignmentChange << sample.sampleVersion << sample.validThrough << sample.bytesWrittenPerKSecond; + return wr.toValue(); +} + +CDCTagLoadSample decodeCDCTagLoadValue(ValueRef const& value) { + CDCTagLoadSample sample; + BinaryReader reader(value, IncludeVersion()); + ASSERT_WE_THINK(reader.protocolVersion().hasNativeCdc()); + reader >> sample.assignmentChange >> sample.sampleVersion >> sample.validThrough >> sample.bytesWrittenPerKSecond; + return sample; +} + Key cdcTagOwnerKeyFor(Tag tag) { BinaryWriter wr(Unversioned()); wr.serializeBytes(cdcTagOwnerKeys.begin); @@ -1982,6 +2011,15 @@ TEST_CASE("/SystemData/NativeCDC") { ASSERT(earlierTagHistoryKey < laterTagHistoryKey); ASSERT(cdcTagHistoryRangeFor(streamId).contains(laterTagHistoryKey)); + const CDCTagLoadSample sample{ "assignment"_sr, minVersion, minVersion + 100, 123000 }; + const CDCTagLoadSample decodedSample = decodeCDCTagLoadValue(cdcTagLoadValue(sample)); + ASSERT_EQ(decodeCDCTagLoadKey(cdcTagLoadKeyFor(tag)), tag); + ASSERT(nonMetadataSystemKeys.contains(cdcTagLoadKeyFor(tag))); + ASSERT_EQ(decodedSample.assignmentChange, sample.assignmentChange); + ASSERT_EQ(decodedSample.sampleVersion, sample.sampleVersion); + ASSERT_EQ(decodedSample.validThrough, sample.validThrough); + ASSERT_EQ(decodedSample.bytesWrittenPerKSecond, sample.bytesWrittenPerKSecond); + const Value serializedTagHistory = ObjectWriter::toValue(decodedTagHistory, Unversioned()); const auto deserializedTagHistory = ObjectReader::fromStringRef(serializedTagHistory, Unversioned()); diff --git a/fdbclient/include/fdbclient/SystemData.h b/fdbclient/include/fdbclient/SystemData.h index 2e2d31e371d..a437e3e7c8b 100644 --- a/fdbclient/include/fdbclient/SystemData.h +++ b/fdbclient/include/fdbclient/SystemData.h @@ -323,6 +323,21 @@ Key cdcTagHistoryKeyFor(CDCStreamId streamId, Version version, Tag tag); KeyRange cdcTagHistoryRangeFor(CDCStreamId streamId); CDCTagHistoryEntry decodeCDCTagHistoryKey(KeyRef const& key); +// Advisory producer-write samples. The assignment generation invalidates every +// comparison when registrations, tag histories, or durable ownership change. +struct CDCTagLoadSample { + Value assignmentChange; + Version sampleVersion = invalidVersion; + Version validThrough = invalidVersion; + int64_t bytesWrittenPerKSecond = 0; +}; + +extern const KeyRangeRef cdcTagLoadKeys; +Key cdcTagLoadKeyFor(Tag tag); +Tag decodeCDCTagLoadKey(KeyRef const& key); +Value cdcTagLoadValue(CDCTagLoadSample const& sample); +CDCTagLoadSample decodeCDCTagLoadValue(ValueRef const& value); + // "\xff\x02/cdc/tagOwner/[[Tag]]" := "[[CDCStreamId]]" // Derived lookup hint, not authoritative ownership. Validate the stream is active // on this tag and read its durable proxy assignment in the same transaction. diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index 0e905b0b1a7..44f7962c241 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -31,6 +31,7 @@ #include "fdbclient/Knobs.h" #include "fdbclient/SystemData.h" #include "NativeCdcInternal.h" +#include "fdbserver/core/NativeCdcMetadata.h" #include "fdbserver/cdcproxy/CDCProxy.h" #include "fdbserver/cdcproxy/CDCProxyTest.h" #include "fdbserver/core/Knobs.h" diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index 8d0e9695d09..a06772b8a5d 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -30,6 +30,7 @@ #include "fdbclient/ClientBooleanParams.h" #include "fdbclient/FDBTypes.h" #include "NativeCdcInternal.h" +#include "fdbserver/core/NativeCdcMetadata.h" #include "fdbclient/SystemData.h" #include "fdbclient/DatabaseContext.h" #include "fdbrpc/FailureMonitor.h" diff --git a/fdbserver/core/CMakeLists.txt b/fdbserver/core/CMakeLists.txt index 78af86ad3b5..763c014ecad 100644 --- a/fdbserver/core/CMakeLists.txt +++ b/fdbserver/core/CMakeLists.txt @@ -19,5 +19,6 @@ target_include_directories(fdbserver_core ${CMAKE_CURRENT_SOURCE_DIR}/include ${CMAKE_CURRENT_BINARY_DIR}/include PRIVATE + ${CMAKE_SOURCE_DIR}/fdbclient ${CMAKE_SOURCE_DIR}/fdbserver/include) target_link_libraries(fdbserver_core PUBLIC fdbclient) diff --git a/fdbserver/core/NativeCdcMetadata.cpp b/fdbserver/core/NativeCdcMetadata.cpp new file mode 100644 index 00000000000..33f2b9ce826 --- /dev/null +++ b/fdbserver/core/NativeCdcMetadata.cpp @@ -0,0 +1,578 @@ +/* + * NativeCdcMetadata.cpp + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include +#include +#include +#include +#include +#include + +#include "fdbclient/DatabaseContext.h" +#include "fdbclient/Knobs.h" +#include "fdbclient/SystemData.h" +#include "NativeCdcInternal.h" +#include "fdbserver/core/NativeCdcMetadata.h" +#include "flow/CodeProbe.h" +#include "flow/Error.h" +#include "flow/Trace.h" +#include "flow/UnitTest.h" + +namespace { + +using CDCTagId = uint16_t; + +constexpr uint32_t maxNativeCdcTagCount = static_cast(std::numeric_limits::max()) + 1; + +bool validNativeCdcTagCount(int tagCount) { + return tagCount > 0 && static_cast(tagCount) <= maxNativeCdcTagCount; +} + +class NativeCdcIdentifierAllocator { + bool sawStream = false; + CDCStreamId maxStreamId = 0; + std::unordered_map tagStreamCounts; + std::unordered_map tagWriteRates; + +public: + void observeStreamId(CDCStreamId streamId) { + sawStream = true; + maxStreamId = std::max(maxStreamId, streamId); + } + + void observeTag(Tag tag) { + ASSERT_WE_THINK(tag.locality == tagLocalityCDC); + ++tagStreamCounts[tag.id]; + } + + void observeTagLoad(Tag tag, CDCTagLoadSample const& sample, ValueRef generation, Version readVersion) { + if (tag.locality == tagLocalityCDC && sample.assignmentChange == generation && sample.sampleVersion >= 0 && + sample.sampleVersion <= readVersion && sample.validThrough >= readVersion && + sample.bytesWrittenPerKSecond >= 0) { + tagWriteRates[tag.id] = sample.bytesWrittenPerKSecond; + } + } + + bool hasStreams(Tag tag) const { return tagStreamCounts.contains(tag.id); } + + std::pair allocate(int tagCount) const { + if (sawStream && maxStreamId == std::numeric_limits::max()) { + throw operation_failed(); + } + + const CDCStreamId streamId = sawStream ? maxStreamId + 1 : 1; + if (!validNativeCdcTagCount(tagCount)) { + throw invalid_option_value(); + } + const bool completeLoad = !tagStreamCounts.empty() && + std::all_of(tagStreamCounts.begin(), tagStreamCounts.end(), [&](auto const& entry) { + return entry.first >= tagCount || tagWriteRates.contains(entry.first); + }); + uint32_t leastStreams = std::numeric_limits::max(); + int64_t leastWriteRate = std::numeric_limits::max(); + CDCTagId selectedTagId = 0; + for (uint32_t tagId = 0; tagId < static_cast(tagCount); ++tagId) { + auto count = tagStreamCounts.find(static_cast(tagId)); + const uint32_t streamCount = count == tagStreamCounts.end() ? 0 : count->second; + const int64_t writeRate = completeLoad && streamCount > 0 ? tagWriteRates.at(tagId) : 0; + if (writeRate < leastWriteRate || (writeRate == leastWriteRate && streamCount < leastStreams)) { + leastWriteRate = writeRate; + leastStreams = streamCount; + selectedTagId = static_cast(tagId); + } + } + CODE_PROBE(completeLoad, "Native CDC registration places streams using fresh producer throughput"); + return { streamId, Tag(tagLocalityCDC, selectedTagId) }; + } +}; + +Future> getNativeCdcProxyAssignment(Transaction* tr, CDCStreamId streamId) { + RangeResult assignments = co_await tr->getRange(cdcProxyRangeFor(streamId), 2); + ASSERT_LE(assignments.size(), 1); + if (assignments.empty()) { + co_return Optional(); + } + const auto [assignedStreamId, proxyId] = decodeCDCProxyKey(assignments[0].key); + ASSERT_WE_THINK(assignedStreamId == streamId); + co_return proxyId; +} + +Future getNativeCdcCurrentTag(Transaction* tr, CDCStreamId streamId) { + // Tag-history keys sort by their big-endian assignment version, so the final + // key in this stream's prefix range contains its current tag. + RangeResult history = co_await tr->getRange(cdcTagHistoryRangeFor(streamId), 1, Snapshot::False, Reverse::True); + if (history.empty()) { + throw client_invalid_operation(); + } + co_return decodeCDCTagHistoryKey(history.front().key).tag; +} + +Future readNativeCdcCurrentTags(Transaction* tr, + std::unordered_map* currentTags, + NativeCdcIdentifierAllocator* allocator = nullptr) { + std::set activeStreamIds; + Key begin = cdcStreamKeys.begin; + while (begin < cdcStreamKeys.end) { + RangeResult streams = co_await tr->getRange(KeyRangeRef(begin, cdcStreamKeys.end), CLIENT_KNOBS->TOO_MANY); + for (const auto& kv : streams) { + const CDCStreamId streamId = decodeCDCStreamKey(kv.key); + activeStreamIds.insert(streamId); + if (allocator) { + allocator->observeStreamId(streamId); + } + } + if (!streams.more) { + break; + } + begin = keyAfter(streams.back().key); + } + + begin = cdcTagHistoryKeys.begin; + while (begin < cdcTagHistoryKeys.end) { + RangeResult histories = + co_await tr->getRange(KeyRangeRef(begin, cdcTagHistoryKeys.end), CLIENT_KNOBS->TOO_MANY); + for (const auto& kv : histories) { + const CDCTagHistoryEntry history = decodeCDCTagHistoryKey(kv.key); + if (allocator) { + allocator->observeStreamId(history.streamId); + } + if (activeStreamIds.contains(history.streamId)) { + (*currentTags)[history.streamId] = history.tag; + } + } + if (!histories.more) { + break; + } + begin = keyAfter(histories.back().key); + } +} + +Future> getNativeCdcProxyAssignmentForTag(Transaction* tr, Tag targetTag) { + const Key ownerKey = cdcTagOwnerKeyFor(targetTag); + Optional indexedStream = co_await tr->get(ownerKey); + if (indexedStream.present()) { + const CDCStreamId streamId = decodeCDCTagOwnerValue(indexedStream.get()); + Future> activeStream = tr->get(cdcStreamKeyFor(streamId)); + RangeResult history = co_await tr->getRange(cdcTagHistoryRangeFor(streamId), 1, Snapshot::False, Reverse::True); + // Keep the await separate so GCC 13 does not evaluate history.front() before the short-circuit guards. + const Optional activeStreamValue = co_await activeStream; + // The index is derived: removal or retagging can invalidate its representative, and the per-stream + // assignment remains authoritative across proxy replacement, including by older metadata writers. + if (activeStreamValue.present() && !history.empty() && + decodeCDCTagHistoryKey(history.front().key).tag == targetTag) { + Optional proxyId = co_await getNativeCdcProxyAssignment(tr, streamId); + if (proxyId.present()) { + CODE_PROBE(true, "Native CDC resolves a shared tag owner from its persisted index"); + co_return proxyId; + } + } + CODE_PROBE(true, "Native CDC rebuilds a stale tag owner index"); + tr->clear(ownerKey); + } + + std::unordered_map currentTags; + co_await readNativeCdcCurrentTags(tr, ¤tTags); + for (const auto& [streamId, tag] : currentTags) { + if (tag == targetTag) { + Optional proxyId = co_await getNativeCdcProxyAssignment(tr, streamId); + if (proxyId.present()) { + tr->set(ownerKey, cdcTagOwnerValue(streamId)); + CODE_PROBE(true, "Native CDC reconstructs a missing tag owner index from active streams"); + co_return proxyId; + } + } + } + co_return Optional(); +} + +void retireNativeCdcTag(Transaction* tr, Tag tag) { + // Dropping a history row must retain its final-pop obligation, including + // when another stream still protects the same tag or recovery intervenes. + tr->set(cdcRetiredTagPopKeyFor(tag), Value()); + tr->atomicOp( + cdcRetiredTagPopVersionKeyFor(tag), cdcVersionstampedMinVersionValue(), MutationRef::SetVersionstampedValue); +} + +void signalNativeCdcProxyAssignmentChange(Transaction* tr) { + // Assignment updates are low-rate control-plane operations. A single + // coalescing signal lets the cluster controller rescan all durable owners. + tr->set(cdcProxyAssignmentChangeKey, + BinaryWriter::toValue(deterministicRandom()->randomUniqueID(), + IncludeVersion(ProtocolVersion::withNativeCdc()))); +} + +Future observeNativeCdcMetadata(Transaction* tr, NativeCdcIdentifierAllocator* allocator) { + Optional maxStreamId = co_await tr->get(cdcMaxStreamIdKey); + if (maxStreamId.present()) { + allocator->observeStreamId(decodeCDCMaxStreamIdValue(maxStreamId.get())); + } + + std::unordered_map currentTags; + co_await readNativeCdcCurrentTags(tr, ¤tTags, allocator); + for (const auto& tagAssignment : currentTags) { + allocator->observeTag(tagAssignment.second); + } + if (!currentTags.empty()) { + const Value generation = (co_await tr->get(cdcProxyAssignmentChangeKey)).orDefault(Value()); + const Version readVersion = co_await tr->getReadVersion(); + Key begin = cdcTagLoadKeys.begin; + while (begin < cdcTagLoadKeys.end) { + RangeResult samples = co_await tr->getRange(KeyRangeRef(begin, cdcTagLoadKeys.end), CLIENT_KNOBS->TOO_MANY); + for (const auto& sample : samples) { + allocator->observeTagLoad( + decodeCDCTagLoadKey(sample.key), decodeCDCTagLoadValue(sample.value), generation, readVersion); + } + if (!samples.more) { + break; + } + begin = keyAfter(samples.back().key); + } + } +} + +Future> readNativeCdcTagStateImpl(Transaction* tr, CDCStreamId streamId) { + Future> keysFuture = tr->get(cdcStreamKeyFor(streamId)); + Future historyFuture = + tr->getRange(cdcTagHistoryRangeFor(streamId), 2, Snapshot::False, Reverse::True); + const Optional keys = co_await keysFuture; + const RangeResult history = co_await historyFuture; + if (!keys.present() || history.size() != 1 || history.more || !history.front().value.empty()) { + co_return Optional(); + } + co_return NativeCdcTagState{ streamId, + decodeCDCStreamKeysValue(keys.get()), + decodeCDCTagHistoryKey(history.front().key) }; +} + +} // namespace + +Future> readNativeCdcTagState(Transaction* tr, CDCStreamId streamId) { + return readNativeCdcTagStateImpl(tr, streamId); +} + +Future>> readNativeCdcTagStates(Transaction* tr, int maxStreams) { + if (maxStreams <= 0 || maxStreams == std::numeric_limits::max()) { + throw invalid_option_value(); + } + const RangeResult streams = co_await tr->getRange(cdcStreamKeys, maxStreams + 1); + if (streams.more || streams.size() > maxStreams) { + co_return Optional>(); + } + std::vector>> reads; + reads.reserve(streams.size()); + for (const auto& stream : streams) { + reads.push_back(readNativeCdcTagState(tr, decodeCDCStreamKey(stream.key))); + } + const std::vector> states = co_await getAll(reads); + std::vector result; + result.reserve(states.size()); + for (const auto& state : states) { + if (!state.present()) { + co_return Optional>(); + } + result.push_back(state.get()); + } + co_return Optional>(std::move(result)); +} + +Future prepareNativeCdcStreamRegistration(Transaction* tr, + Key name, + std::vector ranges, + UID proxyId) { + normalizeNativeCdcStreamRanges(name, ranges); + + const Key nameKey = cdcStreamNameKeyFor(name); + Optional currentId = co_await tr->get(nameKey); + if (currentId.present()) { + const CDCStreamId streamId = decodeCDCStreamNameValue(currentId.get()); + Optional currentKeys = co_await tr->get(cdcStreamKeyFor(streamId)); + if (!currentKeys.present() || decodeCDCStreamKeysValue(currentKeys.get()) != ranges) { + throw client_invalid_operation(); + } + if (!(co_await getNativeCdcProxyAssignment(tr, streamId)).present()) { + CODE_PROBE(true, "Native CDC registration restores missing stream owner", probe::decoration::rare); + const Tag tag = co_await getNativeCdcCurrentTag(tr, streamId); + Optional sharedTagProxy = co_await getNativeCdcProxyAssignmentForTag(tr, tag); + CODE_PROBE( + sharedTagProxy.present(), "Native CDC shared-tag streams use one owner", probe::decoration::rare); + const UID selectedProxy = sharedTagProxy.present() ? sharedTagProxy.get() : proxyId; + tr->set(cdcProxyKeyFor(streamId, selectedProxy), Value()); + if (!sharedTagProxy.present()) { + tr->set(cdcTagOwnerKeyFor(tag), cdcTagOwnerValue(streamId)); + } + signalNativeCdcProxyAssignmentChange(tr); + co_return NativeCdcRegistrationResult{ streamId, true }; + } + co_return NativeCdcRegistrationResult{ streamId, false }; + } + + // Disabling CDC stops new admission, but existing registrations and + // owner repair must remain available so durable streams can drain. + const bool nativeCdcEnabled = tr->getDatabase()->clientInfo->get().nativeCdcEnabled; + const int nativeCdcTagCount = tr->getDatabase()->clientInfo->get().nativeCdcTagCount; + validateNativeCdcEnabled(nativeCdcEnabled); + NativeCdcIdentifierAllocator allocator; + co_await observeNativeCdcMetadata(tr, &allocator); + const auto [streamId, tag] = allocator.allocate(nativeCdcTagCount); + // The read version is a conservative lower bound for tag routing. + // The versionstamped minimum below is the commit version, and stream + // initialization takes their maximum before exposing mutations. + const Version registrationVersion = co_await tr->getReadVersion(); + + tr->set(nameKey, cdcStreamNameValue(streamId)); + tr->set(cdcMaxStreamIdKey, cdcMaxStreamIdValue(streamId)); + tr->set(cdcStreamKeyFor(streamId), cdcStreamKeysValue(ranges)); + tr->set(cdcTagHistoryKeyFor(streamId, registrationVersion, tag), Value()); + tr->atomicOp( + cdcMinVersionKeyFor(streamId), cdcVersionstampedMinVersionValue(), MutationRef::SetVersionstampedValue); + Optional sharedTagProxy; + if (allocator.hasStreams(tag)) { + sharedTagProxy = co_await getNativeCdcProxyAssignmentForTag(tr, tag); + } + const UID selectedProxy = sharedTagProxy.present() ? sharedTagProxy.get() : proxyId; + tr->set(cdcProxyKeyFor(streamId, selectedProxy), Value()); + if (!sharedTagProxy.present()) { + tr->set(cdcTagOwnerKeyFor(tag), cdcTagOwnerValue(streamId)); + } + signalNativeCdcProxyAssignmentChange(tr); + co_return NativeCdcRegistrationResult{ streamId, true }; +} + +Future registerNativeCdcStream(Database cx, Key name, std::vector ranges, UID proxyId) { + normalizeNativeCdcStreamRanges(name, ranges); + + Transaction tr(cx); + while (true) { + Error err; + try { + tr.setOption(FDBTransactionOptions::LOCK_AWARE); + tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); + + const NativeCdcRegistrationResult result = + co_await prepareNativeCdcStreamRegistration(&tr, name, ranges, proxyId); + if (result.requiresCommit) { + co_await tr.commit(); + } + co_return result.streamId; + } catch (Error& e) { + err = e; + } + co_await tr.onError(err); + } +} + +Future removeNativeCdcStream(Database cx, Key name, CDCStreamId streamId, UID proxyId) { + if (name.empty() || streamId == 0) { + throw client_invalid_operation(); + } + + Transaction tr(cx); + while (true) { + Error err; + try { + tr.setOption(FDBTransactionOptions::LOCK_AWARE); + tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); + + const Key nameKey = cdcStreamNameKeyFor(name); + Optional currentId = co_await tr.get(nameKey); + if (!nativeCdcNameMatchesStream(currentId, streamId)) { + CODE_PROBE(currentId.present(), "Native CDC preserves a replacement stream during removal retry"); + if (currentId.present()) { + TraceEvent("NativeCdcRemovalPreservesReplacement") + .detail("RemovedStreamId", streamId) + .detail("ReplacementStreamId", decodeCDCStreamNameValue(currentId.get())); + } + co_return false; + } + + Optional assignedProxy = co_await getNativeCdcProxyAssignment(&tr, streamId); + if (!assignedProxy.present() || assignedProxy.get() != proxyId) { + CODE_PROBE(true, "Native CDC rejects removal through a stale owner"); + throw wrong_shard_server(); + } + + std::set removedTags; + const KeyRange historyRange = cdcTagHistoryRangeFor(streamId); + Key begin = historyRange.begin; + while (begin < historyRange.end) { + RangeResult history = + co_await tr.getRange(KeyRangeRef(begin, historyRange.end), CLIENT_KNOBS->TOO_MANY); + for (const auto& entry : history) { + removedTags.insert(decodeCDCTagHistoryKey(entry.key).tag); + } + if (!history.more) { + break; + } + begin = keyAfter(history.back().key); + } + + tr.clear(nameKey); + tr.clear(cdcStreamKeyFor(streamId)); + for (const Tag& tag : removedTags) { + const Key ownerKey = cdcTagOwnerKeyFor(tag); + Optional indexedStream = co_await tr.get(ownerKey); + if (indexedStream.present() && decodeCDCTagOwnerValue(indexedStream.get()) == streamId) { + tr.clear(ownerKey); + } + retireNativeCdcTag(&tr, tag); + } + tr.clear(cdcTagHistoryRangeFor(streamId)); + tr.clear(cdcMinVersionKeyFor(streamId)); + tr.clear(cdcProxyRangeFor(streamId)); + if (assignedProxy.present()) { + signalNativeCdcProxyAssignmentChange(&tr); + } + co_await tr.commit(); + CODE_PROBE(!removedTags.empty(), "Native CDC removal records final tagged pop work"); + TraceEvent("NativeCdcStreamRemoved") + .detail("StreamId", streamId) + .detail("ProxyID", proxyId) + .detail("CommitVersion", tr.getCommittedVersion()) + .detail("RetiredTagCount", removedTags.size()); + co_return true; + } catch (Error& e) { + if (e.code() == error_code_wrong_shard_server) { + throw; + } + err = e; + } + co_await tr.onError(err); + } +} + +Future reassignNativeCdcStreams(Database cx, UID oldProxyId, UID newProxyId) { + if (oldProxyId == newProxyId) { + co_return; + } + + Transaction tr(cx); + while (true) { + Error err; + try { + tr.setOption(FDBTransactionOptions::LOCK_AWARE); + tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); + tr.setOption(FDBTransactionOptions::PRIORITY_SYSTEM_IMMEDIATE); + + bool changed = false; + Key begin = cdcProxyKeys.begin; + while (begin < cdcProxyKeys.end) { + RangeResult assignments = + co_await tr.getRange(KeyRangeRef(begin, cdcProxyKeys.end), CLIENT_KNOBS->TOO_MANY); + for (const auto& assignment : assignments) { + const auto [streamId, proxyId] = decodeCDCProxyKey(assignment.key); + if (proxyId == oldProxyId) { + tr.clear(assignment.key); + tr.set(cdcProxyKeyFor(streamId, newProxyId), Value()); + changed = true; + } + } + if (!assignments.more) { + break; + } + begin = keyAfter(assignments.back().key); + } + + if (changed) { + CODE_PROBE(true, "Native CDC reassigns streams after proxy replacement"); + signalNativeCdcProxyAssignmentChange(&tr); + co_await tr.commit(); + } + co_return; + } catch (Error& e) { + err = e; + } + co_await tr.onError(err); + } +} + +void forceLinkNativeCdcMetadataTests() {} + +TEST_CASE("/NativeCDC/LifecycleAllocation") { + ASSERT(!validNativeCdcTagCount(-1)); + ASSERT(!validNativeCdcTagCount(0)); + ASSERT(validNativeCdcTagCount(1)); + ASSERT(validNativeCdcTagCount(std::numeric_limits::max() + 1u)); + ASSERT(!validNativeCdcTagCount(std::numeric_limits::max() + 2u)); + + NativeCdcIdentifierAllocator allocator; + auto [initialId, initialTag] = allocator.allocate(CLIENT_KNOBS->NATIVE_CDC_TAG_COUNT); + ASSERT_EQ(initialId, 1); + ASSERT_EQ(initialTag, Tag(tagLocalityCDC, 0)); + + allocator.observeStreamId(9); + allocator.observeTag(initialTag); + allocator.observeTag(Tag(tagLocalityCDC, 2)); + auto [nextId, nextTag] = allocator.allocate(CLIENT_KNOBS->NATIVE_CDC_TAG_COUNT); + ASSERT_EQ(nextId, 10); + ASSERT_EQ(nextTag, Tag(tagLocalityCDC, 1)); + + NativeCdcIdentifierAllocator publishedPoolAllocator; + publishedPoolAllocator.observeTag(Tag(tagLocalityCDC, 0)); + // The cluster-controller-published pool is authoritative even when it differs from this process's knob. + auto [publishedPoolId, publishedPoolTag] = publishedPoolAllocator.allocate(1); + ASSERT_EQ(publishedPoolId, 1); + ASSERT_EQ(publishedPoolTag, Tag(tagLocalityCDC, 0)); + + NativeCdcIdentifierAllocator fullPoolAllocator; + for (uint32_t tagId = 0; tagId < static_cast(CLIENT_KNOBS->NATIVE_CDC_TAG_COUNT); ++tagId) { + fullPoolAllocator.observeTag(Tag(tagLocalityCDC, static_cast(tagId))); + } + auto [sharedId, sharedTag] = fullPoolAllocator.allocate(CLIENT_KNOBS->NATIVE_CDC_TAG_COUNT); + ASSERT_EQ(sharedId, 1); + ASSERT_EQ(sharedTag, Tag(tagLocalityCDC, 0)); + + return Void(); +} + +TEST_CASE("/NativeCDC/ThroughputPlacement") { + const Value generation = "current"_sr; + const Tag hotTag(tagLocalityCDC, 0); + const Tag coldTag(tagLocalityCDC, 1); + const CDCTagLoadSample hot{ generation, 100, 200, 100000 }; + const CDCTagLoadSample cold{ generation, 100, 200, 0 }; + auto select = [&](CDCTagLoadSample const& hotSample, Optional coldSample, int tagCount = 2) { + NativeCdcIdentifierAllocator allocator; + allocator.observeTag(hotTag); + allocator.observeTag(coldTag); + allocator.observeTag(coldTag); + allocator.observeTagLoad(hotTag, hotSample, generation, 150); + if (coldSample.present()) { + allocator.observeTagLoad(coldTag, coldSample.get(), generation, 150); + } + return allocator.allocate(tagCount).second; + }; + ASSERT_EQ(select(hot, cold), coldTag); + ASSERT_EQ(select(hot, Optional()), hotTag); + ASSERT_EQ(select(hot, cold, 3), Tag(tagLocalityCDC, 2)); + + CDCTagLoadSample invalid = cold; + invalid.assignmentChange = "previous"_sr; + ASSERT_EQ(select(hot, invalid), hotTag); + invalid = cold; + invalid.validThrough = 149; + ASSERT_EQ(select(hot, invalid), hotTag); + invalid = cold; + invalid.sampleVersion = 151; + ASSERT_EQ(select(hot, invalid), hotTag); + invalid = cold; + invalid.bytesWrittenPerKSecond = -1; + ASSERT_EQ(select(hot, invalid), hotTag); + return Void(); +} diff --git a/fdbserver/core/ServerKnobs.cpp b/fdbserver/core/ServerKnobs.cpp index 0bd7a4e8d55..82a3a10791c 100644 --- a/fdbserver/core/ServerKnobs.cpp +++ b/fdbserver/core/ServerKnobs.cpp @@ -189,6 +189,14 @@ void ServerKnobs::initialize(Randomize randomize, ClientKnobs* clientKnobs, IsSi init( CDC_PROXY_FAILURE_COALESCE_DELAY, 0.0 ); init( CDC_PROXY_POP_MIN_INTERVAL, 0.1 ); if( randomize && buggify() ) CDC_PROXY_POP_MIN_INTERVAL = 0.01; init( CDC_PROXY_POP_SCAN_INTERVAL, 5.0 ); if( randomize && buggify() ) CDC_PROXY_POP_SCAN_INTERVAL = 0.1; + init( NATIVE_CDC_TAG_BALANCING_ENABLED, true ); + init( NATIVE_CDC_TAG_SAMPLE_INTERVAL, 30.0 ); + init( NATIVE_CDC_TAG_SAMPLE_TIMEOUT, 5.0 ); + init( NATIVE_CDC_TAG_SAMPLE_MAX_AGE, 90.0 ); + init( NATIVE_CDC_TAG_MAX_STREAMS, 1000 ); + init( NATIVE_CDC_TAG_MODEL_MAX_ENTRIES, 2000000 ); + init( NATIVE_CDC_TAG_SAMPLE_CONCURRENCY, 8 ); + init( NATIVE_CDC_TAG_SAMPLE_SHARD_LIMIT, 1000 ); init( APPLY_MUTATION_BYTES, 1e6 ); init( BUGGIFY_RECOVER_MEMORY_LIMIT, 1e6 ); init( BUGGIFY_WORKER_REMOVED_MAX_LAG, 30 ); diff --git a/fdbserver/core/include/fdbserver/core/Knobs.h b/fdbserver/core/include/fdbserver/core/Knobs.h index 2f65564bde0..f2d818aff7a 100644 --- a/fdbserver/core/include/fdbserver/core/Knobs.h +++ b/fdbserver/core/include/fdbserver/core/Knobs.h @@ -80,6 +80,14 @@ class SWIFT_CXX_IMMORTAL_SINGLETON_TYPE ServerKnobs : public KnobsImpl ranges; + CDCTagHistoryEntry assignment; +}; + +// An absent result means the bounded snapshot is incomplete; it must not be used +// as an empty or zero-load configuration. +Future> readNativeCdcTagState(Transaction* tr, CDCStreamId streamId); +Future>> readNativeCdcTagStates(Transaction* tr, int maxStreams); +struct NativeCdcRegistrationResult { + CDCStreamId streamId; + // Describes only mutations prepared by this helper, not unrelated caller mutations. + bool requiresCommit; +}; + +// Prepares one registration without committing or retrying. The caller sets LOCK_AWARE and ACCESS_SYSTEM_KEYS +// and owns commit and retry handling. Transaction does not read its own writes: prepare at most one registration +// per transaction, without earlier mutations to the CDC metadata this operation reads. +Future prepareNativeCdcStreamRegistration(Transaction* tr, + Key name, + std::vector ranges, + UID proxyId); + +// Durable metadata operations used by CDC server roles. Registration is +// feature gated; drain and cleanup operations remain available for streams +// persisted before native CDC is disabled. +Future registerNativeCdcStream(Database cx, Key name, std::vector ranges, UID proxyId); +// Persists per-tag final-pop watermarks before removing stream metadata. +Future removeNativeCdcStream(Database cx, Key name, CDCStreamId streamId, UID proxyId); +// Atomically moves any streams assigned to a failed proxy to its replacement. +Future reassignNativeCdcStreams(Database cx, UID oldProxyId, UID newProxyId); + +#endif // FDBSERVER_CORE_NATIVECDCMETADATA_H diff --git a/fdbserver/datadistributor/CMakeLists.txt b/fdbserver/datadistributor/CMakeLists.txt index e888f7d8718..69f2895c4a5 100644 --- a/fdbserver/datadistributor/CMakeLists.txt +++ b/fdbserver/datadistributor/CMakeLists.txt @@ -13,5 +13,6 @@ target_include_directories(fdbserver_datadistributor PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/include PRIVATE + ${CMAKE_SOURCE_DIR}/fdbclient ${CMAKE_CURRENT_SOURCE_DIR}) target_link_libraries(fdbserver_datadistributor PRIVATE fdbserver_core) diff --git a/fdbserver/datadistributor/DataDistribution.cpp b/fdbserver/datadistributor/DataDistribution.cpp index 9365b10bbfa..dd878ff009e 100644 --- a/fdbserver/datadistributor/DataDistribution.cpp +++ b/fdbserver/datadistributor/DataDistribution.cpp @@ -43,6 +43,7 @@ #include "DDTeamCollection.h" #include "DataDistribution.h" #include "DDRelocationQueue.h" +#include "NativeCdcBalancer.h" #include "fdbserver/core/Knobs.h" #include "fdbserver/core/MoveKeys.h" #include "fdbserver/core/QuietDatabase.h" @@ -3050,6 +3051,8 @@ Future bulkDumpCore(Reference self, Future readyToS } void addDataDistributionActors(Reference self, std::vector>& actors) { + actors.push_back(nativeCdcBalancer( + self->txnProcessor->context(), self->lock, self->context->ddEnabledState.get(), self->initialized.getFuture())); if (bulkLoadIsEnabled(self->initData->bulkLoadMode)) { TraceEvent(SevInfo, "DDBulkLoadModeEnabled", self->ddId) .detail("UsableRegions", self->configuration.usableRegions); diff --git a/fdbserver/datadistributor/NativeCdcBalancer.cpp b/fdbserver/datadistributor/NativeCdcBalancer.cpp new file mode 100644 index 00000000000..0ec7fae2da9 --- /dev/null +++ b/fdbserver/datadistributor/NativeCdcBalancer.cpp @@ -0,0 +1,397 @@ +/* + * NativeCdcBalancer.cpp + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include +#include +#include +#include +#include +#include +#include + +#include "NativeCdcBalancer.h" +#include "fdbserver/core/NativeCdcMetadata.h" +#include "fdbclient/DatabaseContext.h" +#include "fdbclient/KeyRangeMap.h" +#include "fdbclient/StorageServerInterface.h" +#include "fdbclient/SystemData.h" +#include "fdbserver/core/Knobs.h" +#include "flow/CodeProbe.h" +#include "flow/Trace.h" +#include "flow/UnitTest.h" + +namespace { + +bool addNativeCdcLoad(int64_t* total, int64_t value) { + if (value < 0 || value > std::numeric_limits::max() - *total) { + return false; + } + *total += value; + return true; +} + +class NativeCdcLoadModel { + struct Segment { + KeyRange keys; + std::set tags; + Optional load; + }; + + std::vector segments; + std::map tagLoads; + bool complete = false; + + explicit NativeCdcLoadModel(std::vector const& states) { + KeyRangeMap> coveringTags; + for (const auto& state : states) { + for (const auto& keys : state.ranges) { + for (auto range : coveringTags.modify(keys)) { + range->value().insert(state.assignment.tag); + } + } + } + for (auto range : coveringTags.ranges()) { + if (!range.value().empty()) { + segments.push_back(Segment{ KeyRange(range.range()), range.value(), {} }); + } + } + } + +public: + static Optional create(std::vector const& states, int64_t maxEntries) { + if (maxEntries <= 0) { + return {}; + } + if (!states.empty()) { + // M ranges have at most 2M-1 nonempty segments, each containing at most N tags. + // Bound coverage memberships before constructing the segment sets. + const uint64_t maxRanges = static_cast(maxEntries) / states.size() / 2; + uint64_t ranges = 0; + for (const auto& state : states) { + if (state.ranges.empty() || state.ranges.size() > maxRanges - ranges) { + return {}; + } + ranges += state.ranges.size(); + } + } + return NativeCdcLoadModel(states); + } + + size_t segmentCount() const { return segments.size(); } + KeyRange const& segmentKeys(size_t index) const { return segments[index].keys; } + std::map const& loads() const { + ASSERT(complete); + return tagLoads; + } + + bool setSample(size_t index, int64_t load) { + ASSERT(!complete); + if (load < 0) { + return false; + } + segments[index].load = load; + return true; + } + + bool finishSamples() { + tagLoads.clear(); + for (const auto& segment : segments) { + if (!segment.load.present()) { + return false; + } + for (Tag tag : segment.tags) { + if (!addNativeCdcLoad(&tagLoads[tag], segment.load.get())) { + return false; + } + } + } + complete = true; + return true; + } +}; + +Version nativeCdcDurationVersions(double seconds) { + const long double versions = static_cast(seconds) * SERVER_KNOBS->VERSIONS_PER_SECOND; + if (versions >= std::numeric_limits::max()) { + return std::numeric_limits::max(); + } + return static_cast(versions); +} + +bool validNativeCdcBalancerKnobs() { + return SERVER_KNOBS->NATIVE_CDC_TAG_MAX_STREAMS > 0 && + SERVER_KNOBS->NATIVE_CDC_TAG_MAX_STREAMS < std::numeric_limits::max() && + SERVER_KNOBS->NATIVE_CDC_TAG_MODEL_MAX_ENTRIES > 0 && SERVER_KNOBS->NATIVE_CDC_TAG_SAMPLE_CONCURRENCY > 0 && + SERVER_KNOBS->NATIVE_CDC_TAG_SAMPLE_SHARD_LIMIT > 1 && + std::isfinite(SERVER_KNOBS->NATIVE_CDC_TAG_SAMPLE_TIMEOUT) && + SERVER_KNOBS->NATIVE_CDC_TAG_SAMPLE_TIMEOUT > 0 && + std::isfinite(SERVER_KNOBS->NATIVE_CDC_TAG_SAMPLE_MAX_AGE) && + SERVER_KNOBS->NATIVE_CDC_TAG_SAMPLE_MAX_AGE > 0 && SERVER_KNOBS->VERSIONS_PER_SECOND > 0; +} + +Future sampleNativeCdcRanges(Database cx, NativeCdcLoadModel* model, size_t* nextSegment) { + while (*nextSegment < model->segmentCount()) { + const size_t index = (*nextSegment)++; + const StorageMetrics metrics = + co_await cx->getStorageMetrics(model->segmentKeys(index), SERVER_KNOBS->NATIVE_CDC_TAG_SAMPLE_SHARD_LIMIT); + if (!model->setSample(index, metrics.bytesWrittenPerKSecond)) { + throw operation_failed(); + } + co_await yield(TaskPriority::DataDistribution); + } +} + +Future sampleNativeCdcLoads(Database cx, NativeCdcLoadModel* model) { + size_t nextSegment = 0; + std::vector> workers; + const size_t workerCount = + std::min(model->segmentCount(), static_cast(SERVER_KNOBS->NATIVE_CDC_TAG_SAMPLE_CONCURRENCY)); + workers.reserve(workerCount); + for (size_t i = 0; i < workerCount; ++i) { + workers.push_back(sampleNativeCdcRanges(cx, model, &nextSegment)); + } + co_await waitForAll(workers); +} + +struct NativeCdcMetadataSnapshot { + Value assignmentChange; + Version version; + std::vector streams; +}; + +class NativeCdcBalancer { + Database cx; + MoveKeysLock lock; + const DDEnabledState* ddEnabledState; + + bool samplingEnabled() const { + return SERVER_KNOBS->NATIVE_CDC_TAG_BALANCING_ENABLED && cx->clientInfo->get().nativeCdcEnabled; + } + + Future> readSnapshot() { + Transaction tr(cx); + while (true) { + Error err; + try { + tr.setOption(FDBTransactionOptions::READ_LOCK_AWARE); + tr.setOption(FDBTransactionOptions::READ_SYSTEM_KEYS); + const Optional change = co_await tr.get(cdcProxyAssignmentChangeKey); + const Value generation = change.present() ? change.get() : Value(); + Optional> states = + co_await readNativeCdcTagStates(&tr, SERVER_KNOBS->NATIVE_CDC_TAG_MAX_STREAMS); + if (!states.present()) { + TraceEvent("NativeCdcTagMetadataUnavailable", lock.myOwner) + .detail("StreamLimit", SERVER_KNOBS->NATIVE_CDC_TAG_MAX_STREAMS); + co_return Optional(); + } + const Version version = co_await tr.getReadVersion(); + co_return Optional( + NativeCdcMetadataSnapshot{ generation, version, std::move(states.get()) }); + } catch (Error& e) { + err = e; + } + co_await tr.onError(err); + } + } + + Future currentGeneration(Transaction* tr, Value expected, Version sampledAt, Version validThrough) const { + if (!samplingEnabled()) { + co_return false; + } + const Optional generation = co_await tr->get(cdcProxyAssignmentChangeKey); + if ((generation.present() ? generation.get() : Value()) != expected) { + CODE_PROBE(true, "Native CDC DD discards samples after assignment changes"); + co_return false; + } + const Version version = co_await tr->getReadVersion(); + const bool fresh = version >= sampledAt && version <= validThrough; + co_return fresh; + } + + Future publishLoads(NativeCdcMetadataSnapshot const& snapshot, + Version validThrough, + NativeCdcLoadModel const& model) { + Transaction tr(cx); + while (true) { + Error err; + try { + tr.setOption(FDBTransactionOptions::LOCK_AWARE); + tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); + if (!(co_await currentGeneration(&tr, snapshot.assignmentChange, snapshot.version, validThrough))) { + co_return false; + } + tr.clear(cdcTagLoadKeys); + for (const auto& [tag, load] : model.loads()) { + tr.set(cdcTagLoadKeyFor(tag), + cdcTagLoadValue( + CDCTagLoadSample{ snapshot.assignmentChange, snapshot.version, validThrough, load })); + } + co_await checkMoveKeysLock(&tr, lock, ddEnabledState); + co_await tr.commit(); + CODE_PROBE(true, "Native CDC DD publishes producer tag throughput samples"); + TraceEvent("NativeCdcTagLoadSampled", lock.myOwner) + .detail("Streams", snapshot.streams.size()) + .detail("Tags", model.loads().size()) + .detail("Segments", model.segmentCount()) + .detail("SampleVersion", snapshot.version) + .detail("ValidThrough", validThrough); + co_return true; + } catch (Error& e) { + err = e; + } + co_await tr.onError(err); + } + } + + Future runPass() { + if (!samplingEnabled()) { + co_return; + } + if (!validNativeCdcBalancerKnobs()) { + TraceEvent(SevWarn, "NativeCdcTagBalancerInvalidKnobs", lock.myOwner); + co_return; + } + Optional snapshot = co_await readSnapshot(); + if (!snapshot.present() || snapshot.get().streams.empty()) { + co_return; + } + const int tagCount = cx->clientInfo->get().nativeCdcTagCount; + if (tagCount <= 0 || tagCount > static_cast(std::numeric_limits::max()) + 1) { + co_return; + } + const Version lifetime = nativeCdcDurationVersions(SERVER_KNOBS->NATIVE_CDC_TAG_SAMPLE_MAX_AGE); + const Version validThrough = snapshot.get().version > std::numeric_limits::max() - lifetime + ? std::numeric_limits::max() + : snapshot.get().version + lifetime; + Optional boundedModel = + NativeCdcLoadModel::create(snapshot.get().streams, SERVER_KNOBS->NATIVE_CDC_TAG_MODEL_MAX_ENTRIES); + if (!boundedModel.present()) { + CODE_PROBE(true, "Native CDC DD skips an overlap model exceeding its entry budget"); + TraceEvent("NativeCdcTagModelBudgetExceeded", lock.myOwner) + .detail("Streams", snapshot.get().streams.size()) + .detail("MaxEntries", SERVER_KNOBS->NATIVE_CDC_TAG_MODEL_MAX_ENTRIES); + co_return; + } + NativeCdcLoadModel& model = boundedModel.get(); + // One deadline bounds all segment requests. Partial/failed samples are never published as zero load. + const Optional sampled = + co_await timeout(sampleNativeCdcLoads(cx, &model), SERVER_KNOBS->NATIVE_CDC_TAG_SAMPLE_TIMEOUT); + if (!sampled.present() || !model.finishSamples()) { + CODE_PROBE(true, "Native CDC DD skips incomplete throughput samples"); + TraceEvent("NativeCdcTagSamplingIncomplete", lock.myOwner).detail("Segments", model.segmentCount()); + co_return; + } + co_await publishLoads(snapshot.get(), validThrough, model); + } + +public: + NativeCdcBalancer(Database cx, MoveKeysLock lock, const DDEnabledState* ddEnabledState) + : cx(cx), lock(lock), ddEnabledState(ddEnabledState) {} + + Future run(Future initialized) { + co_await initialized; + while (true) { + try { + co_await runPass(); + } catch (Error& e) { + if (e.code() == error_code_actor_cancelled || e.code() == error_code_broken_promise || + e.code() == error_code_movekeys_conflict) { + throw; + } + TraceEvent(SevWarn, "NativeCdcTagBalancerError", lock.myOwner).error(e); + } + const double interval = SERVER_KNOBS->NATIVE_CDC_TAG_SAMPLE_INTERVAL; + co_await delay(std::isfinite(interval) && interval > 0 ? interval : 30.0, TaskPriority::DataDistribution); + } + } +}; + +NativeCdcTagState nativeCdcPolicyTestStream(CDCStreamId streamId, KeyRange keys, uint16_t tag) { + return NativeCdcTagState{ streamId, { keys }, CDCTagHistoryEntry(streamId, 100, Tag(tagLocalityCDC, tag)) }; +} + +TEST_CASE("/NativeCDC/TagBalancing/IncompleteAndZeroSamples") { + auto model = NativeCdcLoadModel::create({ nativeCdcPolicyTestStream(1, KeyRangeRef("a"_sr, "b"_sr), 0) }, + SERVER_KNOBS->NATIVE_CDC_TAG_MODEL_MAX_ENTRIES) + .get(); + ASSERT(!model.finishSamples()); + ASSERT(!model.setSample(0, -1)); + ASSERT(model.setSample(0, 0)); + ASSERT(model.finishSamples()); + ASSERT_EQ(model.loads().at(Tag(tagLocalityCDC, 0)), 0); + int64_t total = std::numeric_limits::max() - 1; + ASSERT(!addNativeCdcLoad(&total, 2)); + ASSERT(addNativeCdcLoad(&total, 1)); + return Void(); +} + +TEST_CASE("/NativeCDC/TagBalancing/OverlappingTagLoads") { + auto model = NativeCdcLoadModel::create({ nativeCdcPolicyTestStream(1, KeyRangeRef("a"_sr, "d"_sr), 0), + nativeCdcPolicyTestStream(2, KeyRangeRef("c"_sr, "f"_sr), 0), + nativeCdcPolicyTestStream(3, KeyRangeRef("c"_sr, "d"_sr), 1) }, + 18) + .get(); + ASSERT_EQ(model.segmentCount(), 3); + ASSERT(model.setSample(0, 2000)); + ASSERT(model.setSample(1, 3000)); + ASSERT(model.setSample(2, 5000)); + ASSERT(model.finishSamples()); + ASSERT_EQ(model.loads().at(Tag(tagLocalityCDC, 0)), 10000); + ASSERT_EQ(model.loads().at(Tag(tagLocalityCDC, 1)), 3000); + return Void(); +} + +TEST_CASE("/NativeCDC/TagBalancing/ModelEntryBudget") { + std::vector streams{ nativeCdcPolicyTestStream(1, KeyRangeRef("a"_sr, "f"_sr), 0), + nativeCdcPolicyTestStream(2, KeyRangeRef("b"_sr, "e"_sr), 0), + nativeCdcPolicyTestStream(3, KeyRangeRef("c"_sr, "d"_sr), 1) }; + // The same stream count needs a larger entry budget once streams cover disjoint range unions. + ASSERT(NativeCdcLoadModel::create(streams, 18).present()); + for (auto& stream : streams) { + stream.ranges.emplace_back(KeyRangeRef("x"_sr, "y"_sr)); + } + ASSERT(!NativeCdcLoadModel::create(streams, 18).present()); + ASSERT(!NativeCdcLoadModel::create(streams, 35).present()); + auto model = NativeCdcLoadModel::create(streams, 36); + ASSERT(model.present()); + ASSERT_EQ(model.get().segmentCount(), 6); + for (size_t i = 0; i < model.get().segmentCount(); ++i) { + ASSERT(model.get().setSample(i, 1000)); + } + ASSERT(model.get().finishSamples()); + ASSERT_EQ(model.get().loads().at(Tag(tagLocalityCDC, 0)), 6000); + ASSERT_EQ(model.get().loads().at(Tag(tagLocalityCDC, 1)), 2000); + ASSERT(!NativeCdcLoadModel::create(streams, 0).present()); + ASSERT(!NativeCdcLoadModel::create(streams, -1).present()); + ASSERT(NativeCdcLoadModel::create(streams, std::numeric_limits::max()).present()); + return Void(); +} + +} // namespace + +Future nativeCdcBalancer(Database cx, + MoveKeysLock lock, + const DDEnabledState* ddEnabledState, + Future initialized) { + NativeCdcBalancer balancer(cx, lock, ddEnabledState); + co_await balancer.run(initialized); +} + +void forceLinkNativeCdcBalancerTests() {} diff --git a/fdbserver/datadistributor/NativeCdcBalancer.h b/fdbserver/datadistributor/NativeCdcBalancer.h new file mode 100644 index 00000000000..50bf3398dd5 --- /dev/null +++ b/fdbserver/datadistributor/NativeCdcBalancer.h @@ -0,0 +1,32 @@ +/* + * NativeCdcBalancer.h + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#pragma once + +#include "fdbclient/NativeAPI.h" +#include "fdbserver/core/MoveKeys.h" + +// The DD epoch fences publication of advisory producer-load samples. +Future nativeCdcBalancer(Database cx, + MoveKeysLock lock, + const DDEnabledState* ddEnabledState, + Future initialized); + +void forceLinkNativeCdcBalancerTests(); diff --git a/fdbserver/workloads/NativeCdcInitialPlacement.cpp b/fdbserver/workloads/NativeCdcInitialPlacement.cpp new file mode 100644 index 00000000000..25ffdc9b55b --- /dev/null +++ b/fdbserver/workloads/NativeCdcInitialPlacement.cpp @@ -0,0 +1,198 @@ +/* + * NativeCdcInitialPlacement.cpp + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include +#include +#include + +#include "fdbclient/DatabaseContext.h" +#include "fdbclient/NativeCdc.h" +#include "fdbclient/SystemData.h" +#include "fdbserver/core/NativeCdcMetadata.h" +#include "fdbserver/tester/workloads.h" +#include "flow/CodeProbe.h" + +class NativeCdcInitialPlacementWorkload : public TestWorkload { + const Key coldName = "native-cdc-placement/cold"_sr; + const Key hotName = "native-cdc-placement/hot"_sr; + const Key duplicateName = "native-cdc-placement/duplicate"_sr; + const Key placedName = "native-cdc-placement/placed"_sr; + const Key coldKey = "native-cdc-placement/data/cold"_sr; + const Key hotKey = "native-cdc-placement/data/hot"_sr; + const Key placedKey = "native-cdc-placement/data/placed"_sr; + const double operationTimeout; + + static Future writeValue(Database cx, Key key, Value value) { + Transaction tr(cx); + while (true) { + Error error; + try { + tr.set(key, value); + co_await tr.commit(); + co_return tr.getCommittedVersion(); + } catch (Error& e) { + error = e; + } + co_await tr.onError(error); + } + } + + static Future produceHotWrites(Database cx, Key key) { + int sequence = 0; + while (true) { + const Value value = StringRef(std::string(32768, 'a' + (++sequence % 26))); + co_await writeValue(cx, key, value); + co_await delay(0.05); + } + } + + static Future readTag(Database cx, CDCStreamId streamId) { + Transaction tr(cx); + while (true) { + Error error; + try { + tr.setOption(FDBTransactionOptions::READ_LOCK_AWARE); + tr.setOption(FDBTransactionOptions::READ_SYSTEM_KEYS); + const RangeResult history = co_await tr.getRange(cdcTagHistoryRangeFor(streamId), 2); + ASSERT_EQ(history.size(), 1); + ASSERT(!history.more); + ASSERT(history.front().value.empty()); + co_return decodeCDCTagHistoryKey(history.front().key).tag; + } catch (Error& e) { + error = e; + } + co_await tr.onError(error); + } + } + + static Future usableLoad(Transaction* tr, Tag coldTag, Tag hotTag) { + const Value generation = (co_await tr->get(cdcProxyAssignmentChangeKey)).orDefault(Value()); + const Version version = co_await tr->getReadVersion(); + const Optional coldValue = co_await tr->get(cdcTagLoadKeyFor(coldTag)); + const Optional hotValue = co_await tr->get(cdcTagLoadKeyFor(hotTag)); + if (!coldValue.present() || !hotValue.present()) { + co_return false; + } + const auto cold = decodeCDCTagLoadValue(coldValue.get()); + const auto hot = decodeCDCTagLoadValue(hotValue.get()); + for (const auto& sample : { cold, hot }) { + if (sample.assignmentChange != generation || sample.sampleVersion < 0 || sample.sampleVersion > version || + sample.validThrough < version || sample.bytesWrittenPerKSecond < 0) { + co_return false; + } + } + co_return hot.bytesWrittenPerKSecond > cold.bytesWrittenPerKSecond; + } + + Future registerWithLoad(Database cx, Tag coldTag, Tag hotTag) { + Transaction tr(cx); + std::set attemptedIds; + while (true) { + Error error; + try { + tr.setOption(FDBTransactionOptions::LOCK_AWARE); + tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); + const Optional existing = co_await tr.get(cdcStreamNameKeyFor(placedName)); + if (existing.present()) { + // A successful but ambiguous commit invalidates its own sample generation. + const CDCStreamId streamId = decodeCDCStreamNameValue(existing.get()); + ASSERT(attemptedIds.contains(streamId)); + co_return streamId; + } + // Guard the actual registration snapshot: a separate readiness check can expire on recovery. + if (!(co_await usableLoad(&tr, coldTag, hotTag)) || cx->clientInfo->get().cdcProxies.empty()) { + tr.reset(); + co_await delay(0.1); + continue; + } + const auto result = co_await prepareNativeCdcStreamRegistration( + &tr, + placedName, + { KeyRange(KeyRangeRef(placedKey, keyAfter(placedKey))) }, + cx->clientInfo->get().cdcProxies.front().id()); + ASSERT(result.requiresCommit); + attemptedIds.insert(result.streamId); + co_await tr.commit(); + co_return result.streamId; + } catch (Error& e) { + error = e; + } + co_await tr.onError(error); + } + } + + Future run(Database cx) { + ASSERT_EQ(cx->clientInfo->get().nativeCdcTagCount, 2); + const KeyRange coldRange(KeyRangeRef(coldKey, keyAfter(coldKey))); + const KeyRange hotRange(KeyRangeRef(hotKey, keyAfter(hotKey))); + const CDCStreamId cold = co_await registerNativeCdcStreamClient(cx, coldName, { coldRange }); + const CDCStreamId hot = co_await registerNativeCdcStreamClient(cx, hotName, { hotRange }); + const CDCStreamId duplicate = co_await registerNativeCdcStreamClient(cx, duplicateName, { coldRange }); + const Tag coldTag = co_await readTag(cx, cold); + const Tag hotTag = co_await readTag(cx, hot); + ASSERT_NE(coldTag, hotTag); + ASSERT_EQ(co_await readTag(cx, duplicate), coldTag); + Future producer = produceHotWrites(cx, hotKey); + const CDCStreamId placed = co_await timeoutError(registerWithLoad(cx, coldTag, hotTag), operationTimeout); + producer.cancel(); + ASSERT_EQ(co_await readTag(cx, placed), coldTag); + // Existing streams retain their original tag; placement requires no history cutover protocol. + ASSERT_EQ(co_await readTag(cx, cold), coldTag); + ASSERT_EQ(co_await readTag(cx, duplicate), coldTag); + ASSERT_EQ(co_await readTag(cx, hot), hotTag); + Reference consumer = co_await createNativeCdcConsumer(cx, placedName); + const Value marker = "placed-stream-delivery"_sr; + const Version committed = co_await writeValue(cx, placedKey, marker); + bool found = false; + while (!found) { + const CDCConsumeReply reply = co_await timeoutError(consumer->consume(), operationTimeout); + for (const auto& versioned : reply.mutations) { + if (versioned.version == committed) { + ASSERT_EQ(versioned.mutations.size(), 1); + ASSERT_EQ(versioned.mutations.front().type, MutationRef::SetValue); + ASSERT_EQ(versioned.mutations.front().param1, placedKey); + ASSERT_EQ(versioned.mutations.front().param2, marker); + found = true; + } + } + } + co_await consumer->acknowledge(); + for (const auto& name : { placedName, duplicateName, hotName, coldName }) { + co_await removeNativeCdcStreamClient(cx, name); + } + CODE_PROBE(true, "Native CDC initial placement favors measured load over stream count"); + TraceEvent("NativeCdcInitialPlacementVerified").detail("PlacedStream", placed).detail("Tag", coldTag); + } + +public: + static constexpr auto NAME = "NativeCdcInitialPlacement"; + explicit NativeCdcInitialPlacementWorkload(WorkloadContext const& wc) + : TestWorkload(wc), operationTimeout(getOption(options, "operationTimeout"_sr, 120.0)) {} + + void disableFailureInjectionWorkloads(std::set& out) const override { out.insert("RandomRangeLock"); } + Future setup(Database const& cx) override { return Void(); } + Future start(Database const& cx) override { + return clientId == 0 ? timeoutError(run(cx), operationTimeout * 2) : Void(); + } + Future check(Database const& cx) override { return true; } + void getMetrics(std::vector& metrics) override {} +}; + +WorkloadFactory NativeCdcInitialPlacementWorkloadFactory; diff --git a/fdbserver/workloads/UnitTests.cpp b/fdbserver/workloads/UnitTests.cpp index 1b3a90ee4fa..20429e3408f 100644 --- a/fdbserver/workloads/UnitTests.cpp +++ b/fdbserver/workloads/UnitTests.cpp @@ -53,6 +53,8 @@ void forceLinkIPagerTests(); void forceLinkMockS3ServerTests(); void forceLinkAuditUtilsTests(); void forceLinkShardsAffectedByTeamFailureTests(); +void forceLinkNativeCdcBalancerTests(); +void forceLinkNativeCdcMetadataTests(); void forceLinkClusterHealthMonitorTests(); void forceLinkGrvQueueDelayTests(); void forceLinkGrvProxyStarvationTests(); @@ -132,6 +134,8 @@ struct UnitTestWorkload : TestWorkload { forceLinkMockS3ServerTests(); forceLinkAuditUtilsTests(); forceLinkShardsAffectedByTeamFailureTests(); + forceLinkNativeCdcBalancerTests(); + forceLinkNativeCdcMetadataTests(); forceLinkClusterHealthMonitorTests(); forceLinkGrvQueueDelayTests(); forceLinkGrvProxyStarvationTests(); diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 91a19bb42ed..4931a016b02 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -225,6 +225,7 @@ if(WITH_PYTHON) add_fdb_test(TEST_FILES fast/RandomUnitTests.toml) add_fdb_test(TEST_FILES fast/RangeLocking.toml) add_fdb_test(TEST_FILES fast/NativeCdcEndToEnd.toml) + add_fdb_test(TEST_FILES fast/NativeCdcInitialPlacement.toml) add_fdb_test(TEST_FILES fast/NativeCdcBuggify.toml) add_fdb_test(TEST_FILES fast/NativeCdcAssignmentPublication.toml) add_fdb_test(TEST_FILES fast/NativeCdcSharedTag.toml) diff --git a/tests/fast/NativeCdcInitialPlacement.toml b/tests/fast/NativeCdcInitialPlacement.toml new file mode 100644 index 00000000000..d522e81ab69 --- /dev/null +++ b/tests/fast/NativeCdcInitialPlacement.toml @@ -0,0 +1,27 @@ +[configuration] +config = 'single commit_proxies=1 grv_proxies=2' +singleRegion = true +datacenters = 1 +machineCount = 12 +statelessProcessClassesPerDC = 3 +buggify = false +faultInjection = false + +[[knobs]] +enable_native_cdc = true +native_cdc_tag_count = 2 +native_cdc_tag_balancing_enabled = true +native_cdc_tag_sample_interval = 0.5 +native_cdc_tag_sample_timeout = 5.0 +native_cdc_tag_sample_max_age = 30.0 + +[[test]] +testTitle = 'NativeCdcInitialPlacement' +useDB = true +waitForQuiescenceEnd = false +connectionFailuresDisableDuration = 1000000 +timeout = 300 + + [[test.workload]] + testName = 'NativeCdcInitialPlacement' + operationTimeout = 120.0 From 32b08f0e30d68b4de6ed266cefadbd3d796ce3d3 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 20 Sep 2026 14:11:02 -0700 Subject: [PATCH 131/170] Use deterministic producer samples in CDC placement coverage --- tests/fast/NativeCdcInitialPlacement.toml | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/tests/fast/NativeCdcInitialPlacement.toml b/tests/fast/NativeCdcInitialPlacement.toml index d522e81ab69..86e0a9c0c3c 100644 --- a/tests/fast/NativeCdcInitialPlacement.toml +++ b/tests/fast/NativeCdcInitialPlacement.toml @@ -14,10 +14,15 @@ native_cdc_tag_balancing_enabled = true native_cdc_tag_sample_interval = 0.5 native_cdc_tag_sample_timeout = 5.0 native_cdc_tag_sample_max_age = 30.0 +# Keep producer measurement deterministic within this focused test window. +storage_metrics_average_interval = 1.0 +storage_metrics_average_interval_per_kseconds = 1000.0 +bytes_written_units_per_sample = 1 [[test]] testTitle = 'NativeCdcInitialPlacement' useDB = true +runFailureWorkloads = false waitForQuiescenceEnd = false connectionFailuresDisableDuration = 1000000 timeout = 300 From faf1e5b8215944902b7d42d42e4e3273aaaa2061 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 20 Sep 2026 14:23:12 -0700 Subject: [PATCH 132/170] Check placement pool after client metadata is established --- fdbserver/workloads/NativeCdcInitialPlacement.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fdbserver/workloads/NativeCdcInitialPlacement.cpp b/fdbserver/workloads/NativeCdcInitialPlacement.cpp index 25ffdc9b55b..7b54b615c4f 100644 --- a/fdbserver/workloads/NativeCdcInitialPlacement.cpp +++ b/fdbserver/workloads/NativeCdcInitialPlacement.cpp @@ -139,10 +139,10 @@ class NativeCdcInitialPlacementWorkload : public TestWorkload { } Future run(Database cx) { - ASSERT_EQ(cx->clientInfo->get().nativeCdcTagCount, 2); const KeyRange coldRange(KeyRangeRef(coldKey, keyAfter(coldKey))); const KeyRange hotRange(KeyRangeRef(hotKey, keyAfter(hotKey))); const CDCStreamId cold = co_await registerNativeCdcStreamClient(cx, coldName, { coldRange }); + ASSERT_EQ(cx->clientInfo->get().nativeCdcTagCount, 2); const CDCStreamId hot = co_await registerNativeCdcStreamClient(cx, hotName, { hotRange }); const CDCStreamId duplicate = co_await registerNativeCdcStreamClient(cx, duplicateName, { coldRange }); const Tag coldTag = co_await readTag(cx, cold); From 305e5f3d69cee972e0d747dac8bc09f46fd3d997 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 20 Sep 2026 18:59:16 -0700 Subject: [PATCH 133/170] Front-load native CDC retag compatibility and cleanup --- design/cdc.md | 116 +-- fdbclient/SystemData.cpp | 63 +- fdbclient/include/fdbclient/SystemData.h | 20 +- fdbserver/cdcproxy/CDCProxy.cpp | 502 +++++++++++-- .../clustercontroller/ClusterController.cpp | 1 - .../clustercontroller/ClusterRecovery.cpp | 2 +- fdbserver/core/NativeCdcMetadata.cpp | 153 ++-- fdbserver/core/ServerKnobs.cpp | 9 +- fdbserver/core/include/fdbserver/core/Knobs.h | 9 +- .../fdbserver/core/NativeCdcMetadata.h | 11 +- .../datadistributor/DataDistribution.cpp | 4 +- .../datadistributor/NativeCdcBalancer.cpp | 397 ----------- .../datadistributor/NativeCdcRetagCleanup.cpp | 207 ++++++ ...eCdcBalancer.h => NativeCdcRetagCleanup.h} | 15 +- fdbserver/logsystem/ApplyMetadataMutation.cpp | 2 +- fdbserver/logsystem/CDCRoutingTable.cpp | 2 +- fdbserver/workloads/CMakeLists.txt | 3 +- fdbserver/workloads/NativeCdcEndToEnd.cpp | 674 +++++++++++++++++- .../workloads/NativeCdcInitialPlacement.cpp | 198 ----- fdbserver/workloads/UnitTests.cpp | 4 +- tests/CMakeLists.txt | 6 +- tests/fast/NativeCdcInitialPlacement.toml | 32 - tests/fast/NativeCdcRetagCompatibility.toml | 38 + .../fast/NativeCdcRetagDisableRestart-1.toml | 33 + .../fast/NativeCdcRetagDisableRestart-2.toml | 28 + tests/fast/NativeCdcRetaggingMemoryBound.toml | 38 + 26 files changed, 1653 insertions(+), 914 deletions(-) delete mode 100644 fdbserver/datadistributor/NativeCdcBalancer.cpp create mode 100644 fdbserver/datadistributor/NativeCdcRetagCleanup.cpp rename fdbserver/datadistributor/{NativeCdcBalancer.h => NativeCdcRetagCleanup.h} (63%) delete mode 100644 fdbserver/workloads/NativeCdcInitialPlacement.cpp delete mode 100644 tests/fast/NativeCdcInitialPlacement.toml create mode 100644 tests/fast/NativeCdcRetagCompatibility.toml create mode 100644 tests/fast/NativeCdcRetagDisableRestart-1.toml create mode 100644 tests/fast/NativeCdcRetagDisableRestart-2.toml create mode 100644 tests/fast/NativeCdcRetaggingMemoryBound.toml diff --git a/design/cdc.md b/design/cdc.md index 86078d1cd03..60addca8da1 100644 --- a/design/cdc.md +++ b/design/cdc.md @@ -351,20 +351,27 @@ in transaction state: | `\xff/cdc/name/` | `CDCStreamId` | Resolves a user-visible name to its durable stream identity. | | `\xff/cdc/maxStreamId` | `CDCStreamId` | Allocates monotonic stream identifiers. | | `\xff/cdc/keys/` | `std::vector` | Stores the canonical immutable registered range set for an active stream. | -| `\xff/cdc/tagHistory///` | empty | Records the CDC tag assignment history used for routing and historical reads. | +| `\xff/cdc/tagHistory///` | empty or commit versionstamp | Records initial assignments and exact committed live-retag boundaries. | | `\xff/cdc/proxies//` | empty | Stores the CDC proxy assigned to an active stream. | | `\xff/cdc/proxyAssignmentChange` | version/change signal | Wakes ownership monitoring when durable assignments change. | | `\xff/cdc/retiredTagPop/` | empty | Retains recovery-visible pending final-pop work after removal. | Tag history is versioned so the data model can support a stream moving between tags without forgetting which old log streams may still contain unread -mutations. The initial implementation writes the initial assignment and reads -the history; dynamic throughput-driven reassignment is future work. The initial -history entry uses the registration transaction's read version as a +mutations. Production writers currently create only the initial assignment; +throughput-driven reassignment is future work. The initial history entry uses +the registration transaction's read version as a conservative inclusive lower bound. The versionstamped `minVersion` uses the commit version, and the proxy starts at the maximum of those two values, so the earlier history boundary cannot expose pre-registration mutations. +A live retag preserves that key layout, but writes a ten-byte commit +versionstamp in the value. Its key uses the transaction read version only to +order successive assignments; readers use the committed version from the value +as the exact cutover. Empty values retain their original interpretation. A +retag transaction reads and revalidates its previous history, so its read version +is later than that history key and the key order remains monotonic. + ### Storage-backed system data These keys are in the storage-server-backed `\xff\x02` system key range rather @@ -374,7 +381,6 @@ than transaction state: | --- | --- | --- | | `\xff\x02/cdc/minVersion/` | `Version` | Earliest version that an active stream may still require. | | `\xff\x02/cdc/retiredTagPopVersion/` | `Version` | Final pop watermark required after a stream using a tag is removed. | -| `\xff\x02/cdc/tagLoad/` | assignment generation, sample version, expiry version, sampled write rate | Advisory producer-load estimate used for placement. | | `\xff\x02/cdc/tagOwner/` | `CDCStreamId` | Derived representative stream used to look up a current tag's proxy owner. | The initial `minVersion` is written with a versionstamp at stream @@ -398,11 +404,9 @@ Registration runs as a durable metadata transaction: same-name/same-range-set rule, even when admission is disabled. 3. For a new name, it validates the feature knob. 4. It allocates a new monotonically increasing `CDCStreamId`. -5. It selects a CDC tag from `NATIVE_CDC_TAG_COUNT` tags (256 by default). - With fresh, complete write-load samples for the current assignment generation, - it chooses the least-loaded tag, breaking ties by stream count then tag ID. - Otherwise it uses the least populated tag, breaking ties by tag ID. An unused - tag has zero current load; an occupied tag without a sample has unknown load. +5. It selects a CDC tag using current active stream counts. The allocator uses + the least populated tag among `NATIVE_CDC_TAG_COUNT` tags (256 by default), + choosing the lowest tag ID on a tie. 6. It records the stream name, canonical ranges, initial tag history entry, and versionstamped initial minimum version. 7. It records an available CDC proxy owner and signals assignment monitoring. @@ -431,33 +435,31 @@ validation also rejects stale representatives left by older metadata writers. The allocator's stream-count scan remains necessary, and ownership discovery still scans global metadata when the representative is absent or invalid. -### Throughput-aware initial placement - -When native CDC admission and `NATIVE_CDC_TAG_BALANCING_ENABLED` are enabled, -the data distributor samples producer writes using storage-server range metrics. -Both sampling and load-aware registration are advisory: existing streams retain -their original tags. The sampling knob defaults to enabled; native CDC admission -remains disabled by default. - -Registered ranges are divided into disjoint segments, and each tag is counted -once per segment so overlapping streams do not inflate a shared tag's load. -The model bounds projected coverage entries by -`2 * streamCount * totalRangeCount` before construction. The default -`NATIVE_CDC_TAG_MODEL_MAX_ENTRIES` budget is 2,000,000 entries; this is a -conservative entry bound, not an exact resident-memory bound. - -Sampling also limits streams, shards, request concurrency, and elapsed time. -The default interval is 30 seconds and sample lifetime is 90 seconds. Partial, -failed, expired, or previous-generation samples never imply zero load. -Publication revalidates both the assignment generation and data-distributor -lock. Registration, removal, and proxy ownership changes invalidate existing -comparisons. Registrations use stream counts until every occupied eligible tag -has a fresh sample from the current generation. - -`NativeCdcInitialPlacement` drives real producer writes, verifies a colder tag -wins despite having more streams, and checks delivery through the new stream. -Storage metrics remain write-cost estimates with their existing sampling and -range-clear attribution semantics; they do not measure exact tagged TLog bytes. +### Retag compatibility and cleanup + +A committed target history row at version `C` divides delivery into the old tag +below `C` and the target tag starting at `C`. Consumers refresh history even +when the owner has not changed. Delivery is bounded by the metadata snapshot's +read version, so an old-tag read cannot skip across a cutover that committed +after that snapshot. Existing unacknowledged delivery positions remain valid +on the same owner. Recovery retains both required tagged log intervals. + +Each stream has at most one pending move. Both history rows remain until its +durable minimum required version reaches `C`. The data distributor then +atomically replaces them with one canonical empty-valued target row at `C` +and records retired-pop work for the old tag. Shared-tag acknowledgement, +recovery, and final-pop checks still govern physical cleanup. + +The cleanup worker pages through durable streams independently of admission +or any future sampling/move policy. `NATIVE_CDC_RETAG_CLEANUP_INTERVAL` defaults +to 30 seconds. Assignment changes trigger another traversal without restarting +an in-progress traversal, and pending histories are revisited even when an +acknowledgement notification is lost. Disabling CDC admission still allows +existing streams and transitions to drain or be removed. + +This compatibility foundation does not schedule new retags. Its transactional +writer helper is exercised by simulation fixtures to create future-format +histories, including writes between the metadata read and commit versions. ### Metadata lifecycle example @@ -806,6 +808,17 @@ rollback must keep CDC-capable binaries available until those records have been consumed or removed and retired cleanup has completed. Disabling the knob stops new allocation but is not a rollback mechanism for already durable CDC state. +Retag activation requires every process that may serve or recover CDC to +understand commit-stamped history values. The original `withNativeCdc` +capability alone does not establish this. This foundation can read, recover, +and finalize those values without enabling a production writer, so a later +patch can add move policy while preserving rollback to the foundation. +Rolling back to a binary without this foundation requires disabling new moves, +acknowledging or removing streams with pending transitions, and verifying that +all retained history rows have canonical empty values. Keep compatible +replacement binaries available until that state is verified. Disabling a +writer does not itself make older readers safe. + ## Correctness properties The implementation is structured around the following properties: @@ -839,9 +852,9 @@ The design records tag history and proxy ownership in forms that support more complete load balancing, but the first implementation intentionally keeps policy simple. -* Producer-write metrics are estimates with the storage-metrics sampling window. - Missing or stale comparisons fall back to active stream counts. Initial - placement does not rebalance existing streams after their loads change. +* Tag selection is based on active stream counts, not observed byte or mutation + throughput. Data distribution could make equally counted tags very + different in cost. * Registration selects an available CDC proxy without balancing aggregate proxy throughput, buffer memory, lag, or number of active readers. * Assignment mutations use one coalescing change key that wakes a full durable @@ -854,8 +867,9 @@ policy simple. size and lifetime limits until these paths are sharded or incrementally maintained. * There is no background process that changes a live stream's CDC tag in - response to load. A future implementation can use versioned tag history to - make such changes without losing the ability to read earlier tagged data. + response to load. Commit-stamped history readers, recovery, and cleanup are + present so a future policy can make such changes without losing earlier + tagged data. * The CDC client surface does not yet provide language-specific bindings beyond the C and Python APIs or a higher-level consumer checkpoint abstraction. Administrative status and identity-guarded removal are available through `fdbcli`. @@ -926,7 +940,20 @@ Unit coverage checks the CDC recovery-recruitment truth table with the feature enabled and disabled, both with and without durable CDC state. The process-static knob transition is covered by a paired restart simulation that creates durable CDC work while enabled, restarts with registration disabled, and drains the -existing stream and its retired state. +existing stream and its retired state. A separate paired restart preserves a +pending retag, consumes exact-version mutations from both sides of the cutover, +and verifies acknowledgement-driven finalization with admission disabled. +The same pair can run the writer phase with a newer binary and the drain phase +with the compatibility foundation to test persisted-state rollback. + +`NativeCdcRetagCompatibility` commits synthetic retags around intervening user +writes, verifies replay after proxy replacement and transaction-system recovery, +and checks shared-tag retention and returning to an old tag. It covers the +reader/cleanup contract without a load-driven move policy. +`NativeCdcRetaggingMemoryBound` verifies a pending retag with a 4.5 KiB proxy +budget and competing reader reservations, then acknowledges and checks history +finalization and retired cleanup. Neither fixture qualifies simultaneous +mixed-version processes or throughput under balancing. The shared-tag workload forces streams to share routing tags and verifies both range filtering and acknowledgement coordination. In particular, removing one @@ -936,7 +963,8 @@ mutations needed by the remaining consumer. The simulation configurations enable CDC explicitly when testing these behaviors, while the default-disabled knob and randomized simulation admission exercise the requirement that clusters without active or pending CDC work do -not carry CDC service overhead. +not recruit CDC proxies or retain CDC TLog tags. The data distributor retains +a low-rate metadata-generation check for pending-history finalization. ## Observability and supportability considerations diff --git a/fdbclient/SystemData.cpp b/fdbclient/SystemData.cpp index 6a1009db9e3..7b9e742356b 100644 --- a/fdbclient/SystemData.cpp +++ b/fdbclient/SystemData.cpp @@ -787,7 +787,6 @@ const KeyRangeRef cdcStreamNameKeys("\xff/cdc/name/"_sr, "\xff/cdc/name0"_sr); const KeyRef cdcMaxStreamIdKey = "\xff/cdc/maxStreamId"_sr; const KeyRangeRef cdcStreamKeys("\xff/cdc/keys/"_sr, "\xff/cdc/keys0"_sr); const KeyRangeRef cdcTagHistoryKeys("\xff/cdc/tagHistory/"_sr, "\xff/cdc/tagHistory0"_sr); -const KeyRangeRef cdcTagLoadKeys("\xff\x02/cdc/tagLoad/"_sr, "\xff\x02/cdc/tagLoad0"_sr); const KeyRangeRef cdcTagOwnerKeys("\xff\x02/cdc/tagOwner/"_sr, "\xff\x02/cdc/tagOwner0"_sr); const KeyRangeRef cdcMinVersionKeys("\xff\x02/cdc/minVersion/"_sr, "\xff\x02/cdc/minVersion0"_sr); const KeyRangeRef cdcRetiredTagPopKeys("\xff/cdc/retiredTagPop/"_sr, "\xff/cdc/retiredTagPop0"_sr); @@ -886,32 +885,19 @@ CDCTagHistoryEntry decodeCDCTagHistoryKey(KeyRef const& key) { return CDCTagHistoryEntry(streamId, bigEndian64(encodedVersion), tag); } -Key cdcTagLoadKeyFor(Tag tag) { - BinaryWriter wr(Unversioned()); - wr.serializeBytes(cdcTagLoadKeys.begin); - wr << tag; - return wr.toValue(); -} - -Tag decodeCDCTagLoadKey(KeyRef const& key) { - Tag tag; - BinaryReader reader(key.removePrefix(cdcTagLoadKeys.begin), Unversioned()); - reader >> tag; - return tag; -} - -Value cdcTagLoadValue(CDCTagLoadSample const& sample) { - BinaryWriter wr(IncludeVersion(ProtocolVersion::withNativeCdc())); - wr << sample.assignmentChange << sample.sampleVersion << sample.validThrough << sample.bytesWrittenPerKSecond; - return wr.toValue(); -} - -CDCTagLoadSample decodeCDCTagLoadValue(ValueRef const& value) { - CDCTagLoadSample sample; - BinaryReader reader(value, IncludeVersion()); - ASSERT_WE_THINK(reader.protocolVersion().hasNativeCdc()); - reader >> sample.assignmentChange >> sample.sampleVersion >> sample.validThrough >> sample.bytesWrittenPerKSecond; - return sample; +CDCTagHistoryEntry decodeCDCTagHistoryEntry(KeyRef const& key, ValueRef const& value) { + CDCTagHistoryEntry result = decodeCDCTagHistoryKey(key); + if (!value.empty()) { + if (value.size() != sizeof(Version) + sizeof(uint16_t)) { + throw serialization_failed(); + } + const Version committedVersion = decodeCDCMinVersionValue(value); + if (committedVersion <= result.version) { + throw serialization_failed(); + } + result.version = committedVersion; + } + return result; } Key cdcTagOwnerKeyFor(Tag tag) { @@ -2010,15 +1996,20 @@ TEST_CASE("/SystemData/NativeCDC") { const Key laterTagHistoryKey = cdcTagHistoryKeyFor(streamId, 256, Tag(tagLocalityCDC, 0)); ASSERT(earlierTagHistoryKey < laterTagHistoryKey); ASSERT(cdcTagHistoryRangeFor(streamId).contains(laterTagHistoryKey)); - - const CDCTagLoadSample sample{ "assignment"_sr, minVersion, minVersion + 100, 123000 }; - const CDCTagLoadSample decodedSample = decodeCDCTagLoadValue(cdcTagLoadValue(sample)); - ASSERT_EQ(decodeCDCTagLoadKey(cdcTagLoadKeyFor(tag)), tag); - ASSERT(nonMetadataSystemKeys.contains(cdcTagLoadKeyFor(tag))); - ASSERT_EQ(decodedSample.assignmentChange, sample.assignmentChange); - ASSERT_EQ(decodedSample.sampleVersion, sample.sampleVersion); - ASSERT_EQ(decodedSample.validThrough, sample.validThrough); - ASSERT_EQ(decodedSample.bytesWrittenPerKSecond, sample.bytesWrittenPerKSecond); + ASSERT_EQ(decodeCDCTagHistoryEntry(tagHistoryKey, ValueRef()).version, minVersion); + const Value committedBoundary = BinaryWriter::toValue(Versionstamp(minVersion + 20, 3), Unversioned()); + const CDCTagHistoryEntry committedHistory = decodeCDCTagHistoryEntry(tagHistoryKey, committedBoundary); + ASSERT_EQ(committedHistory.version, minVersion + 20); + ASSERT_EQ(committedHistory.tag, tag); + ASSERT_EQ(committedHistory.streamId, streamId); + bool invalidBoundaryRejected = false; + try { + decodeCDCTagHistoryEntry(tagHistoryKey, BinaryWriter::toValue(Versionstamp(minVersion, 0), Unversioned())); + } catch (Error& e) { + ASSERT_EQ(e.code(), error_code_serialization_failed); + invalidBoundaryRejected = true; + } + ASSERT(invalidBoundaryRejected); const Value serializedTagHistory = ObjectWriter::toValue(decodedTagHistory, Unversioned()); const auto deserializedTagHistory = diff --git a/fdbclient/include/fdbclient/SystemData.h b/fdbclient/include/fdbclient/SystemData.h index a437e3e7c8b..a1100ea2f3a 100644 --- a/fdbclient/include/fdbclient/SystemData.h +++ b/fdbclient/include/fdbclient/SystemData.h @@ -300,7 +300,9 @@ CDCStreamId decodeCDCStreamKey(KeyRef const& key); Value cdcStreamKeysValue(std::vector const& ranges); std::vector decodeCDCStreamKeysValue(ValueRef const& value); -// "\xff/cdc/tagHistory/[[CDCStreamId]][[Version]][[Tag]]" := "" +// "\xff/cdc/tagHistory/[[CDCStreamId]][[Version]][[Tag]]" := "" | commit versionstamp +// Empty values use the key's version. A pending live retag stores its exact commit +// boundary in the value; the key retains the transaction read version for ordering. struct CDCTagHistoryEntry { constexpr static FileIdentifier file_identifier = 13091844; @@ -322,21 +324,7 @@ extern const KeyRangeRef cdcTagHistoryKeys; Key cdcTagHistoryKeyFor(CDCStreamId streamId, Version version, Tag tag); KeyRange cdcTagHistoryRangeFor(CDCStreamId streamId); CDCTagHistoryEntry decodeCDCTagHistoryKey(KeyRef const& key); - -// Advisory producer-write samples. The assignment generation invalidates every -// comparison when registrations, tag histories, or durable ownership change. -struct CDCTagLoadSample { - Value assignmentChange; - Version sampleVersion = invalidVersion; - Version validThrough = invalidVersion; - int64_t bytesWrittenPerKSecond = 0; -}; - -extern const KeyRangeRef cdcTagLoadKeys; -Key cdcTagLoadKeyFor(Tag tag); -Tag decodeCDCTagLoadKey(KeyRef const& key); -Value cdcTagLoadValue(CDCTagLoadSample const& sample); -CDCTagLoadSample decodeCDCTagLoadValue(ValueRef const& value); +CDCTagHistoryEntry decodeCDCTagHistoryEntry(KeyRef const& key, ValueRef const& value); // "\xff\x02/cdc/tagOwner/[[Tag]]" := "[[CDCStreamId]]" // Derived lookup hint, not authoritative ownership. Validate the stream is active diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index 44f7962c241..81b85d77fb6 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -159,8 +159,10 @@ struct CDCBufferedStream : ReferenceCounted { bool bufferLimitExceeded = false; Version minVersion = invalidVersion; Version bufferedThrough = invalidVersion; + Version metadataReadVersion = invalidVersion; int64_t bufferedBytes = 0; int readDemand = 0; + std::vector> tagAssignments; Reference activeConsume; CDCStreamReadAhead readAhead; std::vector tagIntervals; @@ -199,6 +201,9 @@ Optional nextCDCPrefetchVersion(Reference const& str return {}; } const Version next = std::max(stream->minVersion, stream->bufferedThrough + 1); + if (next > stream->metadataReadVersion) { + return {}; + } for (auto const& interval : stream->tagIntervals) { if (interval.begin <= next && next < interval.end) { return interval.tag == tag->tag ? Optional(next) : Optional(); @@ -361,6 +366,12 @@ Version selectedCDCConsumeReplyThrough(CDCConsumeReplySelection const& selection return selection.firstExcludedVersion.present() ? selection.firstExcludedVersion.get() - 1 : bufferedThrough; } +Version boundedCDCConsumeReplyThrough(Version lastConsumedVersion, Version readVersion, Version bufferedThrough) { + // A later cutover may change the tag for versions beyond this metadata snapshot. An already delivered cursor + // remains valid even when a subsequent metadata read temporarily trails it. + return std::max(lastConsumedVersion, std::min(readVersion, bufferedThrough)); +} + // A transactionally consistent durable-watermark snapshot consumed by one acknowledged-data pop pass. struct CDCPopState { std::unordered_map minVersions; @@ -530,8 +541,11 @@ class CDCProxy { void clearBufferedMutations(Reference stream); void addBufferedBatch(Reference stream, CDCBufferedBatch batch); void reconcileStreamMinVersion(Reference stream, Version minVersion); + void reconcileStreamMetadata(Reference stream, CDCStreamReadState const& metadata); void markTagStreamsBufferLimitExceeded(Reference tag, Version begin, int replyByteLimit); void markTagStreamsRawReplyBudgetExceeded(Reference tag, Version begin, int64_t retainedReplyCount); + void attachStreamToTags(Reference stream); + void detachStreamFromTags(CDCStreamId streamId, std::vector const& intervals); void detachStreamFromTags(Reference stream); void deactivateStream(Reference stream); void refreshStreamTags(Reference stream); @@ -576,7 +590,6 @@ class CDCProxy { int64_t rawPeekReservation, FlowLock::Releaser& reservation, int64_t bufferLimit, - Prefetch prefetch, Future invalidated); Future rotateContendedPeek(); Future bufferTagPass(Reference tag, Version begin, Prefetch prefetch); @@ -694,7 +707,7 @@ AsyncResult readCDCStreamState(Database cx, while (begin < tagHistoryRange.end) { RangeResult history = co_await historyFuture; for (KeyValueRef const& kv : history) { - const CDCTagHistoryEntry historyEntry = decodeCDCTagHistoryKey(kv.key); + const CDCTagHistoryEntry historyEntry = decodeCDCTagHistoryEntry(kv.key, kv.value); ASSERT_WE_THINK(historyEntry.streamId == streamId); ASSERT_WE_THINK(historyEntry.tag.locality == tagLocalityCDC); tagAssignments.emplace_back(historyEntry.version, historyEntry.tag); @@ -870,9 +883,44 @@ void CDCProxy::markTagStreamsRawReplyBudgetExceeded(Reference ta } } -void updateStreamBufferedThrough(Reference stream) { - Version bufferedThrough = stream->minVersion - 1; - for (const auto& interval : stream->tagIntervals) { +Optional firstIncompleteTagInterval(CDCBufferedStream const& stream) { + for (size_t i = 0; i < stream.tagIntervals.size(); ++i) { + if (stream.tagIntervals[i].bufferedThrough < stream.tagIntervals[i].end - 1) { + return i; + } + } + return Optional(); +} + +Optional eligibleTagReadInterval(CDCBufferedStream const& stream, Tag const& tag) { + if (!stream.active || !stream.initialized || stream.bufferLimitExceeded) { + return Optional(); + } + const Optional first = firstIncompleteTagInterval(stream); + if (!first.present()) { + return Optional(); + } + const auto& interval = stream.tagIntervals[first.get()]; + // A later interval cannot release its buffered data until the unread prefix has been delivered. Letting it + // compete for capacity can fill the buffer with that undeliverable tail and prevent the prefix from ever reading. + if (interval.tag != tag || std::max(interval.begin, interval.bufferedThrough + 1) > stream.metadataReadVersion) { + return Optional(); + } + return first; +} + +bool canBufferTagVersion(CDCBufferedStream const& stream, Tag const& tag, Version version) { + const Optional eligible = eligibleTagReadInterval(stream, tag); + if (!eligible.present() || version > stream.metadataReadVersion) { + return false; + } + const auto& interval = stream.tagIntervals[eligible.get()]; + return interval.begin <= version && version < interval.end && version > interval.bufferedThrough; +} + +Version contiguousStreamBufferedThrough(CDCBufferedStream const& stream) { + Version bufferedThrough = stream.minVersion - 1; + for (const auto& interval : stream.tagIntervals) { if (interval.begin > bufferedThrough + 1) { break; } @@ -884,12 +932,30 @@ void updateStreamBufferedThrough(Reference stream) { break; } } + return bufferedThrough; +} + +void updateStreamBufferedThrough(Reference stream) { + const Version bufferedThrough = contiguousStreamBufferedThrough(*stream); if (bufferedThrough > stream->bufferedThrough) { stream->bufferedThrough = bufferedThrough; stream->changed.trigger(); } } +bool advanceStreamTagBufferedThrough(Reference stream, Tag const& tag, Version throughVersion) { + const Optional eligible = eligibleTagReadInterval(*stream, tag); + if (!eligible.present()) { + return false; + } + auto& interval = stream->tagIntervals[eligible.get()]; + interval.bufferedThrough = + std::max(interval.bufferedThrough, std::min({ throughVersion, stream->metadataReadVersion, interval.end - 1 })); + const bool intervalCompleted = interval.bufferedThrough == interval.end - 1; + updateStreamBufferedThrough(stream); + return intervalCompleted; +} + void advanceStreamMinVersion(Reference stream, Version minVersion) { stream->minVersion = std::max(stream->minVersion, minVersion); for (auto& interval : stream->tagIntervals) { @@ -901,7 +967,104 @@ void advanceStreamMinVersion(Reference stream, Version minVer updateStreamBufferedThrough(stream); } +struct CDCStreamMetadataUpdate { + bool historyChanged = false; + bool readVersionAdvanced = false; + int64_t releasedBytes = 0; +}; + +CDCStreamMetadataUpdate reconcileBufferedStreamMetadata(Reference stream, + CDCStreamReadState const& metadata) { + CDCStreamMetadataUpdate update; + stream->minVersion = std::max(stream->minVersion, metadata.minVersion); + std::vector previousIntervals; + if (metadata.readVersion >= stream->metadataReadVersion) { + update.readVersionAdvanced = metadata.readVersion > stream->metadataReadVersion; + stream->metadataReadVersion = metadata.readVersion; + stream->ranges = metadata.ranges; + update.historyChanged = stream->tagAssignments != metadata.tagAssignments; + if (update.historyChanged) { + previousIntervals = std::move(stream->tagIntervals); + stream->tagIntervals.clear(); + for (size_t i = 0; i < metadata.tagAssignments.size(); ++i) { + const Version begin = std::max(stream->minVersion, metadata.tagAssignments[i].first); + const Version end = i + 1 < metadata.tagAssignments.size() ? metadata.tagAssignments[i + 1].first + : std::numeric_limits::max(); + if (begin >= end) { + continue; + } + CDCTagInterval interval(metadata.tagAssignments[i].second, begin, end); + for (const auto& previous : previousIntervals) { + if (previous.tag == interval.tag && previous.begin <= begin && begin < previous.end) { + interval.bufferedThrough = + std::max(interval.bufferedThrough, std::min(previous.bufferedThrough, end - 1)); + } + } + stream->tagIntervals.push_back(interval); + } + stream->tagAssignments = metadata.tagAssignments; + } + } + + for (auto& interval : stream->tagIntervals) { + interval.bufferedThrough = + std::max(interval.bufferedThrough, std::min(stream->minVersion - 1, interval.end - 1)); + } + for (auto buffered = stream->mutations.begin(); buffered != stream->mutations.end();) { + const Version version = buffered->version; + if (!update.historyChanged && version >= stream->minVersion) { + break; + } + const bool sameTag = + !update.historyChanged || + std::any_of(stream->tagIntervals.begin(), stream->tagIntervals.end(), [&](const auto& interval) { + return interval.begin <= version && version < interval.end && + std::any_of(previousIntervals.begin(), previousIntervals.end(), [&](const auto& previous) { + return previous.tag == interval.tag && previous.begin <= version && version < previous.end; + }); + }); + if (version < stream->minVersion || !sameTag) { + update.releasedBytes += estimatedCDCConsumeVersionBytes(*buffered); + buffered = stream->mutations.erase(buffered); + } else { + ++buffered; + } + } + stream->bufferedBytes -= update.releasedBytes; + ASSERT_GE(stream->bufferedBytes, 0); + // A pre-cutover tag may have speculatively buffered farther than its newly discovered end. No reply can have + // delivered that tail: replies are bounded by the metadata snapshot that established their routing history. + stream->bufferedThrough = contiguousStreamBufferedThrough(*stream); + stream->initialized = true; + return update; +} + +void CDCProxy::reconcileStreamMetadata(Reference stream, CDCStreamReadState const& metadata) { + const std::vector previousIntervals = stream->tagIntervals; + const Version previousBufferedThrough = stream->bufferedThrough; + const Optional previousReadInterval = firstIncompleteTagInterval(*stream); + const CDCStreamMetadataUpdate update = reconcileBufferedStreamMetadata(stream, metadata); + if (update.historyChanged) { + CODE_PROBE(!previousIntervals.empty(), "CDC proxy refreshes same-owner tag history"); + detachStreamFromTags(stream->streamId, previousIntervals); + attachStreamToTags(stream); + } + ASSERT_GE(bufferedBytes, update.releasedBytes); + bufferedBytes -= update.releasedBytes; + if (update.releasedBytes > 0) { + bufferLock.release(update.releasedBytes); + } + if (update.historyChanged || update.readVersionAdvanced || + previousReadInterval != firstIncompleteTagInterval(*stream)) { + refreshStreamTags(stream); + } + if (update.historyChanged || previousBufferedThrough != stream->bufferedThrough) { + stream->changed.trigger(); + } +} + void CDCProxy::reconcileStreamMinVersion(Reference stream, Version minVersion) { + const Optional previousReadInterval = firstIncompleteTagInterval(*stream); advanceStreamMinVersion(stream, minVersion); while (!stream->mutations.empty() && stream->mutations.front().version < minVersion) { const int64_t releasedBytes = @@ -913,16 +1076,35 @@ void CDCProxy::reconcileStreamMinVersion(Reference stream, Ve stream->mutations.pop_front(); } ASSERT_GE(stream->bufferedBytes, 0); + if (previousReadInterval != firstIncompleteTagInterval(*stream)) { + refreshStreamTags(stream); + } } -void CDCProxy::detachStreamFromTags(Reference stream) { +void CDCProxy::attachStreamToTags(Reference stream) { for (const auto& interval : stream->tagIntervals) { + auto tag = tags.find(interval.tag); + if (tag == tags.end()) { + auto newTag = makeReference(interval.tag); + tag = tags.emplace(interval.tag, newTag).first; + tag->second->streamIds.insert(stream->streamId); + actors.add(bufferTag(newTag)); + } else { + CODE_PROBE(true, "CDC proxy shares a tag reader across streams"); + tag->second->streamIds.insert(stream->streamId); + tag->second->refresh.trigger(); + } + } +} + +void CDCProxy::detachStreamFromTags(CDCStreamId streamId, std::vector const& intervals) { + for (const auto& interval : intervals) { auto tag = tags.find(interval.tag); if (tag == tags.end()) { continue; } Reference bufferedTag = tag->second; - bufferedTag->streamIds.erase(stream->streamId); + bufferedTag->streamIds.erase(streamId); if (bufferedTag->streamIds.empty()) { bufferedTag->active = false; tags.erase(tag); @@ -933,6 +1115,10 @@ void CDCProxy::detachStreamFromTags(Reference stream) { } } +void CDCProxy::detachStreamFromTags(Reference stream) { + detachStreamFromTags(stream->streamId, stream->tagIntervals); +} + void CDCProxy::deactivateStream(Reference stream) { CODE_PROBE(stream->readDemand > 0, "CDC proxy wakes pending consume when stream is unassigned"); CODE_PROBE(true, "CDC proxy drops removed or reassigned stream state"); @@ -1033,21 +1219,15 @@ void CDCProxy::changeStreamReadDemand(Reference stream, int d Optional CDCProxy::nextTagReadVersionForStream(Reference tag, Reference stream, bool ignoreReadInterest) { - if (!stream->active || !stream->initialized || stream->bufferLimitExceeded || - (!ignoreReadInterest && !hasCDCReadInterest(stream, tag))) { + if (!ignoreReadInterest && !hasCDCReadInterest(stream, tag)) { return Optional(); } - Optional begin; - for (const auto& interval : stream->tagIntervals) { - if (interval.tag != tag->tag) { - continue; - } - const Version next = std::max(interval.begin, interval.bufferedThrough + 1); - if (next < interval.end && (!begin.present() || next < begin.get())) { - begin = next; - } + const Optional eligible = eligibleTagReadInterval(*stream, tag->tag); + if (!eligible.present()) { + return Optional(); } - return begin; + const auto& interval = stream->tagIntervals[eligible.get()]; + return std::max(interval.begin, interval.bufferedThrough + 1); } Optional CDCProxy::nextTagReadVersion(Reference tag) { @@ -1091,13 +1271,10 @@ void CDCProxy::advanceTagBufferedThrough(Reference tag, if (stream == streams.end() || !stream->second->active || !hasCDCReadInterest(stream->second, tag)) { continue; } - for (auto& interval : stream->second->tagIntervals) { - if (interval.tag == tag->tag && bufferedThrough >= interval.begin) { - interval.bufferedThrough = - std::max(interval.bufferedThrough, std::min(bufferedThrough, interval.end - 1)); - } + Reference bufferedStream = stream->second; + if (advanceStreamTagBufferedThrough(bufferedStream, tag->tag, bufferedThrough)) { + refreshStreamTags(bufferedStream); } - updateStreamBufferedThrough(stream->second); } } @@ -1110,7 +1287,8 @@ void CDCProxy::markPoppedTagStreamsTooOld(Reference tag, Version } for (const auto& interval : stream->second->tagIntervals) { const Version next = std::max(interval.begin, interval.bufferedThrough + 1); - if (interval.tag == tag->tag && next < interval.end && next < popped) { + if (interval.tag == tag->tag && next < interval.end && next <= stream->second->metadataReadVersion && + next < popped) { tooOldStreams.push_back(stream->second); break; } @@ -1159,18 +1337,9 @@ void CDCProxy::visitBufferedMutations(Reference tag, continue; } auto stream = streams.find(streamId); - if (stream == streams.end() || !stream->second->active || !hasCDCReadInterest(stream->second, tag) || - !stream->second->ranges.present()) { - continue; - } - const bool coversVersion = - std::any_of(stream->second->tagIntervals.begin(), - stream->second->tagIntervals.end(), - [tag, messageVersion](const auto& interval) { - return interval.tag == tag->tag && interval.begin <= messageVersion && - messageVersion < interval.end && messageVersion > interval.bufferedThrough; - }); - if (!coversVersion) { + if (stream == streams.end() || !hasCDCReadInterest(stream->second, tag) || + !stream->second->ranges.present() || + !canBufferTagVersion(*stream->second, tag->tag, messageVersion)) { continue; } visitClippedCDCMutations(mutation, stream->second->ranges.get(), [&](MutationRef const& clipped) { @@ -1299,7 +1468,6 @@ Future CDCProxy::materializeBufferSelection(Reference invalidated) { const int64_t materializationReservation = reservation.remaining - rawPeekReservation; ASSERT_GE(materializationReservation, 0); @@ -1323,7 +1491,9 @@ Future CDCProxy::materializeBufferSelection(Referenceactive) { co_return CDCBufferTagPassResult::STOP; } - if (prefetch && invalidated.isReady()) { + // A metadata refresh can widen a stream's validated read window. Discard an estimate made before that refresh, + // including when capacity and the refresh become ready together. + if (invalidated.isReady()) { co_return CDCBufferTagPassResult::RETRY; } @@ -1393,9 +1563,8 @@ Future CDCProxy::bufferTagCursor(Reference prefetchDeadline = prefetch ? delay(SERVER_KNOBS->BLOCKING_PEEK_TIMEOUT) : Never(); - Future invalidated = - prefetch ? logSystemChanged || tag->refresh.onTrigger() || tag->stopped.onTrigger() || prefetchDeadline - : Never(); + Future tagChanged = tag->refresh.onTrigger(); + Future invalidated = logSystemChanged || tagChanged || tag->stopped.onTrigger() || prefetchDeadline; const int64_t retainedReplyCount = cursor->getMaxRetainedReplyCount(); // Leave one reply-sized window for filtered mutations instead of rejecting a topology whose maximum-sized // replies would consume the entire buffer. Every retained raw arena remains covered by the reservation. @@ -1423,7 +1592,7 @@ Future CDCProxy::bufferTagCursor(Referencestopped.onTrigger(), - tag->refresh.onTrigger()); + tagChanged); if (capacity.index() == 1 || capacity.index() == 3) { co_return CDCBufferTagPassResult::RETRY; } @@ -1433,7 +1602,7 @@ Future CDCProxy::bufferTagCursor(ReferencehasMessage()) { @@ -1443,7 +1612,7 @@ Future CDCProxy::bufferTagCursor(ReferencegetMore(TaskPriority::TLogPeekReply), logSystemChanged, tag->stopped.onTrigger(), - tag->refresh.onTrigger(), + tagChanged, rotateContendedPeek(), prefetchDeadline); if (result.index() == 1 || result.index() == 3 || result.index() == 4 || result.index() == 5) { @@ -1456,11 +1625,17 @@ Future CDCProxy::bufferTagCursor(Referenceactive) { + co_return CDCBufferTagPassResult::STOP; + } + if (invalidated.isReady()) { co_return CDCBufferTagPassResult::RETRY; } // A newly constructed replay cursor can already contain messages, especially after log-generation @@ -1492,7 +1667,7 @@ Future CDCProxy::bufferTagCursor(Reference CDCProxy::bufferTag(Reference tag) { @@ -1555,32 +1730,8 @@ Future CDCProxy::initializeStream(Reference stream) { CODE_PROBE(true, "CDC proxy discards stale stream initialization"); co_return; } - stream->ranges = metadata.ranges; - stream->minVersion = metadata.minVersion; - stream->bufferedThrough = metadata.minVersion - 1; - for (size_t i = 0; i < metadata.tagAssignments.size(); ++i) { - const Version begin = std::max(metadata.minVersion, metadata.tagAssignments[i].first); - const Version end = i + 1 < metadata.tagAssignments.size() ? metadata.tagAssignments[i + 1].first - : std::numeric_limits::max(); - if (begin < end) { - stream->tagIntervals.emplace_back(metadata.tagAssignments[i].second, begin, end); - } - } - stream->initialized = true; + reconcileStreamMetadata(stream, metadata); stream->changed.trigger(); - for (const auto& interval : stream->tagIntervals) { - auto tag = tags.find(interval.tag); - if (tag == tags.end()) { - auto newTag = makeReference(interval.tag); - tag = tags.emplace(interval.tag, newTag).first; - tag->second->streamIds.insert(stream->streamId); - actors.add(bufferTag(newTag)); - } else { - CODE_PROBE(true, "CDC proxy shares a tag reader across streams"); - tag->second->streamIds.insert(stream->streamId); - tag->second->refresh.trigger(); - } - } } catch (Error& e) { if (e.code() == error_code_client_invalid_operation || e.code() == error_code_wrong_shard_server) { clearBufferedMutations(stream); @@ -1622,7 +1773,7 @@ AsyncResult readPopState(Database cx) { RangeResult histories = co_await tr.getRange(KeyRangeRef(begin, cdcTagHistoryKeys.end), CLIENT_KNOBS->TOO_MANY); for (const auto& kv : histories) { - const CDCTagHistoryEntry history = decodeCDCTagHistoryKey(kv.key); + const CDCTagHistoryEntry history = decodeCDCTagHistoryEntry(kv.key, kv.value); auto minimum = result.minVersions.find(history.streamId); if (minimum == result.minVersions.end()) { continue; @@ -1917,8 +2068,14 @@ Future CDCProxy::consume(CDCConsumeRequest request) { Future CDCProxy::consumeReply(Reference stream, CDCCursor cursor) { const CDCStreamReadState metadata = co_await readCDCStreamState(cx, cursor.streamId, id, true, PrioritizeDrain::True); + if (stream->tooOld) { + throw transaction_too_old(); + } + if (!stream->active) { + throw wrong_shard_server(); + } CODE_PROBE(stream->minVersion < metadata.minVersion, "Native CDC consume reconciles a durable acknowledgement"); - reconcileStreamMinVersion(stream, metadata.minVersion); + reconcileStreamMetadata(stream, metadata); if (!stream->readAhead.provesCursor(cursor.lastConsumedVersion, stream->minVersion)) { // Prefetched data is not proof of delivery. A cursor must have been issued in a reply or covered by a // durable acknowledgement; otherwise it could skip unread data and manufacture more read-ahead work. @@ -1935,6 +2092,12 @@ Future CDCProxy::consumeReply(Reference stre throw transaction_too_old(); } + if (begin > std::max(cursor.lastConsumedVersion, metadata.readVersion)) { + CDCConsumeReply reply; + reply.lastConsumedVersion = cursor.lastConsumedVersion; + co_return reply; + } + auto buffered = co_await race(waitForBufferedVersion(stream, begin), delay(SERVER_KNOBS->CDC_PROXY_CONSUME_POLL_TIMEOUT)); if (buffered.index() == 1) { @@ -1958,11 +2121,13 @@ Future CDCProxy::consumeReply(Reference stre CDCConsumeReply reply; CDCConsumeReplySelection selection; + const Version replyThrough = + boundedCDCConsumeReplyThrough(cursor.lastConsumedVersion, metadata.readVersion, stream->bufferedThrough); for (const auto& versioned : stream->mutations) { if (versioned.version < begin) { continue; } - if (versioned.version > stream->bufferedThrough) { + if (versioned.version > replyThrough) { break; } if (!selectCDCConsumeReplyVersion(&selection, @@ -1984,7 +2149,7 @@ Future CDCProxy::consumeReply(Reference stre .detail("ReplyLimit", SERVER_KNOBS->CDC_PROXY_CONSUME_REPLY_BYTES); throw server_overloaded(); } - reply.lastConsumedVersion = selectedCDCConsumeReplyThrough(selection, stream->bufferedThrough); + reply.lastConsumedVersion = selectedCDCConsumeReplyThrough(selection, replyThrough); co_return reply; } @@ -2014,7 +2179,7 @@ Future CDCProxy::acknowledge(CDCAckRequest request) { // Reconcile the new owner's in-memory frontier to that already verified watermark. const Version minVersion = metadata.minVersion; CODE_PROBE(stream->minVersion < minVersion, "CDC proxy reconciles a durable stream acknowledgement"); - reconcileStreamMinVersion(stream, minVersion); + reconcileStreamMetadata(stream, metadata); requestAcknowledgedDataPop(); request.reply.send(Void()); } catch (Error& e) { @@ -2429,6 +2594,7 @@ class CDCProxyPrefetchTest { stream->initialized = true; stream->minVersion = 1; stream->bufferedThrough = 99; + stream->metadataReadVersion = 1000; stream->ranges = std::vector{ KeyRangeRef("a"_sr, "z"_sr) }; stream->tagIntervals.emplace_back(tag->tag, 1, 200); stream->tagIntervals.back().bufferedThrough = 99; @@ -2743,7 +2909,7 @@ class CDCProxyPrefetchTest { selection.selectedBytes = passLimit.preferredBufferedBytes + 1; auto cursor = makeReference(Void()); co_await test.proxy.materializeBufferSelection( - test.tag, cursor, 100, selection, passLimit.rawReplyBytes, reservation, limit, Prefetch::True, Never()); + test.tag, cursor, 100, selection, passLimit.rawReplyBytes, reservation, limit, Never()); ASSERT_EQ(test.proxy.bufferLock.waiters(), 0); ASSERT_EQ(test.proxy.bufferLock.activePermits(), limit); ASSERT(stream->mutations.empty()); @@ -2834,11 +3000,15 @@ TEST_CASE("/NativeCDC/PrefetchCreditTailAndTags") { stream->initialized = true; stream->minVersion = 1; stream->bufferedThrough = 99; + stream->metadataReadVersion = 1000; stream->tagIntervals.emplace_back(first->tag, 1, 101); stream->tagIntervals.emplace_back(second->tag, 101, 200); stream->tagIntervals[0].bufferedThrough = 99; ASSERT(stream->readAhead.issueReply(99, 99, 1, HasMutations::True)); ASSERT_EQ(nextCDCPrefetchVersion(stream, first).get(), 100); + stream->metadataReadVersion = 99; + ASSERT(!nextCDCPrefetchVersion(stream, first).present()); + stream->metadataReadVersion = 1000; ASSERT(!nextCDCPrefetchVersion(stream, second).present()); { CDCReadAheadPass pass(first); @@ -3252,6 +3422,180 @@ TEST_CASE("/NativeCDC/CommittedDeliveryFrontier") { ASSERT_EQ(committedPeekThrough(149, 120), 120); ASSERT_EQ(committedPeekThrough(149, 200), 149); ASSERT_EQ(committedPeekThrough(103, 102), 102); + ASSERT_EQ(boundedCDCConsumeReplyThrough(100, 150, 200), 150); + ASSERT_EQ(boundedCDCConsumeReplyThrough(175, 150, 200), 175); + ASSERT_EQ(boundedCDCConsumeReplyThrough(invalidVersion, 150, 140), 140); + return Void(); +} + +TEST_CASE("/NativeCDC/ProxyHistoryReconciliation") { + const Tag oldTag(tagLocalityCDC, 1); + const Tag targetTag(tagLocalityCDC, 2); + auto stream = makeReference(1); + CDCStreamReadState initial; + initial.ranges = std::vector{ KeyRangeRef("a"_sr, "z"_sr) }; + initial.minVersion = 100; + initial.readVersion = 150; + initial.tagAssignments = { { 90, oldTag } }; + ASSERT(reconcileBufferedStreamMetadata(stream, initial).historyChanged); + stream->tagIntervals.front().bufferedThrough = 240; + stream->bufferedThrough = 240; + const auto addBufferedVersion = [&](Version version) { + auto& buffered = stream->mutations.emplace_back(); + buffered.version = version; + buffered.mutations.push_back_deep(buffered.arena(), MutationRef(MutationRef::SetValue, "key"_sr, "value"_sr)); + stream->bufferedBytes += estimatedCDCConsumeVersionBytes(buffered); + }; + addBufferedVersion(149); + addBufferedVersion(200); + addBufferedVersion(220); + const int64_t originalBytes = stream->bufferedBytes; + const int64_t preservedBytes = estimatedCDCConsumeVersionBytes(stream->mutations.front()); + + CDCStreamReadState migrating = initial; + migrating.readVersion = 250; + migrating.tagAssignments.emplace_back(200, targetTag); + const CDCStreamMetadataUpdate migrated = reconcileBufferedStreamMetadata(stream, migrating); + ASSERT(migrated.historyChanged); + ASSERT_EQ(migrated.releasedBytes, originalBytes - preservedBytes); + ASSERT_EQ(stream->bufferedBytes, preservedBytes); + ASSERT_EQ(stream->mutations.size(), 1); + ASSERT_EQ(stream->mutations.front().version, 149); + ASSERT_EQ(stream->tagIntervals.size(), 2); + ASSERT_EQ(stream->tagIntervals[0].end, 200); + ASSERT_EQ(stream->tagIntervals[0].bufferedThrough, 199); + ASSERT_EQ(stream->tagIntervals[1].begin, 200); + ASSERT_EQ(stream->tagIntervals[1].bufferedThrough, 199); + ASSERT_EQ(stream->bufferedThrough, 199); + + stream->tagIntervals[1].bufferedThrough = 250; + stream->bufferedThrough = 250; + addBufferedVersion(230); + addBufferedVersion(250); + CDCStreamReadState finalized = migrating; + finalized.minVersion = 220; + finalized.readVersion = 350; + finalized.tagAssignments = { { 200, targetTag } }; + const CDCStreamMetadataUpdate completed = reconcileBufferedStreamMetadata(stream, finalized); + ASSERT(completed.historyChanged); + ASSERT_EQ(completed.releasedBytes, preservedBytes); + ASSERT_EQ(stream->tagIntervals.size(), 1); + ASSERT_EQ(stream->tagIntervals.front().tag, targetTag); + ASSERT_EQ(stream->tagIntervals.front().begin, 220); + ASSERT_EQ(stream->bufferedThrough, 250); + ASSERT_EQ(stream->mutations.size(), 2); + ASSERT_EQ(stream->mutations.front().version, 230); + + const CDCStreamMetadataUpdate stale = reconcileBufferedStreamMetadata(stream, migrating); + ASSERT(!stale.historyChanged); + ASSERT(!stale.readVersionAdvanced); + ASSERT_EQ(stale.releasedBytes, 0); + ASSERT_EQ(stream->metadataReadVersion, 350); + ASSERT_EQ(stream->minVersion, 220); + ASSERT_EQ(stream->tagIntervals.size(), 1); + ASSERT_EQ(stream->bufferedThrough, 250); + + finalized.minVersion = 251; + finalized.readVersion = 400; + const CDCStreamMetadataUpdate acknowledged = reconcileBufferedStreamMetadata(stream, finalized); + ASSERT(!acknowledged.historyChanged); + ASSERT_GT(acknowledged.releasedBytes, 0); + ASSERT(stream->mutations.empty()); + ASSERT_EQ(stream->bufferedBytes, 0); + ASSERT_EQ(stream->bufferedThrough, 250); + return Void(); +} + +TEST_CASE("/NativeCDC/ProxyHistoryReconciliationAfterAcknowledgement") { + const Tag oldTag(tagLocalityCDC, 1); + const Tag targetTag(tagLocalityCDC, 2); + auto stream = makeReference(1); + CDCStreamReadState initial; + initial.minVersion = 100; + initial.readVersion = 150; + initial.tagAssignments = { { 90, oldTag } }; + reconcileBufferedStreamMetadata(stream, initial); + stream->tagIntervals.front().bufferedThrough = 180; + stream->bufferedThrough = 180; + + // The durable acknowledgement can be observed by the pop scanner before either migration history snapshot. + advanceStreamMinVersion(stream, 225); + CDCStreamReadState finalized = initial; + finalized.minVersion = 225; + finalized.readVersion = 250; + finalized.tagAssignments = { { 200, targetTag } }; + ASSERT(reconcileBufferedStreamMetadata(stream, finalized).historyChanged); + ASSERT_EQ(stream->tagIntervals.size(), 1); + ASSERT_EQ(stream->tagIntervals.front().tag, targetTag); + ASSERT_EQ(stream->tagIntervals.front().begin, 225); + ASSERT_EQ(stream->bufferedThrough, 224); + + CDCStreamReadState olderRead = finalized; + olderRead.minVersion = 200; + olderRead.readVersion = 210; + olderRead.tagAssignments = { { 90, oldTag }, { 200, targetTag } }; + ASSERT(!reconcileBufferedStreamMetadata(stream, olderRead).historyChanged); + ASSERT_EQ(stream->bufferedThrough, 224); + ASSERT_EQ(boundedCDCConsumeReplyThrough(224, olderRead.readVersion, stream->bufferedThrough), 224); + return Void(); +} + +TEST_CASE("/NativeCDC/ProxyRetagReadAndAcknowledgementEligibility") { + const Tag firstTag(tagLocalityCDC, 1); + const Tag secondTag(tagLocalityCDC, 2); + auto stream = makeReference(1); + CDCStreamReadState metadata; + metadata.minVersion = 100; + metadata.readVersion = 400; + metadata.tagAssignments = { { 90, firstTag }, { 200, secondTag }, { 300, firstTag } }; + reconcileBufferedStreamMetadata(stream, metadata); + stream->readDemand = 1; + + ASSERT_EQ(eligibleTagReadInterval(*stream, firstTag).get(), 0); + ASSERT(!eligibleTagReadInterval(*stream, secondTag).present()); + ASSERT(canBufferTagVersion(*stream, firstTag, 150)); + ASSERT(!canBufferTagVersion(*stream, secondTag, 200)); + ASSERT(!canBufferTagVersion(*stream, firstTag, 350)); + ASSERT(!advanceStreamTagBufferedThrough(stream, firstTag, 150)); + stream->metadataReadVersion = 150; + ASSERT(!eligibleTagReadInterval(*stream, firstTag).present()); + ASSERT(!eligibleTagReadInterval(*stream, secondTag).present()); + + stream->metadataReadVersion = metadata.readVersion; + ASSERT(advanceStreamTagBufferedThrough(stream, firstTag, 400)); + ASSERT_EQ(stream->bufferedThrough, 199); + ASSERT_EQ(stream->tagIntervals[2].bufferedThrough, 299); + ASSERT_EQ(eligibleTagReadInterval(*stream, secondTag).get(), 1); + ASSERT(canBufferTagVersion(*stream, secondTag, 200)); + ASSERT(!canBufferTagVersion(*stream, firstTag, 350)); + ASSERT(advanceStreamTagBufferedThrough(stream, secondTag, 400)); + ASSERT_EQ(stream->bufferedThrough, 299); + ASSERT(canBufferTagVersion(*stream, firstTag, 350)); + ASSERT(!advanceStreamTagBufferedThrough(stream, firstTag, 450)); + ASSERT_EQ(stream->bufferedThrough, metadata.readVersion); + ASSERT_EQ(stream->minVersion, 100); + + // Acknowledgements can unlock an interval before its prefix is read. + stream = makeReference(1); + reconcileBufferedStreamMetadata(stream, metadata); + stream->readDemand = 1; + + const Optional initialReadInterval = firstIncompleteTagInterval(*stream); + advanceStreamMinVersion(stream, 225); + ASSERT(initialReadInterval != firstIncompleteTagInterval(*stream)); + ASSERT_EQ(eligibleTagReadInterval(*stream, secondTag).get(), 1); + ASSERT(canBufferTagVersion(*stream, secondTag, 225)); + ASSERT(!canBufferTagVersion(*stream, secondTag, 224)); + + const Optional acknowledgedReadInterval = firstIncompleteTagInterval(*stream); + metadata.minVersion = 325; + metadata.readVersion = 350; + const CDCStreamMetadataUpdate update = reconcileBufferedStreamMetadata(stream, metadata); + ASSERT(!update.historyChanged); + ASSERT(!update.readVersionAdvanced); + ASSERT(acknowledgedReadInterval != firstIncompleteTagInterval(*stream)); + ASSERT_EQ(eligibleTagReadInterval(*stream, firstTag).get(), 2); + ASSERT(canBufferTagVersion(*stream, firstTag, 325)); return Void(); } diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index a06772b8a5d..f8636a96aaa 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -29,7 +29,6 @@ #include "fdbclient/ClientBooleanParams.h" #include "fdbclient/FDBTypes.h" -#include "NativeCdcInternal.h" #include "fdbserver/core/NativeCdcMetadata.h" #include "fdbclient/SystemData.h" #include "fdbclient/DatabaseContext.h" diff --git a/fdbserver/clustercontroller/ClusterRecovery.cpp b/fdbserver/clustercontroller/ClusterRecovery.cpp index d1de04e86dd..d4afcd67e1f 100644 --- a/fdbserver/clustercontroller/ClusterRecovery.cpp +++ b/fdbserver/clustercontroller/ClusterRecovery.cpp @@ -1453,7 +1453,7 @@ Future readTransactionSystemState(Reference self, RangeResult rawCdcHistoryTags = co_await self->txnStateStore->readRange(cdcTagHistoryKeys); for (auto& kv : rawCdcHistoryTags) { - const CDCTagHistoryEntry tagHistory = decodeCDCTagHistoryKey(kv.key); + const CDCTagHistoryEntry tagHistory = decodeCDCTagHistoryEntry(kv.key, kv.value); if (activeCdcStreams.contains(tagHistory.streamId)) { self->allTags.push_back(tagHistory.tag); } diff --git a/fdbserver/core/NativeCdcMetadata.cpp b/fdbserver/core/NativeCdcMetadata.cpp index 33f2b9ce826..58fdb60b2b5 100644 --- a/fdbserver/core/NativeCdcMetadata.cpp +++ b/fdbserver/core/NativeCdcMetadata.cpp @@ -49,7 +49,6 @@ class NativeCdcIdentifierAllocator { bool sawStream = false; CDCStreamId maxStreamId = 0; std::unordered_map tagStreamCounts; - std::unordered_map tagWriteRates; public: void observeStreamId(CDCStreamId streamId) { @@ -62,14 +61,6 @@ class NativeCdcIdentifierAllocator { ++tagStreamCounts[tag.id]; } - void observeTagLoad(Tag tag, CDCTagLoadSample const& sample, ValueRef generation, Version readVersion) { - if (tag.locality == tagLocalityCDC && sample.assignmentChange == generation && sample.sampleVersion >= 0 && - sample.sampleVersion <= readVersion && sample.validThrough >= readVersion && - sample.bytesWrittenPerKSecond >= 0) { - tagWriteRates[tag.id] = sample.bytesWrittenPerKSecond; - } - } - bool hasStreams(Tag tag) const { return tagStreamCounts.contains(tag.id); } std::pair allocate(int tagCount) const { @@ -81,24 +72,16 @@ class NativeCdcIdentifierAllocator { if (!validNativeCdcTagCount(tagCount)) { throw invalid_option_value(); } - const bool completeLoad = !tagStreamCounts.empty() && - std::all_of(tagStreamCounts.begin(), tagStreamCounts.end(), [&](auto const& entry) { - return entry.first >= tagCount || tagWriteRates.contains(entry.first); - }); uint32_t leastStreams = std::numeric_limits::max(); - int64_t leastWriteRate = std::numeric_limits::max(); CDCTagId selectedTagId = 0; for (uint32_t tagId = 0; tagId < static_cast(tagCount); ++tagId) { auto count = tagStreamCounts.find(static_cast(tagId)); const uint32_t streamCount = count == tagStreamCounts.end() ? 0 : count->second; - const int64_t writeRate = completeLoad && streamCount > 0 ? tagWriteRates.at(tagId) : 0; - if (writeRate < leastWriteRate || (writeRate == leastWriteRate && streamCount < leastStreams)) { - leastWriteRate = writeRate; + if (streamCount < leastStreams) { leastStreams = streamCount; selectedTagId = static_cast(tagId); } } - CODE_PROBE(completeLoad, "Native CDC registration places streams using fresh producer throughput"); return { streamId, Tag(tagLocalityCDC, selectedTagId) }; } }; @@ -229,36 +212,38 @@ Future observeNativeCdcMetadata(Transaction* tr, NativeCdcIdentifierAlloca for (const auto& tagAssignment : currentTags) { allocator->observeTag(tagAssignment.second); } - if (!currentTags.empty()) { - const Value generation = (co_await tr->get(cdcProxyAssignmentChangeKey)).orDefault(Value()); - const Version readVersion = co_await tr->getReadVersion(); - Key begin = cdcTagLoadKeys.begin; - while (begin < cdcTagLoadKeys.end) { - RangeResult samples = co_await tr->getRange(KeyRangeRef(begin, cdcTagLoadKeys.end), CLIENT_KNOBS->TOO_MANY); - for (const auto& sample : samples) { - allocator->observeTagLoad( - decodeCDCTagLoadKey(sample.key), decodeCDCTagLoadValue(sample.value), generation, readVersion); - } - if (!samples.more) { - break; - } - begin = keyAfter(samples.back().key); - } - } } Future> readNativeCdcTagStateImpl(Transaction* tr, CDCStreamId streamId) { Future> keysFuture = tr->get(cdcStreamKeyFor(streamId)); + Future> minimumFuture = tr->get(cdcMinVersionKeyFor(streamId)); + Future> ownerFuture = getNativeCdcProxyAssignment(tr, streamId); Future historyFuture = - tr->getRange(cdcTagHistoryRangeFor(streamId), 2, Snapshot::False, Reverse::True); + tr->getRange(cdcTagHistoryRangeFor(streamId), 3, Snapshot::False, Reverse::True); const Optional keys = co_await keysFuture; + const Optional minimum = co_await minimumFuture; + const Optional owner = co_await ownerFuture; const RangeResult history = co_await historyFuture; - if (!keys.present() || history.size() != 1 || history.more || !history.front().value.empty()) { + if (!keys.present() || !minimum.present() || !owner.present() || history.empty() || history.more || + history.size() > 2) { co_return Optional(); } - co_return NativeCdcTagState{ streamId, - decodeCDCStreamKeysValue(keys.get()), - decodeCDCTagHistoryKey(history.front().key) }; + NativeCdcTagState state; + state.streamId = streamId; + state.ranges = decodeCDCStreamKeysValue(keys.get()); + state.historyKey = history.front().key; + state.assignment = decodeCDCTagHistoryEntry(history.front().key, history.front().value); + state.proxyId = owner.get(); + state.minVersion = decodeCDCMinVersionValue(minimum.get()); + state.pending = history.size() > 1 || !history.front().value.empty(); + co_return state; +} + +bool sameNativeCdcTagState(NativeCdcTagState const& current, NativeCdcTagState const& expected) { + return current.streamId == expected.streamId && current.ranges == expected.ranges && + current.historyKey == expected.historyKey && current.proxyId == expected.proxyId && + current.assignment.version == expected.assignment.version && + current.assignment.tag == expected.assignment.tag; } } // namespace @@ -292,6 +277,60 @@ Future>> readNativeCdcTagStates(Transact co_return Optional>(std::move(result)); } +Future retagNativeCdcStream(Transaction* tr, NativeCdcTagState expected, Tag destination) { + Optional current = co_await readNativeCdcTagState(tr, expected.streamId); + if (!current.present() || !sameNativeCdcTagState(current.get(), expected) || current.get().pending || + destination.locality != tagLocalityCDC || destination == current.get().assignment.tag) { + co_return false; + } + const Optional destinationOwner = co_await getNativeCdcProxyAssignmentForTag(tr, destination); + if (destinationOwner.present() && destinationOwner.get() != current.get().proxyId) { + co_return false; + } + const Version readVersion = co_await tr->getReadVersion(); + const auto& clientInfo = tr->getDatabase()->clientInfo->get(); + if (!clientInfo.nativeCdcEnabled || !validNativeCdcTagCount(clientInfo.nativeCdcTagCount) || + destination.id >= clientInfo.nativeCdcTagCount) { + co_return false; + } + const Key historyKey = cdcTagHistoryKeyFor(expected.streamId, readVersion, destination); + if (historyKey <= current.get().historyKey) { + co_return false; + } + // The key orders assignments, while the value supplies the exact routing + // cutover. A read-version boundary could skip writes before this commit. + tr->atomicOp(historyKey, cdcVersionstampedMinVersionValue(), MutationRef::SetVersionstampedValue); + signalNativeCdcProxyAssignmentChange(tr); + co_return true; +} + +Future finishNativeCdcRetag(Transaction* tr, NativeCdcTagState expected) { + const Optional current = co_await readNativeCdcTagState(tr, expected.streamId); + if (!current.present() || !sameNativeCdcTagState(current.get(), expected) || !current.get().pending || + current.get().minVersion < current.get().assignment.version) { + co_return false; + } + const RangeResult history = co_await tr->getRange(cdcTagHistoryRangeFor(expected.streamId), 3); + if (history.more || history.empty() || history.size() > 2 || history.back().key != current.get().historyKey) { + co_return false; + } + std::set retiredTags; + for (const auto& row : history) { + const Tag tag = decodeCDCTagHistoryEntry(row.key, row.value).tag; + if (tag != current.get().assignment.tag) { + retiredTags.insert(tag); + } + } + tr->clear(cdcTagHistoryRangeFor(expected.streamId)); + tr->set(cdcTagHistoryKeyFor(expected.streamId, current.get().assignment.version, current.get().assignment.tag), + Value()); + for (const Tag tag : retiredTags) { + retireNativeCdcTag(tr, tag); + } + signalNativeCdcProxyAssignmentChange(tr); + co_return true; +} + Future prepareNativeCdcStreamRegistration(Transaction* tr, Key name, std::vector ranges, @@ -540,39 +579,3 @@ TEST_CASE("/NativeCDC/LifecycleAllocation") { return Void(); } - -TEST_CASE("/NativeCDC/ThroughputPlacement") { - const Value generation = "current"_sr; - const Tag hotTag(tagLocalityCDC, 0); - const Tag coldTag(tagLocalityCDC, 1); - const CDCTagLoadSample hot{ generation, 100, 200, 100000 }; - const CDCTagLoadSample cold{ generation, 100, 200, 0 }; - auto select = [&](CDCTagLoadSample const& hotSample, Optional coldSample, int tagCount = 2) { - NativeCdcIdentifierAllocator allocator; - allocator.observeTag(hotTag); - allocator.observeTag(coldTag); - allocator.observeTag(coldTag); - allocator.observeTagLoad(hotTag, hotSample, generation, 150); - if (coldSample.present()) { - allocator.observeTagLoad(coldTag, coldSample.get(), generation, 150); - } - return allocator.allocate(tagCount).second; - }; - ASSERT_EQ(select(hot, cold), coldTag); - ASSERT_EQ(select(hot, Optional()), hotTag); - ASSERT_EQ(select(hot, cold, 3), Tag(tagLocalityCDC, 2)); - - CDCTagLoadSample invalid = cold; - invalid.assignmentChange = "previous"_sr; - ASSERT_EQ(select(hot, invalid), hotTag); - invalid = cold; - invalid.validThrough = 149; - ASSERT_EQ(select(hot, invalid), hotTag); - invalid = cold; - invalid.sampleVersion = 151; - ASSERT_EQ(select(hot, invalid), hotTag); - invalid = cold; - invalid.bytesWrittenPerKSecond = -1; - ASSERT_EQ(select(hot, invalid), hotTag); - return Void(); -} diff --git a/fdbserver/core/ServerKnobs.cpp b/fdbserver/core/ServerKnobs.cpp index 82a3a10791c..9d246302900 100644 --- a/fdbserver/core/ServerKnobs.cpp +++ b/fdbserver/core/ServerKnobs.cpp @@ -189,14 +189,7 @@ void ServerKnobs::initialize(Randomize randomize, ClientKnobs* clientKnobs, IsSi init( CDC_PROXY_FAILURE_COALESCE_DELAY, 0.0 ); init( CDC_PROXY_POP_MIN_INTERVAL, 0.1 ); if( randomize && buggify() ) CDC_PROXY_POP_MIN_INTERVAL = 0.01; init( CDC_PROXY_POP_SCAN_INTERVAL, 5.0 ); if( randomize && buggify() ) CDC_PROXY_POP_SCAN_INTERVAL = 0.1; - init( NATIVE_CDC_TAG_BALANCING_ENABLED, true ); - init( NATIVE_CDC_TAG_SAMPLE_INTERVAL, 30.0 ); - init( NATIVE_CDC_TAG_SAMPLE_TIMEOUT, 5.0 ); - init( NATIVE_CDC_TAG_SAMPLE_MAX_AGE, 90.0 ); - init( NATIVE_CDC_TAG_MAX_STREAMS, 1000 ); - init( NATIVE_CDC_TAG_MODEL_MAX_ENTRIES, 2000000 ); - init( NATIVE_CDC_TAG_SAMPLE_CONCURRENCY, 8 ); - init( NATIVE_CDC_TAG_SAMPLE_SHARD_LIMIT, 1000 ); + init( NATIVE_CDC_RETAG_CLEANUP_INTERVAL, 30.0 ); init( APPLY_MUTATION_BYTES, 1e6 ); init( BUGGIFY_RECOVER_MEMORY_LIMIT, 1e6 ); init( BUGGIFY_WORKER_REMOVED_MAX_LAG, 30 ); diff --git a/fdbserver/core/include/fdbserver/core/Knobs.h b/fdbserver/core/include/fdbserver/core/Knobs.h index f2d818aff7a..a8a6903e469 100644 --- a/fdbserver/core/include/fdbserver/core/Knobs.h +++ b/fdbserver/core/include/fdbserver/core/Knobs.h @@ -80,14 +80,7 @@ class SWIFT_CXX_IMMORTAL_SINGLETON_TYPE ServerKnobs : public KnobsImpl ranges; + Key historyKey; CDCTagHistoryEntry assignment; + UID proxyId; + Version minVersion = invalidVersion; + bool pending = false; }; // An absent result means the bounded snapshot is incomplete; it must not be used // as an empty or zero-load configuration. Future> readNativeCdcTagState(Transaction* tr, CDCStreamId streamId); Future>> readNativeCdcTagStates(Transaction* tr, int maxStreams); +// These helpers revalidate the durable identity and prepare mutations without +// committing. The caller must fence its controller ownership in this transaction. +Future retagNativeCdcStream(Transaction* tr, NativeCdcTagState expected, Tag destination); +Future finishNativeCdcRetag(Transaction* tr, NativeCdcTagState expected); + struct NativeCdcRegistrationResult { CDCStreamId streamId; // Describes only mutations prepared by this helper, not unrelated caller mutations. diff --git a/fdbserver/datadistributor/DataDistribution.cpp b/fdbserver/datadistributor/DataDistribution.cpp index dd878ff009e..1679fdc7eba 100644 --- a/fdbserver/datadistributor/DataDistribution.cpp +++ b/fdbserver/datadistributor/DataDistribution.cpp @@ -43,7 +43,7 @@ #include "DDTeamCollection.h" #include "DataDistribution.h" #include "DDRelocationQueue.h" -#include "NativeCdcBalancer.h" +#include "NativeCdcRetagCleanup.h" #include "fdbserver/core/Knobs.h" #include "fdbserver/core/MoveKeys.h" #include "fdbserver/core/QuietDatabase.h" @@ -3051,7 +3051,7 @@ Future bulkDumpCore(Reference self, Future readyToS } void addDataDistributionActors(Reference self, std::vector>& actors) { - actors.push_back(nativeCdcBalancer( + actors.push_back(nativeCdcRetagCleanup( self->txnProcessor->context(), self->lock, self->context->ddEnabledState.get(), self->initialized.getFuture())); if (bulkLoadIsEnabled(self->initData->bulkLoadMode)) { TraceEvent(SevInfo, "DDBulkLoadModeEnabled", self->ddId) diff --git a/fdbserver/datadistributor/NativeCdcBalancer.cpp b/fdbserver/datadistributor/NativeCdcBalancer.cpp deleted file mode 100644 index 0ec7fae2da9..00000000000 --- a/fdbserver/datadistributor/NativeCdcBalancer.cpp +++ /dev/null @@ -1,397 +0,0 @@ -/* - * NativeCdcBalancer.cpp - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -#include -#include -#include -#include -#include -#include -#include - -#include "NativeCdcBalancer.h" -#include "fdbserver/core/NativeCdcMetadata.h" -#include "fdbclient/DatabaseContext.h" -#include "fdbclient/KeyRangeMap.h" -#include "fdbclient/StorageServerInterface.h" -#include "fdbclient/SystemData.h" -#include "fdbserver/core/Knobs.h" -#include "flow/CodeProbe.h" -#include "flow/Trace.h" -#include "flow/UnitTest.h" - -namespace { - -bool addNativeCdcLoad(int64_t* total, int64_t value) { - if (value < 0 || value > std::numeric_limits::max() - *total) { - return false; - } - *total += value; - return true; -} - -class NativeCdcLoadModel { - struct Segment { - KeyRange keys; - std::set tags; - Optional load; - }; - - std::vector segments; - std::map tagLoads; - bool complete = false; - - explicit NativeCdcLoadModel(std::vector const& states) { - KeyRangeMap> coveringTags; - for (const auto& state : states) { - for (const auto& keys : state.ranges) { - for (auto range : coveringTags.modify(keys)) { - range->value().insert(state.assignment.tag); - } - } - } - for (auto range : coveringTags.ranges()) { - if (!range.value().empty()) { - segments.push_back(Segment{ KeyRange(range.range()), range.value(), {} }); - } - } - } - -public: - static Optional create(std::vector const& states, int64_t maxEntries) { - if (maxEntries <= 0) { - return {}; - } - if (!states.empty()) { - // M ranges have at most 2M-1 nonempty segments, each containing at most N tags. - // Bound coverage memberships before constructing the segment sets. - const uint64_t maxRanges = static_cast(maxEntries) / states.size() / 2; - uint64_t ranges = 0; - for (const auto& state : states) { - if (state.ranges.empty() || state.ranges.size() > maxRanges - ranges) { - return {}; - } - ranges += state.ranges.size(); - } - } - return NativeCdcLoadModel(states); - } - - size_t segmentCount() const { return segments.size(); } - KeyRange const& segmentKeys(size_t index) const { return segments[index].keys; } - std::map const& loads() const { - ASSERT(complete); - return tagLoads; - } - - bool setSample(size_t index, int64_t load) { - ASSERT(!complete); - if (load < 0) { - return false; - } - segments[index].load = load; - return true; - } - - bool finishSamples() { - tagLoads.clear(); - for (const auto& segment : segments) { - if (!segment.load.present()) { - return false; - } - for (Tag tag : segment.tags) { - if (!addNativeCdcLoad(&tagLoads[tag], segment.load.get())) { - return false; - } - } - } - complete = true; - return true; - } -}; - -Version nativeCdcDurationVersions(double seconds) { - const long double versions = static_cast(seconds) * SERVER_KNOBS->VERSIONS_PER_SECOND; - if (versions >= std::numeric_limits::max()) { - return std::numeric_limits::max(); - } - return static_cast(versions); -} - -bool validNativeCdcBalancerKnobs() { - return SERVER_KNOBS->NATIVE_CDC_TAG_MAX_STREAMS > 0 && - SERVER_KNOBS->NATIVE_CDC_TAG_MAX_STREAMS < std::numeric_limits::max() && - SERVER_KNOBS->NATIVE_CDC_TAG_MODEL_MAX_ENTRIES > 0 && SERVER_KNOBS->NATIVE_CDC_TAG_SAMPLE_CONCURRENCY > 0 && - SERVER_KNOBS->NATIVE_CDC_TAG_SAMPLE_SHARD_LIMIT > 1 && - std::isfinite(SERVER_KNOBS->NATIVE_CDC_TAG_SAMPLE_TIMEOUT) && - SERVER_KNOBS->NATIVE_CDC_TAG_SAMPLE_TIMEOUT > 0 && - std::isfinite(SERVER_KNOBS->NATIVE_CDC_TAG_SAMPLE_MAX_AGE) && - SERVER_KNOBS->NATIVE_CDC_TAG_SAMPLE_MAX_AGE > 0 && SERVER_KNOBS->VERSIONS_PER_SECOND > 0; -} - -Future sampleNativeCdcRanges(Database cx, NativeCdcLoadModel* model, size_t* nextSegment) { - while (*nextSegment < model->segmentCount()) { - const size_t index = (*nextSegment)++; - const StorageMetrics metrics = - co_await cx->getStorageMetrics(model->segmentKeys(index), SERVER_KNOBS->NATIVE_CDC_TAG_SAMPLE_SHARD_LIMIT); - if (!model->setSample(index, metrics.bytesWrittenPerKSecond)) { - throw operation_failed(); - } - co_await yield(TaskPriority::DataDistribution); - } -} - -Future sampleNativeCdcLoads(Database cx, NativeCdcLoadModel* model) { - size_t nextSegment = 0; - std::vector> workers; - const size_t workerCount = - std::min(model->segmentCount(), static_cast(SERVER_KNOBS->NATIVE_CDC_TAG_SAMPLE_CONCURRENCY)); - workers.reserve(workerCount); - for (size_t i = 0; i < workerCount; ++i) { - workers.push_back(sampleNativeCdcRanges(cx, model, &nextSegment)); - } - co_await waitForAll(workers); -} - -struct NativeCdcMetadataSnapshot { - Value assignmentChange; - Version version; - std::vector streams; -}; - -class NativeCdcBalancer { - Database cx; - MoveKeysLock lock; - const DDEnabledState* ddEnabledState; - - bool samplingEnabled() const { - return SERVER_KNOBS->NATIVE_CDC_TAG_BALANCING_ENABLED && cx->clientInfo->get().nativeCdcEnabled; - } - - Future> readSnapshot() { - Transaction tr(cx); - while (true) { - Error err; - try { - tr.setOption(FDBTransactionOptions::READ_LOCK_AWARE); - tr.setOption(FDBTransactionOptions::READ_SYSTEM_KEYS); - const Optional change = co_await tr.get(cdcProxyAssignmentChangeKey); - const Value generation = change.present() ? change.get() : Value(); - Optional> states = - co_await readNativeCdcTagStates(&tr, SERVER_KNOBS->NATIVE_CDC_TAG_MAX_STREAMS); - if (!states.present()) { - TraceEvent("NativeCdcTagMetadataUnavailable", lock.myOwner) - .detail("StreamLimit", SERVER_KNOBS->NATIVE_CDC_TAG_MAX_STREAMS); - co_return Optional(); - } - const Version version = co_await tr.getReadVersion(); - co_return Optional( - NativeCdcMetadataSnapshot{ generation, version, std::move(states.get()) }); - } catch (Error& e) { - err = e; - } - co_await tr.onError(err); - } - } - - Future currentGeneration(Transaction* tr, Value expected, Version sampledAt, Version validThrough) const { - if (!samplingEnabled()) { - co_return false; - } - const Optional generation = co_await tr->get(cdcProxyAssignmentChangeKey); - if ((generation.present() ? generation.get() : Value()) != expected) { - CODE_PROBE(true, "Native CDC DD discards samples after assignment changes"); - co_return false; - } - const Version version = co_await tr->getReadVersion(); - const bool fresh = version >= sampledAt && version <= validThrough; - co_return fresh; - } - - Future publishLoads(NativeCdcMetadataSnapshot const& snapshot, - Version validThrough, - NativeCdcLoadModel const& model) { - Transaction tr(cx); - while (true) { - Error err; - try { - tr.setOption(FDBTransactionOptions::LOCK_AWARE); - tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); - if (!(co_await currentGeneration(&tr, snapshot.assignmentChange, snapshot.version, validThrough))) { - co_return false; - } - tr.clear(cdcTagLoadKeys); - for (const auto& [tag, load] : model.loads()) { - tr.set(cdcTagLoadKeyFor(tag), - cdcTagLoadValue( - CDCTagLoadSample{ snapshot.assignmentChange, snapshot.version, validThrough, load })); - } - co_await checkMoveKeysLock(&tr, lock, ddEnabledState); - co_await tr.commit(); - CODE_PROBE(true, "Native CDC DD publishes producer tag throughput samples"); - TraceEvent("NativeCdcTagLoadSampled", lock.myOwner) - .detail("Streams", snapshot.streams.size()) - .detail("Tags", model.loads().size()) - .detail("Segments", model.segmentCount()) - .detail("SampleVersion", snapshot.version) - .detail("ValidThrough", validThrough); - co_return true; - } catch (Error& e) { - err = e; - } - co_await tr.onError(err); - } - } - - Future runPass() { - if (!samplingEnabled()) { - co_return; - } - if (!validNativeCdcBalancerKnobs()) { - TraceEvent(SevWarn, "NativeCdcTagBalancerInvalidKnobs", lock.myOwner); - co_return; - } - Optional snapshot = co_await readSnapshot(); - if (!snapshot.present() || snapshot.get().streams.empty()) { - co_return; - } - const int tagCount = cx->clientInfo->get().nativeCdcTagCount; - if (tagCount <= 0 || tagCount > static_cast(std::numeric_limits::max()) + 1) { - co_return; - } - const Version lifetime = nativeCdcDurationVersions(SERVER_KNOBS->NATIVE_CDC_TAG_SAMPLE_MAX_AGE); - const Version validThrough = snapshot.get().version > std::numeric_limits::max() - lifetime - ? std::numeric_limits::max() - : snapshot.get().version + lifetime; - Optional boundedModel = - NativeCdcLoadModel::create(snapshot.get().streams, SERVER_KNOBS->NATIVE_CDC_TAG_MODEL_MAX_ENTRIES); - if (!boundedModel.present()) { - CODE_PROBE(true, "Native CDC DD skips an overlap model exceeding its entry budget"); - TraceEvent("NativeCdcTagModelBudgetExceeded", lock.myOwner) - .detail("Streams", snapshot.get().streams.size()) - .detail("MaxEntries", SERVER_KNOBS->NATIVE_CDC_TAG_MODEL_MAX_ENTRIES); - co_return; - } - NativeCdcLoadModel& model = boundedModel.get(); - // One deadline bounds all segment requests. Partial/failed samples are never published as zero load. - const Optional sampled = - co_await timeout(sampleNativeCdcLoads(cx, &model), SERVER_KNOBS->NATIVE_CDC_TAG_SAMPLE_TIMEOUT); - if (!sampled.present() || !model.finishSamples()) { - CODE_PROBE(true, "Native CDC DD skips incomplete throughput samples"); - TraceEvent("NativeCdcTagSamplingIncomplete", lock.myOwner).detail("Segments", model.segmentCount()); - co_return; - } - co_await publishLoads(snapshot.get(), validThrough, model); - } - -public: - NativeCdcBalancer(Database cx, MoveKeysLock lock, const DDEnabledState* ddEnabledState) - : cx(cx), lock(lock), ddEnabledState(ddEnabledState) {} - - Future run(Future initialized) { - co_await initialized; - while (true) { - try { - co_await runPass(); - } catch (Error& e) { - if (e.code() == error_code_actor_cancelled || e.code() == error_code_broken_promise || - e.code() == error_code_movekeys_conflict) { - throw; - } - TraceEvent(SevWarn, "NativeCdcTagBalancerError", lock.myOwner).error(e); - } - const double interval = SERVER_KNOBS->NATIVE_CDC_TAG_SAMPLE_INTERVAL; - co_await delay(std::isfinite(interval) && interval > 0 ? interval : 30.0, TaskPriority::DataDistribution); - } - } -}; - -NativeCdcTagState nativeCdcPolicyTestStream(CDCStreamId streamId, KeyRange keys, uint16_t tag) { - return NativeCdcTagState{ streamId, { keys }, CDCTagHistoryEntry(streamId, 100, Tag(tagLocalityCDC, tag)) }; -} - -TEST_CASE("/NativeCDC/TagBalancing/IncompleteAndZeroSamples") { - auto model = NativeCdcLoadModel::create({ nativeCdcPolicyTestStream(1, KeyRangeRef("a"_sr, "b"_sr), 0) }, - SERVER_KNOBS->NATIVE_CDC_TAG_MODEL_MAX_ENTRIES) - .get(); - ASSERT(!model.finishSamples()); - ASSERT(!model.setSample(0, -1)); - ASSERT(model.setSample(0, 0)); - ASSERT(model.finishSamples()); - ASSERT_EQ(model.loads().at(Tag(tagLocalityCDC, 0)), 0); - int64_t total = std::numeric_limits::max() - 1; - ASSERT(!addNativeCdcLoad(&total, 2)); - ASSERT(addNativeCdcLoad(&total, 1)); - return Void(); -} - -TEST_CASE("/NativeCDC/TagBalancing/OverlappingTagLoads") { - auto model = NativeCdcLoadModel::create({ nativeCdcPolicyTestStream(1, KeyRangeRef("a"_sr, "d"_sr), 0), - nativeCdcPolicyTestStream(2, KeyRangeRef("c"_sr, "f"_sr), 0), - nativeCdcPolicyTestStream(3, KeyRangeRef("c"_sr, "d"_sr), 1) }, - 18) - .get(); - ASSERT_EQ(model.segmentCount(), 3); - ASSERT(model.setSample(0, 2000)); - ASSERT(model.setSample(1, 3000)); - ASSERT(model.setSample(2, 5000)); - ASSERT(model.finishSamples()); - ASSERT_EQ(model.loads().at(Tag(tagLocalityCDC, 0)), 10000); - ASSERT_EQ(model.loads().at(Tag(tagLocalityCDC, 1)), 3000); - return Void(); -} - -TEST_CASE("/NativeCDC/TagBalancing/ModelEntryBudget") { - std::vector streams{ nativeCdcPolicyTestStream(1, KeyRangeRef("a"_sr, "f"_sr), 0), - nativeCdcPolicyTestStream(2, KeyRangeRef("b"_sr, "e"_sr), 0), - nativeCdcPolicyTestStream(3, KeyRangeRef("c"_sr, "d"_sr), 1) }; - // The same stream count needs a larger entry budget once streams cover disjoint range unions. - ASSERT(NativeCdcLoadModel::create(streams, 18).present()); - for (auto& stream : streams) { - stream.ranges.emplace_back(KeyRangeRef("x"_sr, "y"_sr)); - } - ASSERT(!NativeCdcLoadModel::create(streams, 18).present()); - ASSERT(!NativeCdcLoadModel::create(streams, 35).present()); - auto model = NativeCdcLoadModel::create(streams, 36); - ASSERT(model.present()); - ASSERT_EQ(model.get().segmentCount(), 6); - for (size_t i = 0; i < model.get().segmentCount(); ++i) { - ASSERT(model.get().setSample(i, 1000)); - } - ASSERT(model.get().finishSamples()); - ASSERT_EQ(model.get().loads().at(Tag(tagLocalityCDC, 0)), 6000); - ASSERT_EQ(model.get().loads().at(Tag(tagLocalityCDC, 1)), 2000); - ASSERT(!NativeCdcLoadModel::create(streams, 0).present()); - ASSERT(!NativeCdcLoadModel::create(streams, -1).present()); - ASSERT(NativeCdcLoadModel::create(streams, std::numeric_limits::max()).present()); - return Void(); -} - -} // namespace - -Future nativeCdcBalancer(Database cx, - MoveKeysLock lock, - const DDEnabledState* ddEnabledState, - Future initialized) { - NativeCdcBalancer balancer(cx, lock, ddEnabledState); - co_await balancer.run(initialized); -} - -void forceLinkNativeCdcBalancerTests() {} diff --git a/fdbserver/datadistributor/NativeCdcRetagCleanup.cpp b/fdbserver/datadistributor/NativeCdcRetagCleanup.cpp new file mode 100644 index 00000000000..3446ef207d4 --- /dev/null +++ b/fdbserver/datadistributor/NativeCdcRetagCleanup.cpp @@ -0,0 +1,207 @@ +/* + * NativeCdcRetagCleanup.cpp + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include +#include + +#include "NativeCdcRetagCleanup.h" +#include "fdbserver/core/NativeCdcMetadata.h" +#include "fdbclient/DatabaseContext.h" +#include "fdbclient/SystemData.h" +#include "fdbserver/core/Knobs.h" +#include "flow/CodeProbe.h" +#include "flow/Trace.h" +#include "flow/UnitTest.h" + +namespace { + +class NativeCdcCleanupProgress { + Optional assignmentChange; + Key next = cdcStreamKeys.begin; + bool cycleComplete = false; + bool sawPending = false; + bool rescan = false; + + bool sameGeneration(ValueRef generation) const { + return assignmentChange.present() && assignmentChange.get() == generation; + } + +public: + bool needsScan(ValueRef generation) const { + return !sameGeneration(generation) || !cycleComplete || sawPending || rescan; + } + + Key begin() const { return cycleComplete ? Key(cdcStreamKeys.begin) : next; } + + void scanned(Value generation, Key nextBegin, bool lastPage, bool pagePending) { + const bool continuing = assignmentChange.present() && !cycleComplete; + sawPending = (continuing && sawPending) || pagePending; + rescan = continuing && (rescan || !sameGeneration(generation)); + assignmentChange = std::move(generation); + next = std::move(nextBegin); + cycleComplete = lastPage; + } + + // Churn requires another traversal, not a restart that can starve later pages. + void changed() { rescan = true; } +}; + +class NativeCdcRetagCleanup { + Database cx; + MoveKeysLock lock; + const DDEnabledState* ddEnabledState; + NativeCdcCleanupProgress cleanupProgress; + + Future finishPendingPage() { + constexpr int cleanupPageSize = 100; + Transaction tr(cx); + while (true) { + Error err; + try { + tr.setOption(FDBTransactionOptions::LOCK_AWARE); + tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); + const Optional change = co_await tr.get(cdcProxyAssignmentChangeKey); + const Value generation = change.present() ? change.get() : Value(); + if (!cleanupProgress.needsScan(generation)) { + co_return false; + } + const Key begin = cleanupProgress.begin(); + const RangeResult page = co_await tr.getRange(KeyRangeRef(begin, cdcStreamKeys.end), cleanupPageSize); + std::vector>> reads; + for (const auto& row : page) { + reads.push_back(readNativeCdcTagState(&tr, decodeCDCStreamKey(row.key))); + } + const std::vector> states = co_await getAll(reads); + int finished = 0; + bool pagePending = false; + for (const auto& state : states) { + if (!state.present()) { + // Incomplete ownership/metadata is not evidence that all transitions have drained. + pagePending = true; + continue; + } + pagePending = pagePending || state.get().pending; + if (state.get().pending && (co_await finishNativeCdcRetag(&tr, state.get()))) { + ++finished; + } + } + if (finished == 0) { + cleanupProgress.scanned(generation, + page.more ? keyAfter(page.back().key) : Key(cdcStreamKeys.end), + !page.more, + pagePending); + co_return false; + } + co_await checkMoveKeysLock(&tr, lock, ddEnabledState); + co_await tr.commit(); + cleanupProgress.scanned(generation, + page.more ? keyAfter(page.back().key) : Key(cdcStreamKeys.end), + !page.more, + pagePending); + cleanupProgress.changed(); + CODE_PROBE(true, "Native CDC DD finishes acknowledged tag transitions"); + TraceEvent("NativeCdcTagTransitionsFinished", lock.myOwner).detail("Streams", finished); + co_return true; + } catch (Error& e) { + err = e; + } + co_await tr.onError(err); + } + } + +public: + NativeCdcRetagCleanup(Database cx, MoveKeysLock lock, const DDEnabledState* ddEnabledState) + : cx(cx), lock(lock), ddEnabledState(ddEnabledState) {} + + Future run(Future initialized) { + co_await initialized; + while (true) { + try { + co_await finishPendingPage(); + } catch (Error& e) { + if (e.code() == error_code_actor_cancelled || e.code() == error_code_broken_promise || + e.code() == error_code_movekeys_conflict) { + throw; + } + TraceEvent(SevWarn, "NativeCdcRetagCleanupError", lock.myOwner).error(e); + } + const double interval = SERVER_KNOBS->NATIVE_CDC_RETAG_CLEANUP_INTERVAL; + co_await delay(std::isfinite(interval) && interval > 0 ? interval : 30.0, TaskPriority::DataDistribution); + } + } +}; + +TEST_CASE("/NativeCDC/RetagCleanup/Paging") { + NativeCdcCleanupProgress progress; + const Value firstGeneration = "first"_sr; + const Value secondGeneration = "second"_sr; + const Key nextPage = keyAfter(cdcStreamKeyFor(100)); + ASSERT(progress.needsScan(firstGeneration)); + ASSERT_EQ(progress.begin(), cdcStreamKeys.begin); + progress.scanned(firstGeneration, nextPage, false, false); + ASSERT(progress.needsScan(firstGeneration)); + ASSERT_EQ(progress.begin(), nextPage); + progress.scanned(firstGeneration, cdcStreamKeys.end, true, true); + // Acknowledgements do not change the assignment generation, so pending cycles must repeat. + ASSERT(progress.needsScan(firstGeneration)); + ASSERT_EQ(progress.begin(), cdcStreamKeys.begin); + progress.scanned(firstGeneration, cdcStreamKeys.end, true, false); + ASSERT(!progress.needsScan(firstGeneration)); + ASSERT(progress.needsScan(secondGeneration)); + ASSERT_EQ(progress.begin(), cdcStreamKeys.begin); + progress.changed(); + ASSERT(progress.needsScan(firstGeneration)); + return Void(); +} + +TEST_CASE("/NativeCDC/RetagCleanup/ProgressAcrossChurn") { + NativeCdcCleanupProgress progress; + const Value firstGeneration = "first"_sr; + const Value secondGeneration = "second"_sr; + const Value thirdGeneration = "third"_sr; + const Key secondPage = keyAfter(cdcStreamKeyFor(100)); + const Key thirdPage = keyAfter(cdcStreamKeyFor(200)); + progress.scanned(firstGeneration, secondPage, false, true); + progress.changed(); // A successful cleanup on the first page changes the generation. + ASSERT(progress.needsScan(secondGeneration)); + ASSERT_EQ(progress.begin(), secondPage); + progress.scanned(secondGeneration, thirdPage, false, true); + progress.changed(); + ASSERT(progress.needsScan(thirdGeneration)); + ASSERT_EQ(progress.begin(), thirdPage); + progress.scanned(thirdGeneration, cdcStreamKeys.end, true, false); + ASSERT(progress.needsScan(thirdGeneration)); + ASSERT_EQ(progress.begin(), cdcStreamKeys.begin); + progress.scanned(thirdGeneration, cdcStreamKeys.end, true, false); + ASSERT(!progress.needsScan(thirdGeneration)); + return Void(); +} + +} // namespace + +Future nativeCdcRetagCleanup(Database cx, + MoveKeysLock lock, + const DDEnabledState* ddEnabledState, + Future initialized) { + NativeCdcRetagCleanup cleanup(cx, lock, ddEnabledState); + co_await cleanup.run(initialized); +} + +void forceLinkNativeCdcRetagCleanupTests() {} diff --git a/fdbserver/datadistributor/NativeCdcBalancer.h b/fdbserver/datadistributor/NativeCdcRetagCleanup.h similarity index 63% rename from fdbserver/datadistributor/NativeCdcBalancer.h rename to fdbserver/datadistributor/NativeCdcRetagCleanup.h index 50bf3398dd5..16978ce7e1b 100644 --- a/fdbserver/datadistributor/NativeCdcBalancer.h +++ b/fdbserver/datadistributor/NativeCdcRetagCleanup.h @@ -1,5 +1,5 @@ /* - * NativeCdcBalancer.h + * NativeCdcRetagCleanup.h * * This source file is part of the FoundationDB open source project * @@ -23,10 +23,11 @@ #include "fdbclient/NativeAPI.h" #include "fdbserver/core/MoveKeys.h" -// The DD epoch fences publication of advisory producer-load samples. -Future nativeCdcBalancer(Database cx, - MoveKeysLock lock, - const DDEnabledState* ddEnabledState, - Future initialized); +// The DD epoch fences finalization of acknowledged transitions. Cleanup remains +// active while CDC admission and production of new transitions are disabled. +Future nativeCdcRetagCleanup(Database cx, + MoveKeysLock lock, + const DDEnabledState* ddEnabledState, + Future initialized); -void forceLinkNativeCdcBalancerTests(); +void forceLinkNativeCdcRetagCleanupTests(); diff --git a/fdbserver/logsystem/ApplyMetadataMutation.cpp b/fdbserver/logsystem/ApplyMetadataMutation.cpp index fe2c60f5010..71013155fef 100644 --- a/fdbserver/logsystem/ApplyMetadataMutation.cpp +++ b/fdbserver/logsystem/ApplyMetadataMutation.cpp @@ -575,7 +575,7 @@ class ApplyMetadataMutationsImpl { if (cdcStreamKeys.contains(m.param1)) { cdcRouting->setRanges(decodeCDCStreamKey(m.param1), decodeCDCStreamKeysValue(m.param2)); } else if (cdcTagHistoryKeys.contains(m.param1)) { - const CDCTagHistoryEntry history = decodeCDCTagHistoryKey(m.param1); + const CDCTagHistoryEntry history = decodeCDCTagHistoryEntry(m.param1, m.param2); cdcRouting->setTag(history.streamId, history.version, history.tag); } } diff --git a/fdbserver/logsystem/CDCRoutingTable.cpp b/fdbserver/logsystem/CDCRoutingTable.cpp index d720adb987d..9adf0a85b0f 100644 --- a/fdbserver/logsystem/CDCRoutingTable.cpp +++ b/fdbserver/logsystem/CDCRoutingTable.cpp @@ -76,7 +76,7 @@ void CDCRoutingTable::reload(IKeyValueStore* txnStateStore) { } const RangeResult tagHistoryRows = txnStateStore->readRange(cdcTagHistoryKeys).get(); for (const auto& kv : tagHistoryRows) { - const CDCTagHistoryEntry history = decodeCDCTagHistoryKey(kv.key); + const CDCTagHistoryEntry history = decodeCDCTagHistoryEntry(kv.key, kv.value); updateTag(history.streamId, history.version, history.tag); } rebuildRanges(); diff --git a/fdbserver/workloads/CMakeLists.txt b/fdbserver/workloads/CMakeLists.txt index f88415501a2..1b44c28a345 100644 --- a/fdbserver/workloads/CMakeLists.txt +++ b/fdbserver/workloads/CMakeLists.txt @@ -10,7 +10,8 @@ target_sources(fdbserver_workloads_test PRIVATE ../MemoryTrackerTest.cpp ../Glob configure_fdbserver_common_includes(fdbserver_workloads) target_include_directories(fdbserver_workloads PRIVATE - ${CMAKE_CURRENT_SOURCE_DIR}) + ${CMAKE_CURRENT_SOURCE_DIR} + ${CMAKE_SOURCE_DIR}/fdbclient) target_link_libraries(fdbserver_workloads PRIVATE fdbserver_cdcproxy fdbserver_consistencyscan diff --git a/fdbserver/workloads/NativeCdcEndToEnd.cpp b/fdbserver/workloads/NativeCdcEndToEnd.cpp index d78faed7f4f..e1a04591faf 100644 --- a/fdbserver/workloads/NativeCdcEndToEnd.cpp +++ b/fdbserver/workloads/NativeCdcEndToEnd.cpp @@ -21,12 +21,15 @@ #include #include #include +#include #include #include #include #include +#include "NativeCdcInternal.h" +#include "fdbserver/core/NativeCdcMetadata.h" #include "fdbclient/DatabaseContext.h" #include "fdbclient/NativeCdc.h" #include "fdbclient/SystemData.h" @@ -34,6 +37,8 @@ #include "fdbserver/core/Knobs.h" #include "fdbserver/core/RecoveryState.h" #include "fdbserver/core/ServerDBInfo.h" +#include "fdbserver/logsystem/LogSystemConsumer.h" +#include "fdbserver/logsystem/LogSystemFactory.h" #include "fdbserver/tester/workloads.h" #include "fdbrpc/simulator.h" #include "flow/DeterministicRandom.h" @@ -64,6 +69,110 @@ class NativeCdcEndToEndWorkload : public TestWorkload { std::unordered_map, ExpectedWrite, KeyValueHash> expected; }; + class RetagMarkerLedger : public ReferenceCounted { + Key markerKey; + std::unordered_map writes; + std::unordered_map markerKeys; + std::unordered_map> epochObservations; + Version committedThrough = invalidVersion; + Version acknowledgedThrough = invalidVersion; + int nextValue = 0; + int replayedMutations = 0; + + public: + explicit RetagMarkerLedger(Key key) : markerKey(std::move(key)) {} + const Key& key() const { return markerKey; } + Version lastCommittedVersion() const { return committedThrough; } + int replayCount() const { return replayedMutations; } + // Every unacknowledged marker must be delivered again after replacement, independently of earlier observations. + void allowReplay() { epochObservations.clear(); } + + Value expectWrite(int valueBytes, Optional key = Optional()) { + std::string bytes = format("retag/%010d/", nextValue++); + bytes.resize(valueBytes, 'x'); + Value value{ StringRef(bytes) }; + ASSERT(writes.emplace(value, ExpectedWrite{ invalidVersion, {} }).second); + markerKeys.emplace(value, key.present() ? key.get() : markerKey); + return value; + } + + void committed(Value const& value, Version version) { + writes.at(value).committedVersion = version; + committedThrough = std::max(committedThrough, version); + } + + void observe(CDCConsumeReply const& reply) { + Version previousGroup = invalidVersion; + for (const auto& versioned : reply.mutations) { + ASSERT_GT(versioned.version, previousGroup); + ASSERT_GT(versioned.version, acknowledgedThrough); + ASSERT_LE(versioned.version, reply.lastConsumedVersion); + previousGroup = versioned.version; + for (const auto& mutation : versioned.mutations) { + ASSERT_EQ(mutation.type, MutationRef::SetValue); + const Value value(mutation.param2); + auto expected = writes.find(value); + ASSERT(expected != writes.end()); + ASSERT_EQ(mutation.param1, markerKeys.at(value)); + ASSERT(epochObservations[value].insert(versioned.version).second); + if (expected->second.committedVersion != invalidVersion) { + ASSERT_LE(versioned.version, expected->second.committedVersion); + } + if (!expected->second.observedVersions.insert(versioned.version).second) { + ++replayedMutations; + } + } + } + } + + void verifyThrough(Version version) const { + for (const auto& [value, expected] : writes) { + ASSERT_NE(expected.committedVersion, invalidVersion); + if (expected.committedVersion > acknowledgedThrough && expected.committedVersion <= version) { + const auto observed = epochObservations.find(value); + ASSERT(observed != epochObservations.end()); + ASSERT(observed->second.contains(expected.committedVersion)); + } + } + } + + void verifyBoundary(Version boundary) const { + bool before = false; + bool after = false; + for (const auto& [value, expected] : writes) { + before |= expected.committedVersion < boundary; + after |= expected.committedVersion >= boundary; + } + ASSERT(before && after); + verifyThrough(committedThrough); + } + + void acknowledged(Version version) { + verifyThrough(version); + acknowledgedThrough = version; + } + }; + + struct RetagSnapshot { + NativeCdcTagState state; + std::vector history; + }; + + struct RetagFixtureAttempt { + Version readVersion; + Version gapVersion; + Version committedVersion = invalidVersion; + }; + + struct RetagRestartMarkers { + CDCStreamId streamId; + Version before; + Version cutover; + Version after; + Tag oldTag; + Tag newTag; + }; + int initialStreamCount; int minStreamCount; int maxStreamCount; @@ -84,8 +193,12 @@ class NativeCdcEndToEndWorkload : public TestWorkload { bool testRetiredRecovery; bool blockRetiredPopWithLiveStream; bool testRetiredSharedTagSnapshot; + bool testRetagCompatibility; + bool testRetaggingMemoryBound; bool prepareRestartDrain; bool drainAfterRestart; + bool testRetaggedRestart; + bool testRetagTransactionRetries; int memoryTestValueBytes; double retentionValidationDelay; double drainProbability; @@ -376,6 +489,436 @@ class NativeCdcEndToEndWorkload : public TestWorkload { co_await addStream(cx, KeyRange(KeyRangeRef(keyForIndex(0), keyForIndex(keyCount)))); } + Future initializeRetaggingStreams(Database cx) { + for (int i = 0; i < initialStreamCount; ++i) { + const Key key = keyForIndex(i); + const Key end = testRetaggingMemoryBound ? keyForIndex(i + 1) : keyAfter(key); + co_await addStream(cx, KeyRange(KeyRangeRef(key, end))); + } + } + + Future readRetagSnapshot(Transaction* tr, int index, int maxStreams) { + const CDCStreamId streamId = streams[index].consumer->position().streamId; + const auto states = co_await readNativeCdcTagStates(tr, maxStreams); + ASSERT(states.present()); + const auto found = std::find_if(states.get().begin(), states.get().end(), [streamId](const auto& state) { + return state.streamId == streamId; + }); + ASSERT(found != states.get().end()); + RetagSnapshot result; + result.state = *found; + ASSERT(result.state.ranges == std::vector{ streams[index].keys }); + const RangeResult history = co_await tr->getRange(cdcTagHistoryRangeFor(streamId), 3); + ASSERT(!history.more && !history.empty() && history.size() <= 2); + for (const auto& row : history) { + result.history.push_back(decodeCDCTagHistoryEntry(row.key, row.value)); + } + co_return result; + } + + Future readRetagSnapshot(Database cx, int index, int maxStreams = 16) { + RetagSnapshot result; + // NOLINTNEXTLINE(cppcoreguidelines-avoid-capturing-lambda-coroutines) Database::run owns the closure. + co_await cx.run([this, &result, index, maxStreams](Transaction* tr) -> Future { + tr->setOption(FDBTransactionOptions::READ_LOCK_AWARE); + tr->setOption(FDBTransactionOptions::READ_SYSTEM_KEYS); + result = co_await readRetagSnapshot(tr, index, maxStreams); + }); + co_return result; + } + + Future writeRetagMarkers(Database cx, + std::vector indices, + std::vector> ledgers) { + std::vector> values; + values.reserve(indices.size()); + for (const int index : indices) { + values.emplace_back(ledgers[index]->key(), ledgers[index]->expectWrite(memoryTestValueBytes)); + } + const Version committed = co_await writeValues(cx, values); + for (int i = 0; i < static_cast(indices.size()); ++i) { + ledgers[indices[i]]->committed(values[i].second, committed); + } + co_return committed; + } + + Future drainRetagMarkers(int index, Reference ledger, bool acknowledge) { + const double deadline = now() + operationTimeout; + while (streams[index].consumer->position().lastConsumedVersion < ledger->lastCommittedVersion()) { + ledger->observe(co_await timeoutError(streams[index].consumer->consume(), operationTimeout)); + ASSERT_LT(now(), deadline); + co_await delay(0.01); + } + ledger->verifyThrough(ledger->lastCommittedVersion()); + if (acknowledge) { + const Version position = streams[index].consumer->position().lastConsumedVersion; + co_await timeoutError(streams[index].consumer->acknowledge(), operationTimeout); + ledger->acknowledged(position); + } + } + + Future waitForCanonicalRetag(Database cx, int index, CDCTagHistoryEntry assignment) { + const double deadline = now() + operationTimeout; + while (true) { + const RetagSnapshot snapshot = co_await readRetagSnapshot(cx, index); + ASSERT_EQ(snapshot.state.assignment.tag, assignment.tag); + ASSERT_EQ(snapshot.state.assignment.version, assignment.version); + if (!snapshot.state.pending) { + ASSERT_EQ(snapshot.history.size(), 1); + co_return; + } + ASSERT_LT(now(), deadline); + co_await delay(0.05); + } + } + + Future commitRetagFixture(Database cx, + int index, + RetagSnapshot original, + Tag destination, + int maxStreams, + std::vector> gapLedgers) { + ASSERT(!original.state.pending); + ASSERT_EQ(original.history.size(), 1); + std::unordered_map> attempts; + bool readRetryInjected = false; + bool commitRetryInjected = false; + bool ambiguousCommit = false; + Transaction tr(cx); + while (true) { + Error err; + try { + tr.setOption(FDBTransactionOptions::LOCK_AWARE); + tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); + if (testRetagTransactionRetries && !readRetryInjected) { + readRetryInjected = true; + throw future_version(); + } + RetagSnapshot snapshot = co_await readRetagSnapshot(&tr, index, maxStreams); + ASSERT_EQ(snapshot.state.streamId, original.state.streamId); + ASSERT(snapshot.state.ranges == original.state.ranges); + ASSERT_EQ(snapshot.state.minVersion, original.state.minVersion); + if (snapshot.state.pending) { + ASSERT_EQ(snapshot.history.size(), 2); + ASSERT_EQ(snapshot.history.front().tag, original.state.assignment.tag); + ASSERT_EQ(snapshot.history.front().version, original.state.assignment.version); + ASSERT_EQ(snapshot.state.assignment.tag, destination); + // An ambiguous commit may already have installed our exact history row. Never turn that retry into + // another move, or accept an unrelated move merely because it chose the same destination. + const auto submitted = attempts.find(snapshot.state.historyKey); + ASSERT(submitted != attempts.end()); + const Version cutover = snapshot.state.assignment.version; + ASSERT_LT(snapshot.state.minVersion, cutover); + bool matchesAttempt = false; + for (const auto& attempt : submitted->second) { + if (attempt.committedVersion != invalidVersion) { + ASSERT_EQ(cutover, attempt.committedVersion); + } + matchesAttempt |= cutover > attempt.readVersion && + (attempt.gapVersion == invalidVersion || + (attempt.gapVersion > attempt.readVersion && cutover > attempt.gapVersion)); + } + ASSERT(matchesAttempt); + ASSERT(!testRetagTransactionRetries || (readRetryInjected && ambiguousCommit)); + CODE_PROBE(ambiguousCommit, "Native CDC retag fixture recognizes its own ambiguous committed move"); + co_return snapshot; + } + ASSERT_EQ(snapshot.history.size(), 1); + ASSERT_EQ(snapshot.state.historyKey, original.state.historyKey); + ASSERT_EQ(snapshot.state.assignment.tag, original.state.assignment.tag); + ASSERT_EQ(snapshot.state.assignment.version, original.state.assignment.version); + const bool prepared = co_await retagNativeCdcStream(&tr, snapshot.state, destination); + ASSERT(prepared); + const Version readVersion = co_await tr.getReadVersion(); + Version gapVersion = invalidVersion; + if (!gapLedgers.empty()) { + // Each retried attempt needs its own separately committed marker after its fresh read version. + // Failed attempts remain in the ledger and must still be delivered. + gapVersion = co_await writeRetagMarkers(cx, { index }, gapLedgers); + ASSERT_GT(gapVersion, readVersion); + } + const Key historyKey = cdcTagHistoryKeyFor(snapshot.state.streamId, readVersion, destination); + auto& attempt = attempts[historyKey].emplace_back(RetagFixtureAttempt{ readVersion, gapVersion }); + co_await tr.commit(); + attempt.committedVersion = tr.getCommittedVersion(); + if (testRetagTransactionRetries && !commitRetryInjected) { + commitRetryInjected = true; + throw commit_unknown_result(); + } + tr.reset(); + continue; + } catch (Error& e) { + ambiguousCommit |= e.code() == error_code_commit_unknown_result; + err = e; + } + co_await tr.onError(err); + } + } + + Future assertRetagRejected(Database cx, NativeCdcTagState expected, Tag destination) { + // NOLINTNEXTLINE(cppcoreguidelines-avoid-capturing-lambda-coroutines) Database::run owns the closure. + co_await cx.run([expected = std::move(expected), destination](Transaction* tr) -> Future { + tr->setOption(FDBTransactionOptions::LOCK_AWARE); + tr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); + const bool prepared = co_await retagNativeCdcStream(tr, expected, destination); + ASSERT(!prepared); + }); + } + + Future retagAcrossConcurrentWrite(Database cx, + int index, + Tag destination, + std::vector> ledgers) { + const RetagSnapshot original = co_await readRetagSnapshot(cx, index); + RetagSnapshot snapshot = co_await commitRetagFixture(cx, index, original, destination, 16, ledgers); + co_await assertRetagRejected(cx, original.state, destination); + co_await assertRetagRejected(cx, snapshot.state, original.state.assignment.tag); + CODE_PROBE(true, "Native CDC live retag uses its commit boundary and rejects stale or pending moves"); + co_return snapshot; + } + + Future writeRetagBatch(Database cx, Reference ledger) { + std::vector> values; + // Small independent mutations fit in one raw peek but expand beyond it when materialized. + for (int i = 0; i < 12; ++i) { + Key key = ledger->key().withSuffix(StringRef(format("/%02d", i))); + values.emplace_back(key, ledger->expectWrite(32, key)); + } + const Version committed = co_await writeValues(cx, values); + for (const auto& [key, value] : values) { + ledger->committed(value, committed); + } + co_return committed; + } + + Future consumeContendedRetag(int index, + Reference ledger, + Version through, + Reference> firstBatches, + Future releaseAcknowledgements) { + bool first = true; + while (streams[index].consumer->position().lastConsumedVersion < through) { + const CDCConsumeReply reply = co_await streams[index].consumer->consume(); + ledger->observe(reply); + ledger->verifyThrough(reply.lastConsumedVersion); + if (first && !reply.mutations.empty()) { + first = false; + firstBatches->set(firstBatches->get() + 1); + co_await releaseAcknowledgements; + } + co_await streams[index].consumer->acknowledge(); + ledger->acknowledged(reply.lastConsumedVersion); + } + ASSERT(!first); + ledger->verifyThrough(ledger->lastCommittedVersion()); + } + + Future retainedTagBytes(Tag tag, Version begin, Version end) { + Reference logs = makeLogSystemConsumerFromServerDBInfo(UID(), dbInfo->get()); + Reference cursor = logs->peekSingle(UID(), begin, tag); + int64_t bytes = 0; + while (cursor->version().version <= end) { + if (!cursor->hasMessage()) { + co_await cursor->getMore(); + ASSERT_LE(cursor->popped(), begin); + continue; + } + bytes += cursor->getMessageWithTags().size(); + cursor->nextMessage(); + } + co_return bytes; + } + + void checkRetagBufferStatus(CDCProxyBufferStatus const& status) const { + ASSERT_GE(status.bufferedBytes, 0); + ASSERT_LE(status.bufferedBytes, status.activePermits); + ASSERT_LE(status.activePermits, status.bufferLimit); + ASSERT_LE(status.peakActivePermits, status.bufferLimit); + } + + Future getRetagBufferStatus(Database cx, UID owner) { + const auto result = co_await timeoutError( + getAssignedProxyStatus(cx, streams.front().consumer->position().streamId), operationTimeout); + ASSERT_EQ(result.first.id(), owner); + checkRetagBufferStatus(result.second); + co_return result.second; + } + + Future validateRetaggingMemoryBound(Database cx) { + ASSERT_EQ(streams.size(), 2); + const RetagSnapshot original = co_await readRetagSnapshot(cx, 0); + const RetagSnapshot destination = co_await readRetagSnapshot(cx, 1); + ASSERT_NE(original.state.assignment.tag, destination.state.assignment.tag); + ASSERT_EQ(original.state.proxyId, destination.state.proxyId); + const Tag oldTag = original.state.assignment.tag; + const Tag newTag = destination.state.assignment.tag; + auto moving = makeReference(streams[0].keys.begin); + auto active = makeReference(streams[1].keys.begin); + const Version before = co_await writeRetagBatch(cx, moving); + const double oldestCommittedAt = now(); + const RetagSnapshot pending = co_await commitRetagFixture(cx, 0, original, newTag, 16, {}); + const Version destinationVersion = co_await writeRetagBatch(cx, active); + const Version after = co_await writeRetagBatch(cx, moving); + ASSERT_LT(before, pending.state.assignment.version); + ASSERT_GE(after, pending.state.assignment.version); + + auto barrier = makeReference(original.state.proxyId, oldTag, newTag); + CDCProxyMaterializationTest::install(barrier); + ScopeExit removeBarrier([] { CDCProxyMaterializationTest::uninstall(); }); + Promise releaseAcknowledgements; + auto firstBatches = makeReference>(0); + std::vector> consumers{ + consumeContendedRetag(0, moving, after, firstBatches, releaseAcknowledgements.getFuture()), + consumeContendedRetag(1, active, after, firstBatches, releaseAcknowledgements.getFuture()) + }; + const double deadline = now() + operationTimeout; + while (!barrier->bothReadersHeld()) { + for (const auto& consumer : consumers) { + if (consumer.isReady()) { + consumer.get(); + ASSERT(false); + } + } + ASSERT_LT(now(), deadline); + co_await delay(0.01); + } + auto status = co_await getRetagBufferStatus(cx, original.state.proxyId); + ASSERT_EQ(status.activePermits, barrier->heldBytes()); + ASSERT_EQ(status.activePermits, status.bufferLimit); + ASSERT_EQ(status.bufferedBytes, 0); + ASSERT_EQ(firstBatches->get(), 0); + ASSERT((co_await readRetagSnapshot(cx, 0)).state.pending); + TraceEvent("NativeCdcRetagContendedReaders") + .detail("Cutover", pending.state.assignment.version) + .detail("ActivePermits", status.activePermits) + .detail("BufferLimit", status.bufferLimit); + + const double releasedAt = now(); + barrier->release(); + // Both real readers must deliver before either acknowledges, and well before a consume lease can expire. + while (firstBatches->get() < 2) { + co_await timeoutError(firstBatches->onChange(), std::max(0.0, releasedAt + operationTimeout - now())); + } + ASSERT_EQ(barrier->leaseExpiries(), 0); + ASSERT_LT(now() - releasedAt, SERVER_KNOBS->CDC_PROXY_CONSUME_POLL_TIMEOUT); + status = co_await getRetagBufferStatus(cx, original.state.proxyId); + ASSERT_GT(status.bufferedBytes, 0); + const int64_t oldBytes = co_await timeoutError(retainedTagBytes(oldTag, before, before), operationTimeout); + const int64_t newBytes = + co_await timeoutError(retainedTagBytes(newTag, destinationVersion, after), operationTimeout); + ASSERT_GT(oldBytes, 0); + ASSERT_GT(newBytes, 0); + const double pauseStarted = now(); + co_await delay(retentionValidationDelay); + ASSERT_EQ(co_await timeoutError(retainedTagBytes(oldTag, before, before), operationTimeout), oldBytes); + ASSERT_EQ(co_await timeoutError(retainedTagBytes(newTag, destinationVersion, after), operationTimeout), + newBytes); + const RetagSnapshot held = co_await readRetagSnapshot(cx, 0); + ASSERT(held.state.pending); + ASSERT_EQ(held.state.minVersion, original.state.minVersion); + status = co_await getRetagBufferStatus(cx, original.state.proxyId); + TraceEvent("NativeCdcRetagRetentionPause") + .detail("OldTagBytes", oldBytes) + .detail("DestinationTagBytes", newBytes) + .detail("PauseSeconds", now() - pauseStarted) + .detail("OldestCommitAgeSeconds", now() - oldestCommittedAt) + .detail("BufferedBytes", status.bufferedBytes) + .detail("PeakActivePermits", status.peakActivePermits); + releaseAcknowledgements.send(Void()); + co_await timeoutError(waitForAll(consumers), operationTimeout); + ASSERT_EQ(barrier->leaseExpiries(), 0); + co_await waitForCanonicalRetag(cx, 0, pending.state.assignment); + Reference logs = makeLogSystemConsumerFromServerDBInfo(UID(), dbInfo->get()); + co_await timeoutError(logs->waitForPopped(pending.state.assignment.version, oldTag), operationTimeout); + co_await timeoutError(logs->waitForPopped(after + 1, newTag), operationTimeout); + status = co_await getRetagBufferStatus(cx, original.state.proxyId); + ASSERT_EQ(status.bufferedBytes, 0); + CODE_PROBE(true, "Native CDC retagging progresses under competing expanded reservations without lease expiry"); + CODE_PROBE(true, "Native CDC retains both retag histories during an acknowledgement pause then drains"); + for (const auto& stream : streams) { + co_await removeNativeCdcStreamClient(cx, stream.name); + } + streams.clear(); + co_await timeoutError(waitForRetiredTagCleanup(cx), operationTimeout); + } + + Future validateRetagCompatibility(Database cx) { + ASSERT_EQ(streams.size(), 4); + ASSERT_EQ(cx->clientInfo->get().nativeCdcTagCount, 2); + std::vector> ledgers; + std::vector initial; + for (int i = 0; i < static_cast(streams.size()); ++i) { + ledgers.push_back(makeReference(streams[i].keys.begin)); + initial.push_back(co_await readRetagSnapshot(cx, i)); + ASSERT(!initial.back().state.pending); + ASSERT_EQ(initial.back().state.proxyId, initial.front().state.proxyId); + } + const Tag originalTag = initial.front().state.assignment.tag; + const Tag destination(tagLocalityCDC, originalTag.id == 0 ? 1 : 0); + int sibling = -1; + for (int i = 1; i < static_cast(streams.size()); ++i) { + if (initial[i].state.assignment.tag == originalTag) { + sibling = i; + break; + } + } + ASSERT_GE(sibling, 0); + co_await writeRetagMarkers(cx, { 0, 1, 2, 3 }, ledgers); + const RetagSnapshot pending = co_await retagAcrossConcurrentWrite(cx, 0, destination, ledgers); + co_await writeRetagMarkers(cx, { 0, 1, 2, 3 }, ledgers); + co_await drainRetagMarkers(0, ledgers[0], false); + ledgers[0]->verifyBoundary(pending.state.assignment.version); + const RetagSnapshot unacknowledged = co_await readRetagSnapshot(cx, 0); + ASSERT_EQ(unacknowledged.history.size(), 2); + + const CDCProxyInterface originalOwner = + co_await timeoutError(waitForAssignedProxy(cx, pending.state.streamId), operationTimeout); + ledgers[0]->allowReplay(); + const int previousReplays = ledgers[0]->replayCount(); + co_await timeoutError(haltProxyUntilReplaced(cx, originalOwner, false), operationTimeout); + co_await timeoutError(waitForAssignedProxy(cx, pending.state.streamId, originalOwner.id()), operationTimeout); + co_await forceTransactionSystemRecovery(); + co_await writeRetagMarkers(cx, { 0, sibling }, ledgers); + co_await drainRetagMarkers(0, ledgers[0], false); + ASSERT_GT(ledgers[0]->replayCount(), previousReplays); + ledgers[0]->verifyBoundary(pending.state.assignment.version); + const RetagSnapshot recovered = co_await readRetagSnapshot(cx, 0); + ASSERT_EQ(recovered.state.minVersion, initial[0].state.minVersion); + co_await drainRetagMarkers(0, ledgers[0], true); + co_await waitForCanonicalRetag(cx, 0, pending.state.assignment); + CODE_PROBE(true, "Native CDC compatibility reader replays both retag intervals after recovery"); + + // An unacknowledged sibling still protects the retired tag after this stream's transition is canonical. + const RetagSnapshot lagging = co_await readRetagSnapshot(cx, sibling); + ASSERT_EQ(lagging.state.minVersion, initial[sibling].state.minVersion); + ASSERT_EQ(lagging.state.assignment.tag, originalTag); + // NOLINTNEXTLINE(cppcoreguidelines-avoid-capturing-lambda-coroutines) Database::run owns the closure. + co_await cx.run([originalTag, pending](Transaction* tr) -> Future { + tr->setOption(FDBTransactionOptions::READ_LOCK_AWARE); + tr->setOption(FDBTransactionOptions::READ_SYSTEM_KEYS); + const Optional retired = co_await tr->get(cdcRetiredTagPopKeyFor(originalTag)); + const Optional watermark = co_await tr->get(cdcRetiredTagPopVersionKeyFor(originalTag)); + ASSERT(retired.present() && watermark.present()); + ASSERT_GE(decodeCDCMinVersionValue(watermark.get()), pending.state.assignment.version); + }); + const RetagSnapshot returned = co_await retagAcrossConcurrentWrite(cx, 0, originalTag, ledgers); + co_await writeRetagMarkers(cx, { 0 }, ledgers); + co_await drainRetagMarkers(0, ledgers[0], true); + ledgers[0]->verifyBoundary(returned.state.assignment.version); + co_await waitForCanonicalRetag(cx, 0, returned.state.assignment); + for (int i = 1; i < static_cast(streams.size()); ++i) { + co_await drainRetagMarkers(i, ledgers[i], true); + } + CODE_PROBE(true, + "Native CDC retag compatibility preserves shared-tag data and supports returning to an old tag"); + for (const auto& stream : streams) { + co_await timeoutError(removeNativeCdcStreamClient(cx, stream.name), operationTimeout); + } + streams.clear(); + co_await timeoutError(waitForRetiredTagCleanup(cx), operationTimeout); + co_await timeoutError(waitForFullyRecovered(), operationTimeout); + } + Future validatePublicLifecycle(Database cx) { const Key name = "native-cdc-e2e/lifecycle"_sr; const KeyRange keys(KeyRangeRef("native-cdc-e2e/lifecycle/"_sr, "native-cdc-e2e/lifecycle0"_sr)); @@ -2083,9 +2626,84 @@ class NativeCdcEndToEndWorkload : public TestWorkload { CODE_PROBE(true, "Native CDC retired tag cleanup allows recovery to complete"); } + Future prepareRetaggedRestartState(Database cx, Version before) { + ASSERT_EQ(cx->clientInfo->get().nativeCdcTagCount, 2); + const RetagSnapshot original = co_await readRetagSnapshot(cx, 0, 1); + const Tag destination(tagLocalityCDC, original.state.assignment.tag.id == 0 ? 1 : 0); + const RetagSnapshot committed = co_await commitRetagFixture(cx, 0, original, destination, 1, {}); + const Version cutover = committed.state.assignment.version; + const Version after = co_await writeValue(cx, keyForIndex(keyCount / 2), "native-cdc-restart-after-retag"_sr); + ASSERT_LT(before, cutover); + ASSERT_GT(after, cutover); + + // Keep exact marker versions outside the tracked range so the restarted reader can verify both log intervals. + BinaryWriter fixture{ Unversioned() }; + fixture << original.state.streamId << before << cutover << after << original.state.assignment.tag + << destination; + co_await writeValue(cx, "native-cdc-e2e/restart-retag-state"_sr, fixture.toValue()); + const RetagSnapshot pending = co_await readRetagSnapshot(cx, 0); + ASSERT(pending.state.pending); + ASSERT_EQ(pending.history.size(), 2); + ASSERT_EQ(pending.state.assignment.version, cutover); + ASSERT_LT(pending.state.minVersion, cutover); + CODE_PROBE(true, "Native CDC restart preserves an unacknowledged retag boundary"); + } + + Future loadRetaggedRestartState(Database cx, Key name, Reference consumer) { + const std::vector listed = + co_await timeoutError(listNativeCdcStreamsClient(cx), operationTimeout); + ASSERT_EQ(listed.size(), 1); + ASSERT_EQ(listed.front().name, name); + ASSERT_EQ(listed.front().streamId, consumer->position().streamId); + StreamState stream; + stream.name = name; + ASSERT_EQ(listed.front().ranges.size(), 1); + stream.keys = listed.front().ranges.front(); + stream.consumer = consumer; + streams.push_back(std::move(stream)); + + RetagRestartMarkers markers; + // NOLINTNEXTLINE(cppcoreguidelines-avoid-capturing-lambda-coroutines) Database::run owns the closure. + co_await cx.run([this, &markers, &consumer, &cx](Transaction* tr) -> Future { + const Optional fixture = co_await tr->get("native-cdc-e2e/restart-retag-state"_sr); + ASSERT(fixture.present()); + BinaryReader reader(fixture.get(), Unversioned()); + reader >> markers.streamId >> markers.before >> markers.cutover >> markers.after >> markers.oldTag >> + markers.newTag; + ASSERT_EQ(markers.streamId, consumer->position().streamId); + const RetagSnapshot pending = co_await readRetagSnapshot(cx, 0); + ASSERT(pending.state.pending); + ASSERT_EQ(pending.history.size(), 2); + ASSERT_EQ(pending.history.front().tag, markers.oldTag); + ASSERT_EQ(pending.state.assignment.tag, markers.newTag); + ASSERT_EQ(pending.state.assignment.version, markers.cutover); + ASSERT_LT(pending.state.minVersion, markers.cutover); + }); + co_return markers; + } + + Future finishRetaggedRestartState(Database cx, RetagRestartMarkers markers) { + const RetagSnapshot acknowledged = co_await readRetagSnapshot(cx, 0); + co_await waitForCanonicalRetag(cx, 0, acknowledged.state.assignment); + // NOLINTNEXTLINE(cppcoreguidelines-avoid-capturing-lambda-coroutines) Database::run owns the closure. + co_await cx.run([this, markers](Transaction* tr) -> Future { + tr->setOption(FDBTransactionOptions::LOCK_AWARE); + tr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); + const RetagSnapshot completed = co_await readRetagSnapshot(tr, 0, 1); + ASSERT(!completed.state.pending); + ASSERT_EQ(completed.state.assignment.version, markers.cutover); + const bool prepared = co_await retagNativeCdcStream(tr, completed.state, markers.oldTag); + ASSERT(!prepared); + }); + CODE_PROBE(true, "Native CDC disabled admission finishes pending retags without admitting new moves"); + } + Future prepareRestartDrainState(Database cx) { ASSERT_EQ(streams.size(), 1); - co_await writeValue(cx, keyForIndex(keyCount / 2), "native-cdc-restart-drain"_sr); + const Version before = co_await writeValue(cx, keyForIndex(keyCount / 2), "native-cdc-restart-drain"_sr); + if (testRetaggedRestart) { + co_await timeoutError(prepareRetaggedRestartState(cx, before), operationTimeout); + } CODE_PROBE(true, "Native CDC restart marker is durable before save and kill"); } @@ -2107,18 +2725,44 @@ class NativeCdcEndToEndWorkload : public TestWorkload { ASSERT_EQ(activeStatus.streams.size(), 1); ASSERT_EQ(activeStatus.streams.front().info.name, name); ASSERT_EQ(activeStatus.streams.front().info.streamId, consumer->position().streamId); + Optional retagMarkers; + if (testRetaggedRestart) { + retagMarkers = co_await timeoutError(loadRetaggedRestartState(cx, name, consumer), operationTimeout); + } bool observed = false; - while (!observed) { + bool observedAfterRetag = !testRetaggedRestart; + const double retagDeadline = now() + operationTimeout; + while (!observed || !observedAfterRetag) { + if (testRetaggedRestart) { + ASSERT_LT(now(), retagDeadline); + } CDCConsumeReply reply = co_await timeoutError(consumer->consume(), operationTimeout); for (const auto& versioned : reply.mutations) { for (const auto& mutation : versioned.mutations) { if (mutation.type == MutationRef::SetValue && mutation.param1 == keyForIndex(keyCount / 2) && mutation.param2 == "native-cdc-restart-drain"_sr) { - observed = true; + if (retagMarkers.present()) { + ASSERT_LT(versioned.version, retagMarkers.get().cutover); + observed |= versioned.version == retagMarkers.get().before; + } else { + observed = true; + } + } + if (retagMarkers.present() && mutation.type == MutationRef::SetValue && + mutation.param1 == keyForIndex(keyCount / 2) && + mutation.param2 == "native-cdc-restart-after-retag"_sr) { + ASSERT_GE(versioned.version, retagMarkers.get().cutover); + observedAfterRetag |= versioned.version == retagMarkers.get().after; } } } + if (!testRetaggedRestart) { + co_await timeoutError(consumer->acknowledge(), operationTimeout); + } + } + if (testRetaggedRestart) { co_await timeoutError(consumer->acknowledge(), operationTimeout); + co_await timeoutError(finishRetaggedRestartState(cx, retagMarkers.get()), operationTimeout); } const NativeCdcRemoveResult removed = co_await timeoutError( removeNativeCdcStreamGuarded(cx, name, consumer->position().streamId), operationTimeout); @@ -2220,6 +2864,14 @@ class NativeCdcEndToEndWorkload : public TestWorkload { } Future run(Database cx) { + if (testRetagCompatibility) { + co_await validateRetagCompatibility(cx); + co_return; + } + if (testRetaggingMemoryBound) { + co_await validateRetaggingMemoryBound(cx); + co_return; + } if (testBufferContention) { co_await validateBufferContention(cx); co_return; @@ -2335,8 +2987,12 @@ class NativeCdcEndToEndWorkload : public TestWorkload { testRetiredRecovery = getOption(options, "testRetiredRecovery"_sr, false); blockRetiredPopWithLiveStream = getOption(options, "blockRetiredPopWithLiveStream"_sr, false); testRetiredSharedTagSnapshot = getOption(options, "testRetiredSharedTagSnapshot"_sr, false); + testRetagCompatibility = getOption(options, "testRetagCompatibility"_sr, false); + testRetaggingMemoryBound = getOption(options, "testRetaggingMemoryBound"_sr, false); prepareRestartDrain = getOption(options, "prepareRestartDrain"_sr, false); drainAfterRestart = getOption(options, "drainAfterRestart"_sr, false); + testRetaggedRestart = getOption(options, "testRetaggedRestart"_sr, false); + testRetagTransactionRetries = getOption(options, "testRetagTransactionRetries"_sr, false); memoryTestValueBytes = getOption(options, "memoryTestValueBytes"_sr, 1024); retentionValidationDelay = getOption(options, "retentionValidationDelay"_sr, 0.0); drainProbability = getOption(options, "drainProbability"_sr, 0.25); @@ -2353,9 +3009,18 @@ class NativeCdcEndToEndWorkload : public TestWorkload { ASSERT_GT(memoryTestValueBytes, 0); ASSERT_GE(retentionValidationDelay, 0.0); ASSERT(!(prepareRestartDrain && drainAfterRestart)); + ASSERT(!testRetaggedRestart || prepareRestartDrain || drainAfterRestart); + ASSERT(!testRetagTransactionRetries || testRetagCompatibility || testRetaggedRestart); ASSERT(!(testReplyChunking && (testOversizedPeek || testDurableAckScan))); ASSERT(!(testOversizedPeek && testDurableAckScan)); ASSERT(!(testRetiredSharedTagSnapshot && testRetiredRecovery)); + ASSERT(!(testRetagCompatibility && testMemoryBound)); + ASSERT(!testRetaggingMemoryBound || (!testRetagCompatibility && !testMemoryBound && !prepareRestartDrain && + !drainAfterRestart && initialStreamCount == 2 && keyCount >= 2)); + ASSERT(!testRetagCompatibility || + (initialStreamCount == 4 && keyCount >= 4 && memoryTestValueBytes >= 32 && !prepareRestartDrain && + !drainAfterRestart && !testRetiredSharedTagSnapshot && !testOversizedPeek && !testReplyChunking && + !testDurableAckScan)); ASSERT(!blockRetiredPopWithLiveStream || testRetiredRecovery); ASSERT(!testBufferContention || (initialStreamCount == 2 && !prepareRestartDrain && !drainAfterRestart && !testMultipleRanges && !testRetiredSharedTagSnapshot && !testMemoryBound)); @@ -2377,6 +3042,9 @@ class NativeCdcEndToEndWorkload : public TestWorkload { if (prepareRestartDrain) { return prepareRestartDrainSetup(cx); } + if (testRetagCompatibility || testRetaggingMemoryBound) { + return initializeRetaggingStreams(cx); + } if (testRetiredSharedTagSnapshot) { return Void(); } diff --git a/fdbserver/workloads/NativeCdcInitialPlacement.cpp b/fdbserver/workloads/NativeCdcInitialPlacement.cpp deleted file mode 100644 index 7b54b615c4f..00000000000 --- a/fdbserver/workloads/NativeCdcInitialPlacement.cpp +++ /dev/null @@ -1,198 +0,0 @@ -/* - * NativeCdcInitialPlacement.cpp - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -#include -#include -#include - -#include "fdbclient/DatabaseContext.h" -#include "fdbclient/NativeCdc.h" -#include "fdbclient/SystemData.h" -#include "fdbserver/core/NativeCdcMetadata.h" -#include "fdbserver/tester/workloads.h" -#include "flow/CodeProbe.h" - -class NativeCdcInitialPlacementWorkload : public TestWorkload { - const Key coldName = "native-cdc-placement/cold"_sr; - const Key hotName = "native-cdc-placement/hot"_sr; - const Key duplicateName = "native-cdc-placement/duplicate"_sr; - const Key placedName = "native-cdc-placement/placed"_sr; - const Key coldKey = "native-cdc-placement/data/cold"_sr; - const Key hotKey = "native-cdc-placement/data/hot"_sr; - const Key placedKey = "native-cdc-placement/data/placed"_sr; - const double operationTimeout; - - static Future writeValue(Database cx, Key key, Value value) { - Transaction tr(cx); - while (true) { - Error error; - try { - tr.set(key, value); - co_await tr.commit(); - co_return tr.getCommittedVersion(); - } catch (Error& e) { - error = e; - } - co_await tr.onError(error); - } - } - - static Future produceHotWrites(Database cx, Key key) { - int sequence = 0; - while (true) { - const Value value = StringRef(std::string(32768, 'a' + (++sequence % 26))); - co_await writeValue(cx, key, value); - co_await delay(0.05); - } - } - - static Future readTag(Database cx, CDCStreamId streamId) { - Transaction tr(cx); - while (true) { - Error error; - try { - tr.setOption(FDBTransactionOptions::READ_LOCK_AWARE); - tr.setOption(FDBTransactionOptions::READ_SYSTEM_KEYS); - const RangeResult history = co_await tr.getRange(cdcTagHistoryRangeFor(streamId), 2); - ASSERT_EQ(history.size(), 1); - ASSERT(!history.more); - ASSERT(history.front().value.empty()); - co_return decodeCDCTagHistoryKey(history.front().key).tag; - } catch (Error& e) { - error = e; - } - co_await tr.onError(error); - } - } - - static Future usableLoad(Transaction* tr, Tag coldTag, Tag hotTag) { - const Value generation = (co_await tr->get(cdcProxyAssignmentChangeKey)).orDefault(Value()); - const Version version = co_await tr->getReadVersion(); - const Optional coldValue = co_await tr->get(cdcTagLoadKeyFor(coldTag)); - const Optional hotValue = co_await tr->get(cdcTagLoadKeyFor(hotTag)); - if (!coldValue.present() || !hotValue.present()) { - co_return false; - } - const auto cold = decodeCDCTagLoadValue(coldValue.get()); - const auto hot = decodeCDCTagLoadValue(hotValue.get()); - for (const auto& sample : { cold, hot }) { - if (sample.assignmentChange != generation || sample.sampleVersion < 0 || sample.sampleVersion > version || - sample.validThrough < version || sample.bytesWrittenPerKSecond < 0) { - co_return false; - } - } - co_return hot.bytesWrittenPerKSecond > cold.bytesWrittenPerKSecond; - } - - Future registerWithLoad(Database cx, Tag coldTag, Tag hotTag) { - Transaction tr(cx); - std::set attemptedIds; - while (true) { - Error error; - try { - tr.setOption(FDBTransactionOptions::LOCK_AWARE); - tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); - const Optional existing = co_await tr.get(cdcStreamNameKeyFor(placedName)); - if (existing.present()) { - // A successful but ambiguous commit invalidates its own sample generation. - const CDCStreamId streamId = decodeCDCStreamNameValue(existing.get()); - ASSERT(attemptedIds.contains(streamId)); - co_return streamId; - } - // Guard the actual registration snapshot: a separate readiness check can expire on recovery. - if (!(co_await usableLoad(&tr, coldTag, hotTag)) || cx->clientInfo->get().cdcProxies.empty()) { - tr.reset(); - co_await delay(0.1); - continue; - } - const auto result = co_await prepareNativeCdcStreamRegistration( - &tr, - placedName, - { KeyRange(KeyRangeRef(placedKey, keyAfter(placedKey))) }, - cx->clientInfo->get().cdcProxies.front().id()); - ASSERT(result.requiresCommit); - attemptedIds.insert(result.streamId); - co_await tr.commit(); - co_return result.streamId; - } catch (Error& e) { - error = e; - } - co_await tr.onError(error); - } - } - - Future run(Database cx) { - const KeyRange coldRange(KeyRangeRef(coldKey, keyAfter(coldKey))); - const KeyRange hotRange(KeyRangeRef(hotKey, keyAfter(hotKey))); - const CDCStreamId cold = co_await registerNativeCdcStreamClient(cx, coldName, { coldRange }); - ASSERT_EQ(cx->clientInfo->get().nativeCdcTagCount, 2); - const CDCStreamId hot = co_await registerNativeCdcStreamClient(cx, hotName, { hotRange }); - const CDCStreamId duplicate = co_await registerNativeCdcStreamClient(cx, duplicateName, { coldRange }); - const Tag coldTag = co_await readTag(cx, cold); - const Tag hotTag = co_await readTag(cx, hot); - ASSERT_NE(coldTag, hotTag); - ASSERT_EQ(co_await readTag(cx, duplicate), coldTag); - Future producer = produceHotWrites(cx, hotKey); - const CDCStreamId placed = co_await timeoutError(registerWithLoad(cx, coldTag, hotTag), operationTimeout); - producer.cancel(); - ASSERT_EQ(co_await readTag(cx, placed), coldTag); - // Existing streams retain their original tag; placement requires no history cutover protocol. - ASSERT_EQ(co_await readTag(cx, cold), coldTag); - ASSERT_EQ(co_await readTag(cx, duplicate), coldTag); - ASSERT_EQ(co_await readTag(cx, hot), hotTag); - Reference consumer = co_await createNativeCdcConsumer(cx, placedName); - const Value marker = "placed-stream-delivery"_sr; - const Version committed = co_await writeValue(cx, placedKey, marker); - bool found = false; - while (!found) { - const CDCConsumeReply reply = co_await timeoutError(consumer->consume(), operationTimeout); - for (const auto& versioned : reply.mutations) { - if (versioned.version == committed) { - ASSERT_EQ(versioned.mutations.size(), 1); - ASSERT_EQ(versioned.mutations.front().type, MutationRef::SetValue); - ASSERT_EQ(versioned.mutations.front().param1, placedKey); - ASSERT_EQ(versioned.mutations.front().param2, marker); - found = true; - } - } - } - co_await consumer->acknowledge(); - for (const auto& name : { placedName, duplicateName, hotName, coldName }) { - co_await removeNativeCdcStreamClient(cx, name); - } - CODE_PROBE(true, "Native CDC initial placement favors measured load over stream count"); - TraceEvent("NativeCdcInitialPlacementVerified").detail("PlacedStream", placed).detail("Tag", coldTag); - } - -public: - static constexpr auto NAME = "NativeCdcInitialPlacement"; - explicit NativeCdcInitialPlacementWorkload(WorkloadContext const& wc) - : TestWorkload(wc), operationTimeout(getOption(options, "operationTimeout"_sr, 120.0)) {} - - void disableFailureInjectionWorkloads(std::set& out) const override { out.insert("RandomRangeLock"); } - Future setup(Database const& cx) override { return Void(); } - Future start(Database const& cx) override { - return clientId == 0 ? timeoutError(run(cx), operationTimeout * 2) : Void(); - } - Future check(Database const& cx) override { return true; } - void getMetrics(std::vector& metrics) override {} -}; - -WorkloadFactory NativeCdcInitialPlacementWorkloadFactory; diff --git a/fdbserver/workloads/UnitTests.cpp b/fdbserver/workloads/UnitTests.cpp index 20429e3408f..968d7f11149 100644 --- a/fdbserver/workloads/UnitTests.cpp +++ b/fdbserver/workloads/UnitTests.cpp @@ -53,7 +53,7 @@ void forceLinkIPagerTests(); void forceLinkMockS3ServerTests(); void forceLinkAuditUtilsTests(); void forceLinkShardsAffectedByTeamFailureTests(); -void forceLinkNativeCdcBalancerTests(); +void forceLinkNativeCdcRetagCleanupTests(); void forceLinkNativeCdcMetadataTests(); void forceLinkClusterHealthMonitorTests(); void forceLinkGrvQueueDelayTests(); @@ -134,7 +134,7 @@ struct UnitTestWorkload : TestWorkload { forceLinkMockS3ServerTests(); forceLinkAuditUtilsTests(); forceLinkShardsAffectedByTeamFailureTests(); - forceLinkNativeCdcBalancerTests(); + forceLinkNativeCdcRetagCleanupTests(); forceLinkNativeCdcMetadataTests(); forceLinkClusterHealthMonitorTests(); forceLinkGrvQueueDelayTests(); diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 4931a016b02..f656855bf45 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -225,12 +225,13 @@ if(WITH_PYTHON) add_fdb_test(TEST_FILES fast/RandomUnitTests.toml) add_fdb_test(TEST_FILES fast/RangeLocking.toml) add_fdb_test(TEST_FILES fast/NativeCdcEndToEnd.toml) - add_fdb_test(TEST_FILES fast/NativeCdcInitialPlacement.toml) add_fdb_test(TEST_FILES fast/NativeCdcBuggify.toml) add_fdb_test(TEST_FILES fast/NativeCdcAssignmentPublication.toml) add_fdb_test(TEST_FILES fast/NativeCdcSharedTag.toml) add_fdb_test(TEST_FILES fast/NativeCdcRetiredSharedTagSnapshot.toml) add_fdb_test(TEST_FILES fast/NativeCdcMemoryBound.toml) + add_fdb_test(TEST_FILES fast/NativeCdcRetagCompatibility.toml) + add_fdb_test(TEST_FILES fast/NativeCdcRetaggingMemoryBound.toml) add_fdb_test(TEST_FILES fast/NativeCdcBufferContention.toml) add_fdb_test(TEST_FILES fast/NativeCdcReplyChunking.toml) add_fdb_test(TEST_FILES fast/NativeCdcMultipleRanges.toml) @@ -243,6 +244,9 @@ if(WITH_PYTHON) add_fdb_test( TEST_FILES fast/NativeCdcDisableRestart-1.toml fast/NativeCdcDisableRestart-2.toml) + add_fdb_test( + TEST_FILES fast/NativeCdcRetagDisableRestart-1.toml + fast/NativeCdcRetagDisableRestart-2.toml) add_fdb_test(TEST_FILES fast/RangeLockCycle.toml) add_fdb_test(TEST_FILES fast/ReadHotDetectionCorrectness.toml IGNORE) # TODO re-enable once read hot detection is enabled. add_fdb_test(TEST_FILES fast/ReportConflictingKeys.toml) diff --git a/tests/fast/NativeCdcInitialPlacement.toml b/tests/fast/NativeCdcInitialPlacement.toml deleted file mode 100644 index 86e0a9c0c3c..00000000000 --- a/tests/fast/NativeCdcInitialPlacement.toml +++ /dev/null @@ -1,32 +0,0 @@ -[configuration] -config = 'single commit_proxies=1 grv_proxies=2' -singleRegion = true -datacenters = 1 -machineCount = 12 -statelessProcessClassesPerDC = 3 -buggify = false -faultInjection = false - -[[knobs]] -enable_native_cdc = true -native_cdc_tag_count = 2 -native_cdc_tag_balancing_enabled = true -native_cdc_tag_sample_interval = 0.5 -native_cdc_tag_sample_timeout = 5.0 -native_cdc_tag_sample_max_age = 30.0 -# Keep producer measurement deterministic within this focused test window. -storage_metrics_average_interval = 1.0 -storage_metrics_average_interval_per_kseconds = 1000.0 -bytes_written_units_per_sample = 1 - -[[test]] -testTitle = 'NativeCdcInitialPlacement' -useDB = true -runFailureWorkloads = false -waitForQuiescenceEnd = false -connectionFailuresDisableDuration = 1000000 -timeout = 300 - - [[test.workload]] - testName = 'NativeCdcInitialPlacement' - operationTimeout = 120.0 diff --git a/tests/fast/NativeCdcRetagCompatibility.toml b/tests/fast/NativeCdcRetagCompatibility.toml new file mode 100644 index 00000000000..54335c23a02 --- /dev/null +++ b/tests/fast/NativeCdcRetagCompatibility.toml @@ -0,0 +1,38 @@ +[configuration] +config = 'double commit_proxies=1 grv_proxies=1' +singleRegion = true +datacenters = 1 +machineCount = 16 +statelessProcessClassesPerDC = 4 +buggify = false +faultInjection = false + +[[knobs]] +enable_native_cdc = true +native_cdc_tag_count = 2 +native_cdc_retag_cleanup_interval = 0.25 +cdc_proxy_failure_coalesce_delay = 0.1 +cdc_proxy_consume_poll_timeout = 0.25 +cdc_proxy_buffer_bytes = 33554432 +maximum_peek_bytes = 65536 +cdc_proxy_consume_reply_bytes = 65536 + +[[test]] +testTitle = 'NativeCdcRetagCompatibility' +useDB = true +runFailureWorkloads = false +runConsistencyCheck = false +waitForQuiescenceEnd = false +connectionFailuresDisableDuration = 1000000 +timeout = 600 + + [[test.workload]] + testName = 'NativeCdcEndToEnd' + initialStreamCount = 4 + minStreamCount = 4 + keyCount = 4 + writesPerRound = 1 + testRetagCompatibility = true + testRetagTransactionRetries = true + memoryTestValueBytes = 512 + operationTimeout = 180.0 diff --git a/tests/fast/NativeCdcRetagDisableRestart-1.toml b/tests/fast/NativeCdcRetagDisableRestart-1.toml new file mode 100644 index 00000000000..d06cb1c1890 --- /dev/null +++ b/tests/fast/NativeCdcRetagDisableRestart-1.toml @@ -0,0 +1,33 @@ +[configuration] +config = 'single' +singleRegion = true +storageEngineExcludeTypes = [3, 4, 5] +buggify = false +faultInjection = false + +[[knobs]] +enable_native_cdc = true +native_cdc_tag_count = 2 +native_cdc_retag_cleanup_interval = 0.25 + +[[test]] +testTitle = 'NativeCdcRetagDisableRestart' +clearAfterTest = false +runFailureWorkloads = false +runConsistencyCheck = false +waitForQuiescenceEnd = false + + [[test.workload]] + testName = 'NativeCdcEndToEnd' + initialStreamCount = 1 + minStreamCount = 1 + keyCount = 4 + writesPerRound = 1 + prepareRestartDrain = true + testRetaggedRestart = true + testRetagTransactionRetries = true + + [[test.workload]] + testName = 'SaveAndKill' + restartInfoLocation = 'simfdb/restartInfo.ini' + testDuration = 20.0 diff --git a/tests/fast/NativeCdcRetagDisableRestart-2.toml b/tests/fast/NativeCdcRetagDisableRestart-2.toml new file mode 100644 index 00000000000..47ce7ab3329 --- /dev/null +++ b/tests/fast/NativeCdcRetagDisableRestart-2.toml @@ -0,0 +1,28 @@ +[configuration] +storageEngineExcludeTypes = [4, 5] +buggify = false +faultInjection = false + +[[knobs]] +enable_native_cdc = false +native_cdc_tag_count = 2 +native_cdc_retag_cleanup_interval = 0.25 + +[[test]] +testTitle = 'NativeCdcRetagDisableRestart' +runSetup = false +runFailureWorkloads = false +runConsistencyCheck = false +waitForQuiescenceBegin = false +waitForQuiescenceEnd = false +connectionFailuresDisableDuration = 1000000 + + [[test.workload]] + testName = 'NativeCdcEndToEnd' + initialStreamCount = 1 + minStreamCount = 1 + keyCount = 4 + writesPerRound = 1 + drainAfterRestart = true + testRetaggedRestart = true + operationTimeout = 180.0 diff --git a/tests/fast/NativeCdcRetaggingMemoryBound.toml b/tests/fast/NativeCdcRetaggingMemoryBound.toml new file mode 100644 index 00000000000..1740a8a1df3 --- /dev/null +++ b/tests/fast/NativeCdcRetaggingMemoryBound.toml @@ -0,0 +1,38 @@ +[configuration] +config = 'single logs=1 commit_proxies=1 grv_proxies=1' +singleRegion = true +datacenters = 1 +machineCount = 10 +statelessProcessClassesPerDC = 4 +buggify = false +faultInjection = false + +[[knobs]] +enable_native_cdc = true +native_cdc_tag_count = 2 +native_cdc_retag_cleanup_interval = 0.25 +cdc_proxy_buffer_bytes = 4608 +maximum_peek_bytes = 1152 +cdc_proxy_consume_reply_bytes = 4096 +# Progress has a much shorter deadline, so expiry cannot release a contended reservation. +cdc_proxy_consume_poll_timeout = 600.0 +cdc_proxy_pop_scan_interval = 600.0 + +[[test]] +testTitle = 'NativeCdcRetaggingMemoryBound' +useDB = true +runFailureWorkloads = false +runConsistencyCheck = false +waitForQuiescenceEnd = false +connectionFailuresDisableDuration = 1000000 +timeout = 600 + + [[test.workload]] + testName = 'NativeCdcEndToEnd' + initialStreamCount = 2 + minStreamCount = 2 + keyCount = 2 + writesPerRound = 1 + testRetaggingMemoryBound = true + retentionValidationDelay = 1.0 + operationTimeout = 20.0 From 66e56f488c489dc56d0a32c9cb7dbd03412b6b6e Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 20 Sep 2026 14:08:54 -0700 Subject: [PATCH 134/170] Release partial CDC buffer reservations before retrying --- design/cdc.md | 5 +- fdbserver/cdcproxy/CDCProxy.cpp | 33 ++--- .../include/fdbserver/cdcproxy/CDCProxyTest.h | 75 ++++++++++ fdbserver/workloads/CMakeLists.txt | 1 + fdbserver/workloads/NativeCdcEndToEnd.cpp | 128 ++++++++++++++++++ tests/CMakeLists.txt | 1 + tests/fast/NativeCdcBufferContention.toml | 38 ++++++ 7 files changed, 264 insertions(+), 17 deletions(-) create mode 100644 fdbserver/cdcproxy/include/fdbserver/cdcproxy/CDCProxyTest.h create mode 100644 tests/fast/NativeCdcBufferContention.toml diff --git a/design/cdc.md b/design/cdc.md index b98f3f8663d..d5c98f2aab8 100644 --- a/design/cdc.md +++ b/design/cdc.md @@ -533,7 +533,10 @@ and cap each reply at the smaller of `MAXIMUM_PEEK_BYTES` and budget plus one reply-sized materialization window before issuing a peek. It marks these delivery cursors with the same per-reply limit; recovery cursors remain uncapped so that transaction-system replay is not constrained -by a delivery memory knob. The pass retains the aggregate raw reservation +by a delivery memory knob. If materialization needs a larger window, the reader +releases its cursor and reservation before retrying with the full required +reservation. Competing readers cannot hold partial reservations while waiting +for each other to release capacity. The pass retains the aggregate raw reservation while filtering and copying, then releases it and transfers only accepted filtered bytes to the stream buffers. Acknowledgement or stream removal releases those retained permits. The usable retained-batch capacity is diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index 92cf4ec3531..0e905b0b1a7 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -32,6 +32,7 @@ #include "fdbclient/SystemData.h" #include "NativeCdcInternal.h" #include "fdbserver/cdcproxy/CDCProxy.h" +#include "fdbserver/cdcproxy/CDCProxyTest.h" #include "fdbserver/core/Knobs.h" #include "fdbserver/core/LogProtocolMessage.h" #include "fdbserver/core/OTELSpanContextMessage.h" @@ -178,6 +179,7 @@ struct CDCBufferedBatch { struct CDCBufferedTag : ReferenceCounted { Tag tag; bool active = true; + int64_t nextPassReservation = 0; std::set streamIds; AsyncTrigger refresh; AsyncTrigger stopped; @@ -1305,22 +1307,17 @@ Future CDCProxy::materializeBufferSelection(ReferenceonChange(), - tag->stopped.onTrigger(), - tag->refresh.onTrigger()); - if (exactCapacity.index() == 1 || exactCapacity.index() == 3) { - co_return CDCBufferTagPassResult::RETRY; - } - if (exactCapacity.index() == 2) { - co_return CDCBufferTagPassResult::STOP; + if (auto test = CDCProxyMaterializationTest::get()) { + // Poll shared simulation state without running callbacks in another simulated process. + while (test->holdExpansion(id, tag->tag, reservation.remaining)) { + co_await delay(0.01); + } } - reservation.remaining += additionalBytes; - recordBufferUsage(); + // Two readers can exhaust the budget with initial reservations and then both wait for an expansion. + // Drop this cursor and reservation before reacquiring the full amount in one request. + tag->nextPassReservation = rawPeekReservation + selection.selectedBytes; + ASSERT_LE(tag->nextPassReservation, bufferLimit); + co_return CDCBufferTagPassResult::RETRY; } if (!tag->active) { co_return CDCBufferTagPassResult::STOP; @@ -1358,6 +1355,7 @@ Future CDCProxy::materializeBufferSelection(ReferencenextPassReservation = 0; advanceTagBufferedThrough(tag, throughVersion, selection.selectedStreamIds); // Every raw cursor arena is covered by rawPeekReservation only for this pass. Reopen from the shared minimum // after releasing it so no cursor response remains live outside the proxy memory budget. @@ -1413,7 +1411,7 @@ Future CDCProxy::bufferTagCursor(ReferencenextPassReservation); if (prefetch && (bufferLock.waiters() != 0 || bufferLock.available() < passReservation)) { co_return CDCBufferTagPassResult::RETRY; } @@ -1939,6 +1937,9 @@ Future CDCProxy::consumeReply(Reference stre auto buffered = co_await race(waitForBufferedVersion(stream, begin), delay(SERVER_KNOBS->CDC_PROXY_CONSUME_POLL_TIMEOUT)); if (buffered.index() == 1) { + if (auto test = CDCProxyMaterializationTest::get()) { + test->recordLeaseExpiry(id); + } CODE_PROBE(true, "CDC proxy expires an idle consume lease"); CDCConsumeReply reply; reply.lastConsumedVersion = cursor.lastConsumedVersion; diff --git a/fdbserver/cdcproxy/include/fdbserver/cdcproxy/CDCProxyTest.h b/fdbserver/cdcproxy/include/fdbserver/cdcproxy/CDCProxyTest.h new file mode 100644 index 00000000000..1fc55f9d6f0 --- /dev/null +++ b/fdbserver/cdcproxy/include/fdbserver/cdcproxy/CDCProxyTest.h @@ -0,0 +1,75 @@ +/* + * CDCProxyTest.h + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#pragma once + +#include +#include "fdbclient/FDBTypes.h" +#include "flow/flow.h" + +// Simulation-only barrier at the point where two real tag readers need to expand their reservations. +class CDCProxyMaterializationTest : public ReferenceCounted { + UID owner; + std::map reservations; + bool released = false; + int expiredLeases = 0; + inline static Reference installed; + +public: + CDCProxyMaterializationTest(UID owner, Tag first, Tag second) : owner(owner) { + ASSERT_NE(first, second); + reservations.emplace(first, 0); + reservations.emplace(second, 0); + } + + static Reference get() { + return g_network->isSimulated() ? installed : Reference(); + } + static void install(Reference test) { + ASSERT(g_network->isSimulated()); + ASSERT(!installed); + installed = test; + } + static void uninstall() { + if (installed) { + installed->released = true; + installed.clear(); + } + } + bool holdExpansion(UID proxy, Tag tag, int64_t reserved) { + if (proxy != owner || released || !reservations.contains(tag)) { + return false; + } + reservations.at(tag) = reserved; + return true; + } + bool bothReadersHeld() const { return reservations.begin()->second > 0 && reservations.rbegin()->second > 0; } + int64_t heldBytes() const { return reservations.begin()->second + reservations.rbegin()->second; } + void release() { + ASSERT(bothReadersHeld()); + released = true; + } + void recordLeaseExpiry(UID proxy) { + if (proxy == owner) { + ++expiredLeases; + } + } + int leaseExpiries() const { return expiredLeases; } +}; diff --git a/fdbserver/workloads/CMakeLists.txt b/fdbserver/workloads/CMakeLists.txt index aadaf00b858..f88415501a2 100644 --- a/fdbserver/workloads/CMakeLists.txt +++ b/fdbserver/workloads/CMakeLists.txt @@ -12,6 +12,7 @@ configure_fdbserver_common_includes(fdbserver_workloads) target_include_directories(fdbserver_workloads PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}) target_link_libraries(fdbserver_workloads PRIVATE + fdbserver_cdcproxy fdbserver_consistencyscan fdbserver_core fdbserver_checkpoint diff --git a/fdbserver/workloads/NativeCdcEndToEnd.cpp b/fdbserver/workloads/NativeCdcEndToEnd.cpp index 8ada52bcf78..d78faed7f4f 100644 --- a/fdbserver/workloads/NativeCdcEndToEnd.cpp +++ b/fdbserver/workloads/NativeCdcEndToEnd.cpp @@ -30,11 +30,14 @@ #include "fdbclient/DatabaseContext.h" #include "fdbclient/NativeCdc.h" #include "fdbclient/SystemData.h" +#include "fdbserver/cdcproxy/CDCProxyTest.h" +#include "fdbserver/core/Knobs.h" #include "fdbserver/core/RecoveryState.h" #include "fdbserver/core/ServerDBInfo.h" #include "fdbserver/tester/workloads.h" #include "fdbrpc/simulator.h" #include "flow/DeterministicRandom.h" +#include "flow/ScopeExit.h" // Exercises native CDC by registering overlapping streams, writing mutations, consuming and acknowledging them, // and checking delivery, retention, assignment publication, failure recovery, and drain behavior. Test options @@ -72,6 +75,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { bool testTagOwnership; bool injectUndeliveredProxyHalt; bool testMemoryBound; + bool testBufferContention; bool testReplyChunking; bool testMultipleRanges; bool testOversizedPeek; @@ -247,6 +251,120 @@ class NativeCdcEndToEndWorkload : public TestWorkload { } } + Future initializeBufferContentionStreams(Database cx) { + for (int i = 0; i < 2; ++i) { + co_await addStream(cx, KeyRange(KeyRangeRef(keyForIndex(i), keyForIndex(i + 1)))); + } + } + + Future consumeContendedBuffer(int index, + Version through, + Reference> delivered, + Future releaseAcknowledgements) { + auto& stream = streams[index]; + while (stream.consumer->position().lastConsumedVersion < through) { + const Version previous = stream.consumer->position().lastConsumedVersion; + const CDCConsumeReply reply = co_await stream.consumer->consume(); + for (const auto& versioned : reply.mutations) { + ASSERT_GT(versioned.version, previous); + ASSERT_LE(versioned.version, reply.lastConsumedVersion); + ASSERT_EQ(versioned.mutations.size(), stream.expected.size()); + for (const auto& mutation : versioned.mutations) { + ASSERT_EQ(mutation.type, MutationRef::SetValue); + auto found = stream.expected.find(std::make_pair(Key(mutation.param1), Value(mutation.param2))); + ASSERT(found != stream.expected.end()); + ASSERT_LE(versioned.version, found->second.committedVersion); + ASSERT(found->second.observedVersions.insert(versioned.version).second); + } + } + } + for (const auto& [value, expected] : stream.expected) { + ASSERT(expected.observedVersions.contains(expected.committedVersion)); + } + delivered->set(delivered->get() + 1); + co_await releaseAcknowledgements; + co_await stream.consumer->acknowledge(); + } + + Future validateBufferContention(Database cx) { + const NativeCdcStatus metadata = co_await timeoutError(getNativeCdcStatus(cx), operationTimeout); + ASSERT(metadata.metadataComplete); + ASSERT_EQ(metadata.streams.size(), 2); + const auto& first = metadata.streams[0]; + const auto& second = metadata.streams[1]; + ASSERT(first.owner.present()); + ASSERT_EQ(first.owner, second.owner); + ASSERT_EQ(first.tags.size(), 1); + ASSERT_EQ(second.tags.size(), 1); + ASSERT_NE(first.tags.front(), second.tags.front()); + const UID owner = first.owner.get(); + std::vector committed; + for (int index = 0; index < 2; ++index) { + std::vector> values; + // These small mutations fit one raw reply but need more space after materialization. + for (int i = 0; i < 12; ++i) { + const Key key = streams[index].keys.begin.withSuffix(StringRef(format("/%02d", i))); + values.emplace_back(key, Value(StringRef(std::string(32, 'x')))); + } + committed.push_back(co_await writeValues(cx, values)); + recordExpectedWrites(values, committed.back()); + } + auto barrier = makeReference(owner, first.tags.front(), second.tags.front()); + CDCProxyMaterializationTest::install(barrier); + ScopeExit removeBarrier([] { CDCProxyMaterializationTest::uninstall(); }); + Promise releaseAcknowledgements; + auto delivered = makeReference>(0); + std::vector> consumers{ + consumeContendedBuffer(0, committed[0], delivered, releaseAcknowledgements.getFuture()), + consumeContendedBuffer(1, committed[1], delivered, releaseAcknowledgements.getFuture()) + }; + const double deadline = now() + operationTimeout; + while (!barrier->bothReadersHeld()) { + for (const auto& consumer : consumers) { + if (consumer.isReady()) { + consumer.get(); + ASSERT(false); + } + } + ASSERT_LT(now(), deadline); + co_await delay(0.01); + } + auto status = co_await timeoutError(getAssignedProxyStatus(cx, streams.front().consumer->position().streamId), + operationTimeout); + ASSERT_EQ(status.first.id(), owner); + ASSERT_EQ(status.second.activePermits, barrier->heldBytes()); + ASSERT_EQ(status.second.activePermits, status.second.bufferLimit); + ASSERT_EQ(status.second.bufferedBytes, 0); + ASSERT_EQ(delivered->get(), 0); + TraceEvent("NativeCdcBufferContendedReaders") + .detail("ActivePermits", status.second.activePermits) + .detail("BufferLimit", status.second.bufferLimit); + const double releasedAt = now(); + barrier->release(); + while (delivered->get() < 2) { + co_await timeoutError(delivered->onChange(), std::max(0.0, releasedAt + operationTimeout - now())); + } + ASSERT_EQ(barrier->leaseExpiries(), 0); + ASSERT_LT(now() - releasedAt, SERVER_KNOBS->CDC_PROXY_CONSUME_POLL_TIMEOUT); + status = co_await timeoutError(getAssignedProxyStatus(cx, streams.front().consumer->position().streamId), + operationTimeout); + ASSERT_EQ(status.first.id(), owner); + ASSERT_GT(status.second.bufferedBytes, 0); + ASSERT_LE(status.second.bufferedBytes, status.second.activePermits); + ASSERT_LE(status.second.activePermits, status.second.bufferLimit); + ASSERT_LE(status.second.peakActivePermits, status.second.bufferLimit); + releaseAcknowledgements.send(Void()); + co_await timeoutError(waitForAll(consumers), operationTimeout); + ASSERT_EQ(barrier->leaseExpiries(), 0); + CODE_PROBE(true, + "Native CDC expanded reservations progress on two tags without acknowledgements or lease expiry"); + for (const auto& stream : streams) { + co_await removeNativeCdcStreamClient(cx, stream.name); + } + streams.clear(); + co_await timeoutError(waitForRetiredTagCleanup(cx), operationTimeout); + } + Future initializeOversizedPeekStreams(Database cx) { ASSERT_GE(keyCount, 4); co_await addStream(cx, KeyRange(KeyRangeRef(keyForIndex(0), keyForIndex(2)))); @@ -2102,6 +2220,10 @@ class NativeCdcEndToEndWorkload : public TestWorkload { } Future run(Database cx) { + if (testBufferContention) { + co_await validateBufferContention(cx); + co_return; + } if (testMultipleRanges) { co_await validateMultipleRanges(cx); co_return; @@ -2204,6 +2326,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { testTagOwnership = getOption(options, "testTagOwnership"_sr, false); injectUndeliveredProxyHalt = getOption(options, "injectUndeliveredProxyHalt"_sr, false); testMemoryBound = getOption(options, "testMemoryBound"_sr, false); + testBufferContention = getOption(options, "testBufferContention"_sr, false); testReplyChunking = getOption(options, "testReplyChunking"_sr, false); testMultipleRanges = getOption(options, "testMultipleRanges"_sr, false); testOversizedPeek = getOption(options, "testOversizedPeek"_sr, false); @@ -2234,6 +2357,8 @@ class NativeCdcEndToEndWorkload : public TestWorkload { ASSERT(!(testOversizedPeek && testDurableAckScan)); ASSERT(!(testRetiredSharedTagSnapshot && testRetiredRecovery)); ASSERT(!blockRetiredPopWithLiveStream || testRetiredRecovery); + ASSERT(!testBufferContention || (initialStreamCount == 2 && !prepareRestartDrain && !drainAfterRestart && + !testMultipleRanges && !testRetiredSharedTagSnapshot && !testMemoryBound)); } // RandomRangeLock can outlive this bounded CDC workload and mask its progress check. @@ -2255,6 +2380,9 @@ class NativeCdcEndToEndWorkload : public TestWorkload { if (testRetiredSharedTagSnapshot) { return Void(); } + if (testBufferContention) { + return initializeBufferContentionStreams(cx); + } if (testOversizedPeek) { return initializeOversizedPeekStreams(cx); } diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 17a46f04af2..91a19bb42ed 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -230,6 +230,7 @@ if(WITH_PYTHON) add_fdb_test(TEST_FILES fast/NativeCdcSharedTag.toml) add_fdb_test(TEST_FILES fast/NativeCdcRetiredSharedTagSnapshot.toml) add_fdb_test(TEST_FILES fast/NativeCdcMemoryBound.toml) + add_fdb_test(TEST_FILES fast/NativeCdcBufferContention.toml) add_fdb_test(TEST_FILES fast/NativeCdcReplyChunking.toml) add_fdb_test(TEST_FILES fast/NativeCdcMultipleRanges.toml) add_fdb_test(TEST_FILES fast/NativeCdcLargeVersionBatching.toml) diff --git a/tests/fast/NativeCdcBufferContention.toml b/tests/fast/NativeCdcBufferContention.toml new file mode 100644 index 00000000000..dfea7f93ef2 --- /dev/null +++ b/tests/fast/NativeCdcBufferContention.toml @@ -0,0 +1,38 @@ +[configuration] +config = 'single logs=1 commit_proxies=1 grv_proxies=1' +singleRegion = true +datacenters = 1 +machineCount = 10 +statelessProcessClassesPerDC = 4 +buggify = false +faultInjection = false + +[[knobs]] +enable_native_cdc = true +native_cdc_tag_count = 2 +cdc_proxy_buffer_bytes = 4608 +maximum_peek_bytes = 1152 +cdc_proxy_consume_reply_bytes = 4096 +# Progress must precede any lease expiry or periodic pop that could free capacity. +cdc_proxy_consume_poll_timeout = 600.0 +cdc_proxy_pop_scan_interval = 600.0 + +[[test]] +testTitle = 'NativeCdcBufferContention' +useDB = true +runFailureWorkloads = false +runConsistencyCheck = false +waitForQuiescenceEnd = false +connectionFailuresDisableDuration = 1000000 +timeout = 600 + + [[test.workload]] + testName = 'NativeCdcEndToEnd' + initialStreamCount = 2 + minStreamCount = 2 + maxStreamCount = 2 + keyCount = 2 + writesPerRound = 1 + rounds = 0 + testBufferContention = true + operationTimeout = 20.0 From 9052b055095316bcf90d01dcd21a3c6092662e04 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Sun, 20 Sep 2026 20:10:45 -0700 Subject: [PATCH 135/170] Replace CDC materialization barrier with direct reservation regression --- fdbserver/cdcproxy/CDCProxy.cpp | 25 ++-- .../include/fdbserver/cdcproxy/CDCProxyTest.h | 75 ---------- fdbserver/workloads/CMakeLists.txt | 1 - fdbserver/workloads/NativeCdcEndToEnd.cpp | 128 ------------------ tests/CMakeLists.txt | 1 - tests/fast/NativeCdcBufferContention.toml | 38 ------ 6 files changed, 11 insertions(+), 257 deletions(-) delete mode 100644 fdbserver/cdcproxy/include/fdbserver/cdcproxy/CDCProxyTest.h delete mode 100644 tests/fast/NativeCdcBufferContention.toml diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index 0e905b0b1a7..8c70d1ae2f5 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -32,7 +32,6 @@ #include "fdbclient/SystemData.h" #include "NativeCdcInternal.h" #include "fdbserver/cdcproxy/CDCProxy.h" -#include "fdbserver/cdcproxy/CDCProxyTest.h" #include "fdbserver/core/Knobs.h" #include "fdbserver/core/LogProtocolMessage.h" #include "fdbserver/core/OTELSpanContextMessage.h" @@ -1307,12 +1306,6 @@ Future CDCProxy::materializeBufferSelection(ReferenceholdExpansion(id, tag->tag, reservation.remaining)) { - co_await delay(0.01); - } - } // Two readers can exhaust the budget with initial reservations and then both wait for an expansion. // Drop this cursor and reservation before reacquiring the full amount in one request. tag->nextPassReservation = rawPeekReservation + selection.selectedBytes; @@ -1937,9 +1930,6 @@ Future CDCProxy::consumeReply(Reference stre auto buffered = co_await race(waitForBufferedVersion(stream, begin), delay(SERVER_KNOBS->CDC_PROXY_CONSUME_POLL_TIMEOUT)); if (buffered.index() == 1) { - if (auto test = CDCProxyMaterializationTest::get()) { - test->recordLeaseExpiry(id); - } CODE_PROBE(true, "CDC proxy expires an idle consume lease"); CDCConsumeReply reply; reply.lastConsumedVersion = cursor.lastConsumedVersion; @@ -2721,7 +2711,7 @@ class CDCProxyPrefetchTest { co_return; } - static Future extraCapacity() { + static Future extraCapacity(Prefetch prefetch) { CDCProxyPrefetchTest test; auto stream = test.addStream(1); const int64_t limit = SERVER_KNOBS->CDC_PROXY_BUFFER_BYTES; @@ -2741,8 +2731,12 @@ class CDCProxyPrefetchTest { selection.selectedStreamIds.insert(1); selection.selectedBytes = passLimit.preferredBufferedBytes + 1; auto cursor = makeReference(Void()); - co_await test.proxy.materializeBufferSelection( - test.tag, cursor, 100, selection, passLimit.rawReplyBytes, reservation, limit, Prefetch::True, Never()); + auto work = test.proxy.materializeBufferSelection( + test.tag, cursor, 100, selection, passLimit.rawReplyBytes, reservation, limit, prefetch, Never()); + // Waiting for an incremental reservation would deadlock against the other reader's held capacity. + ASSERT(work.isReady()); + ASSERT(work.get() == CDCBufferTagPassResult::RETRY); + ASSERT_EQ(test.tag->nextPassReservation, passLimit.rawReplyBytes + selection.selectedBytes); ASSERT_EQ(test.proxy.bufferLock.waiters(), 0); ASSERT_EQ(test.proxy.bufferLock.activePermits(), limit); ASSERT(stream->mutations.empty()); @@ -2873,7 +2867,10 @@ TEST_CASE("/NativeCDC/PrefetchDeclinesQueuedCapacity") { return CDCProxyPrefetchTest::capacity(true); } TEST_CASE("/NativeCDC/PrefetchDeclinesExtraCapacity") { - return CDCProxyPrefetchTest::extraCapacity(); + return CDCProxyPrefetchTest::extraCapacity(Prefetch::True); +} +TEST_CASE("/NativeCDC/DemandRetriesExpandedReservation") { + return CDCProxyPrefetchTest::extraCapacity(Prefetch::False); } TEST_CASE("/NativeCDC/PrefetchRefreshCancels") { return CDCProxyPrefetchTest::interrupted(0); diff --git a/fdbserver/cdcproxy/include/fdbserver/cdcproxy/CDCProxyTest.h b/fdbserver/cdcproxy/include/fdbserver/cdcproxy/CDCProxyTest.h deleted file mode 100644 index 1fc55f9d6f0..00000000000 --- a/fdbserver/cdcproxy/include/fdbserver/cdcproxy/CDCProxyTest.h +++ /dev/null @@ -1,75 +0,0 @@ -/* - * CDCProxyTest.h - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -#pragma once - -#include -#include "fdbclient/FDBTypes.h" -#include "flow/flow.h" - -// Simulation-only barrier at the point where two real tag readers need to expand their reservations. -class CDCProxyMaterializationTest : public ReferenceCounted { - UID owner; - std::map reservations; - bool released = false; - int expiredLeases = 0; - inline static Reference installed; - -public: - CDCProxyMaterializationTest(UID owner, Tag first, Tag second) : owner(owner) { - ASSERT_NE(first, second); - reservations.emplace(first, 0); - reservations.emplace(second, 0); - } - - static Reference get() { - return g_network->isSimulated() ? installed : Reference(); - } - static void install(Reference test) { - ASSERT(g_network->isSimulated()); - ASSERT(!installed); - installed = test; - } - static void uninstall() { - if (installed) { - installed->released = true; - installed.clear(); - } - } - bool holdExpansion(UID proxy, Tag tag, int64_t reserved) { - if (proxy != owner || released || !reservations.contains(tag)) { - return false; - } - reservations.at(tag) = reserved; - return true; - } - bool bothReadersHeld() const { return reservations.begin()->second > 0 && reservations.rbegin()->second > 0; } - int64_t heldBytes() const { return reservations.begin()->second + reservations.rbegin()->second; } - void release() { - ASSERT(bothReadersHeld()); - released = true; - } - void recordLeaseExpiry(UID proxy) { - if (proxy == owner) { - ++expiredLeases; - } - } - int leaseExpiries() const { return expiredLeases; } -}; diff --git a/fdbserver/workloads/CMakeLists.txt b/fdbserver/workloads/CMakeLists.txt index f88415501a2..aadaf00b858 100644 --- a/fdbserver/workloads/CMakeLists.txt +++ b/fdbserver/workloads/CMakeLists.txt @@ -12,7 +12,6 @@ configure_fdbserver_common_includes(fdbserver_workloads) target_include_directories(fdbserver_workloads PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}) target_link_libraries(fdbserver_workloads PRIVATE - fdbserver_cdcproxy fdbserver_consistencyscan fdbserver_core fdbserver_checkpoint diff --git a/fdbserver/workloads/NativeCdcEndToEnd.cpp b/fdbserver/workloads/NativeCdcEndToEnd.cpp index d78faed7f4f..8ada52bcf78 100644 --- a/fdbserver/workloads/NativeCdcEndToEnd.cpp +++ b/fdbserver/workloads/NativeCdcEndToEnd.cpp @@ -30,14 +30,11 @@ #include "fdbclient/DatabaseContext.h" #include "fdbclient/NativeCdc.h" #include "fdbclient/SystemData.h" -#include "fdbserver/cdcproxy/CDCProxyTest.h" -#include "fdbserver/core/Knobs.h" #include "fdbserver/core/RecoveryState.h" #include "fdbserver/core/ServerDBInfo.h" #include "fdbserver/tester/workloads.h" #include "fdbrpc/simulator.h" #include "flow/DeterministicRandom.h" -#include "flow/ScopeExit.h" // Exercises native CDC by registering overlapping streams, writing mutations, consuming and acknowledging them, // and checking delivery, retention, assignment publication, failure recovery, and drain behavior. Test options @@ -75,7 +72,6 @@ class NativeCdcEndToEndWorkload : public TestWorkload { bool testTagOwnership; bool injectUndeliveredProxyHalt; bool testMemoryBound; - bool testBufferContention; bool testReplyChunking; bool testMultipleRanges; bool testOversizedPeek; @@ -251,120 +247,6 @@ class NativeCdcEndToEndWorkload : public TestWorkload { } } - Future initializeBufferContentionStreams(Database cx) { - for (int i = 0; i < 2; ++i) { - co_await addStream(cx, KeyRange(KeyRangeRef(keyForIndex(i), keyForIndex(i + 1)))); - } - } - - Future consumeContendedBuffer(int index, - Version through, - Reference> delivered, - Future releaseAcknowledgements) { - auto& stream = streams[index]; - while (stream.consumer->position().lastConsumedVersion < through) { - const Version previous = stream.consumer->position().lastConsumedVersion; - const CDCConsumeReply reply = co_await stream.consumer->consume(); - for (const auto& versioned : reply.mutations) { - ASSERT_GT(versioned.version, previous); - ASSERT_LE(versioned.version, reply.lastConsumedVersion); - ASSERT_EQ(versioned.mutations.size(), stream.expected.size()); - for (const auto& mutation : versioned.mutations) { - ASSERT_EQ(mutation.type, MutationRef::SetValue); - auto found = stream.expected.find(std::make_pair(Key(mutation.param1), Value(mutation.param2))); - ASSERT(found != stream.expected.end()); - ASSERT_LE(versioned.version, found->second.committedVersion); - ASSERT(found->second.observedVersions.insert(versioned.version).second); - } - } - } - for (const auto& [value, expected] : stream.expected) { - ASSERT(expected.observedVersions.contains(expected.committedVersion)); - } - delivered->set(delivered->get() + 1); - co_await releaseAcknowledgements; - co_await stream.consumer->acknowledge(); - } - - Future validateBufferContention(Database cx) { - const NativeCdcStatus metadata = co_await timeoutError(getNativeCdcStatus(cx), operationTimeout); - ASSERT(metadata.metadataComplete); - ASSERT_EQ(metadata.streams.size(), 2); - const auto& first = metadata.streams[0]; - const auto& second = metadata.streams[1]; - ASSERT(first.owner.present()); - ASSERT_EQ(first.owner, second.owner); - ASSERT_EQ(first.tags.size(), 1); - ASSERT_EQ(second.tags.size(), 1); - ASSERT_NE(first.tags.front(), second.tags.front()); - const UID owner = first.owner.get(); - std::vector committed; - for (int index = 0; index < 2; ++index) { - std::vector> values; - // These small mutations fit one raw reply but need more space after materialization. - for (int i = 0; i < 12; ++i) { - const Key key = streams[index].keys.begin.withSuffix(StringRef(format("/%02d", i))); - values.emplace_back(key, Value(StringRef(std::string(32, 'x')))); - } - committed.push_back(co_await writeValues(cx, values)); - recordExpectedWrites(values, committed.back()); - } - auto barrier = makeReference(owner, first.tags.front(), second.tags.front()); - CDCProxyMaterializationTest::install(barrier); - ScopeExit removeBarrier([] { CDCProxyMaterializationTest::uninstall(); }); - Promise releaseAcknowledgements; - auto delivered = makeReference>(0); - std::vector> consumers{ - consumeContendedBuffer(0, committed[0], delivered, releaseAcknowledgements.getFuture()), - consumeContendedBuffer(1, committed[1], delivered, releaseAcknowledgements.getFuture()) - }; - const double deadline = now() + operationTimeout; - while (!barrier->bothReadersHeld()) { - for (const auto& consumer : consumers) { - if (consumer.isReady()) { - consumer.get(); - ASSERT(false); - } - } - ASSERT_LT(now(), deadline); - co_await delay(0.01); - } - auto status = co_await timeoutError(getAssignedProxyStatus(cx, streams.front().consumer->position().streamId), - operationTimeout); - ASSERT_EQ(status.first.id(), owner); - ASSERT_EQ(status.second.activePermits, barrier->heldBytes()); - ASSERT_EQ(status.second.activePermits, status.second.bufferLimit); - ASSERT_EQ(status.second.bufferedBytes, 0); - ASSERT_EQ(delivered->get(), 0); - TraceEvent("NativeCdcBufferContendedReaders") - .detail("ActivePermits", status.second.activePermits) - .detail("BufferLimit", status.second.bufferLimit); - const double releasedAt = now(); - barrier->release(); - while (delivered->get() < 2) { - co_await timeoutError(delivered->onChange(), std::max(0.0, releasedAt + operationTimeout - now())); - } - ASSERT_EQ(barrier->leaseExpiries(), 0); - ASSERT_LT(now() - releasedAt, SERVER_KNOBS->CDC_PROXY_CONSUME_POLL_TIMEOUT); - status = co_await timeoutError(getAssignedProxyStatus(cx, streams.front().consumer->position().streamId), - operationTimeout); - ASSERT_EQ(status.first.id(), owner); - ASSERT_GT(status.second.bufferedBytes, 0); - ASSERT_LE(status.second.bufferedBytes, status.second.activePermits); - ASSERT_LE(status.second.activePermits, status.second.bufferLimit); - ASSERT_LE(status.second.peakActivePermits, status.second.bufferLimit); - releaseAcknowledgements.send(Void()); - co_await timeoutError(waitForAll(consumers), operationTimeout); - ASSERT_EQ(barrier->leaseExpiries(), 0); - CODE_PROBE(true, - "Native CDC expanded reservations progress on two tags without acknowledgements or lease expiry"); - for (const auto& stream : streams) { - co_await removeNativeCdcStreamClient(cx, stream.name); - } - streams.clear(); - co_await timeoutError(waitForRetiredTagCleanup(cx), operationTimeout); - } - Future initializeOversizedPeekStreams(Database cx) { ASSERT_GE(keyCount, 4); co_await addStream(cx, KeyRange(KeyRangeRef(keyForIndex(0), keyForIndex(2)))); @@ -2220,10 +2102,6 @@ class NativeCdcEndToEndWorkload : public TestWorkload { } Future run(Database cx) { - if (testBufferContention) { - co_await validateBufferContention(cx); - co_return; - } if (testMultipleRanges) { co_await validateMultipleRanges(cx); co_return; @@ -2326,7 +2204,6 @@ class NativeCdcEndToEndWorkload : public TestWorkload { testTagOwnership = getOption(options, "testTagOwnership"_sr, false); injectUndeliveredProxyHalt = getOption(options, "injectUndeliveredProxyHalt"_sr, false); testMemoryBound = getOption(options, "testMemoryBound"_sr, false); - testBufferContention = getOption(options, "testBufferContention"_sr, false); testReplyChunking = getOption(options, "testReplyChunking"_sr, false); testMultipleRanges = getOption(options, "testMultipleRanges"_sr, false); testOversizedPeek = getOption(options, "testOversizedPeek"_sr, false); @@ -2357,8 +2234,6 @@ class NativeCdcEndToEndWorkload : public TestWorkload { ASSERT(!(testOversizedPeek && testDurableAckScan)); ASSERT(!(testRetiredSharedTagSnapshot && testRetiredRecovery)); ASSERT(!blockRetiredPopWithLiveStream || testRetiredRecovery); - ASSERT(!testBufferContention || (initialStreamCount == 2 && !prepareRestartDrain && !drainAfterRestart && - !testMultipleRanges && !testRetiredSharedTagSnapshot && !testMemoryBound)); } // RandomRangeLock can outlive this bounded CDC workload and mask its progress check. @@ -2380,9 +2255,6 @@ class NativeCdcEndToEndWorkload : public TestWorkload { if (testRetiredSharedTagSnapshot) { return Void(); } - if (testBufferContention) { - return initializeBufferContentionStreams(cx); - } if (testOversizedPeek) { return initializeOversizedPeekStreams(cx); } diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 91a19bb42ed..17a46f04af2 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -230,7 +230,6 @@ if(WITH_PYTHON) add_fdb_test(TEST_FILES fast/NativeCdcSharedTag.toml) add_fdb_test(TEST_FILES fast/NativeCdcRetiredSharedTagSnapshot.toml) add_fdb_test(TEST_FILES fast/NativeCdcMemoryBound.toml) - add_fdb_test(TEST_FILES fast/NativeCdcBufferContention.toml) add_fdb_test(TEST_FILES fast/NativeCdcReplyChunking.toml) add_fdb_test(TEST_FILES fast/NativeCdcMultipleRanges.toml) add_fdb_test(TEST_FILES fast/NativeCdcLargeVersionBatching.toml) diff --git a/tests/fast/NativeCdcBufferContention.toml b/tests/fast/NativeCdcBufferContention.toml deleted file mode 100644 index dfea7f93ef2..00000000000 --- a/tests/fast/NativeCdcBufferContention.toml +++ /dev/null @@ -1,38 +0,0 @@ -[configuration] -config = 'single logs=1 commit_proxies=1 grv_proxies=1' -singleRegion = true -datacenters = 1 -machineCount = 10 -statelessProcessClassesPerDC = 4 -buggify = false -faultInjection = false - -[[knobs]] -enable_native_cdc = true -native_cdc_tag_count = 2 -cdc_proxy_buffer_bytes = 4608 -maximum_peek_bytes = 1152 -cdc_proxy_consume_reply_bytes = 4096 -# Progress must precede any lease expiry or periodic pop that could free capacity. -cdc_proxy_consume_poll_timeout = 600.0 -cdc_proxy_pop_scan_interval = 600.0 - -[[test]] -testTitle = 'NativeCdcBufferContention' -useDB = true -runFailureWorkloads = false -runConsistencyCheck = false -waitForQuiescenceEnd = false -connectionFailuresDisableDuration = 1000000 -timeout = 600 - - [[test.workload]] - testName = 'NativeCdcEndToEnd' - initialStreamCount = 2 - minStreamCount = 2 - maxStreamCount = 2 - keyCount = 2 - writesPerRound = 1 - rounds = 0 - testBufferContention = true - operationTimeout = 20.0 From 3aa37282067abaded587419e5cc2a29cf6b9bd29 Mon Sep 17 00:00:00 2001 From: Jingyu Zhou Date: Mon, 21 Sep 2026 13:45:28 -0700 Subject: [PATCH 136/170] Fix CP admission control (#14090) * Fix CP admission control getKeyLocation_internal() constructs the request with a hardcoded limit of 100, which is exactly STORAGE_METRICS_SHARD_LIMIT. This means the admission control is never used. The problem is after a bounce, all clients will send this request to CP. So if the cluster is large enough and has a lot of traffic, key location request can saturate CP CPU, leading to a metastable failure state. * Format * Fix format --- fdbserver/commitproxy/CommitProxyServer.cpp | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/fdbserver/commitproxy/CommitProxyServer.cpp b/fdbserver/commitproxy/CommitProxyServer.cpp index 12779b2b510..8181459d0d8 100644 --- a/fdbserver/commitproxy/CommitProxyServer.cpp +++ b/fdbserver/commitproxy/CommitProxyServer.cpp @@ -2303,10 +2303,9 @@ static Future readRequestServer(CommitProxyInterface proxy, while (true) { GetKeyServerLocationsRequest req = co_await proxy.getKeyServersLocations.getFuture(); // WARNING: this code is run at a high priority, so it needs to do as little work as possible - if (req.limit != CLIENT_KNOBS->STORAGE_METRICS_SHARD_LIMIT && // Always do data distribution requests - (commitData->stats.keyServerLocationIn.getValue() - commitData->stats.keyServerLocationOut.getValue() > - SERVER_KNOBS->KEY_LOCATION_MAX_QUEUE_SIZE || - (g_network->isSimulated() && buggify(0.001)))) { + if (commitData->stats.keyServerLocationIn.getValue() - commitData->stats.keyServerLocationOut.getValue() > + SERVER_KNOBS->KEY_LOCATION_MAX_QUEUE_SIZE || + (g_network->isSimulated() && buggify(0.001))) { ++commitData->stats.keyServerLocationErrors; req.reply.sendError(commit_proxy_memory_limit_exceeded()); TraceEvent(SevWarnAlways, "ProxyLocationRequestThresholdExceeded").suppressFor(60); From baf46ffa372fe26f7f1e38bdae874d3f707eb4f7 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Mon, 21 Sep 2026 14:01:36 -0700 Subject: [PATCH 137/170] Clarify CDC proxy balancing goals and limits --- design/cdc.md | 15 +++++++++------ 1 file changed, 9 insertions(+), 6 deletions(-) diff --git a/design/cdc.md b/design/cdc.md index 0ead0c8f6ea..1df60ab40a5 100644 --- a/design/cdc.md +++ b/design/cdc.md @@ -430,12 +430,20 @@ still scans global metadata when the representative is absent or invalid. ### Balancing ownership between live CDC proxies +Live CDC proxies can own unequal numbers of active streams. The balancer reduces +this stream-count skew while preserving ownership of each shared current-tag +group. It does not balance producer bytes, filtering cost, or consumer lag, and +cannot divide one hot stream among proxies. Groups with mixed owners are left +unchanged; repairing their ownership is outside this policy. Throughput-aware +proxy placement needs measured load and a separate policy; the opt-in +tag-retagging controller can inform a later version. + `CDC_PROXY_REBALANCE_ENABLED` is disabled by default. When enabled, the cluster controller makes at most one live ownership move per `CDC_PROXY_REBALANCE_INTERVAL` (60 seconds by default) while fully recovered. It groups active streams by their current CDC tag and moves a complete group only when that strictly reduces the stream-count difference between two -published proxies. A group with mixed owners is never moved. The controller +published proxies. The controller skips a pass when the metadata exceeds 512 active streams or assignments, or 2,048 tag-history rows, or one MiB in any of those three ranges. Each pass has a five-second transaction timeout and at most three attempts. It skips @@ -451,11 +459,6 @@ but unacknowledged mutations, as with proxy replacement. Tag routing, stream identities, acknowledgement, and safe-pop metadata do not change. Disabling the balancer or CDC admission stops future moves without requiring a stream drain. -This policy balances the number of streams, not producer bytes, filtering cost, -or consumer lag. One hot stream cannot be divided among proxies by moving its -whole tag group. Throughput-aware proxy placement needs measured load and a -separate policy; the opt-in tag-retagging controller can inform a later version. - ### Metadata lifecycle example Assume a client registers stream name `orders` for range From 64e91ca113cc201087360a99056f89af4da0c3ed Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Mon, 21 Sep 2026 14:03:10 -0700 Subject: [PATCH 138/170] Buggify CDC proxy rebalance enablement and interval --- fdbserver/core/ServerKnobs.cpp | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fdbserver/core/ServerKnobs.cpp b/fdbserver/core/ServerKnobs.cpp index 73644ef817e..df6bcd92417 100644 --- a/fdbserver/core/ServerKnobs.cpp +++ b/fdbserver/core/ServerKnobs.cpp @@ -189,8 +189,8 @@ void ServerKnobs::initialize(Randomize randomize, ClientKnobs* clientKnobs, IsSi init( CDC_PROXY_FAILURE_COALESCE_DELAY, 0.0 ); init( CDC_PROXY_POP_MIN_INTERVAL, 0.1 ); if( randomize && buggify() ) CDC_PROXY_POP_MIN_INTERVAL = 0.01; init( CDC_PROXY_POP_SCAN_INTERVAL, 5.0 ); if( randomize && buggify() ) CDC_PROXY_POP_SCAN_INTERVAL = 0.1; - init( CDC_PROXY_REBALANCE_ENABLED, false ); - init( CDC_PROXY_REBALANCE_INTERVAL, 60.0 ); + init( CDC_PROXY_REBALANCE_ENABLED, false ); if( randomize && buggify() ) CDC_PROXY_REBALANCE_ENABLED = true; + init( CDC_PROXY_REBALANCE_INTERVAL, 60.0 ); if( randomize && buggify() ) CDC_PROXY_REBALANCE_INTERVAL = deterministicRandom()->randomInt(10, 121); init( APPLY_MUTATION_BYTES, 1e6 ); init( BUGGIFY_RECOVER_MEMORY_LIMIT, 1e6 ); init( BUGGIFY_WORKER_REMOVED_MAX_LAG, 30 ); From e630517975e8daefb40395c3dbee0ae63b402162 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Mon, 21 Sep 2026 19:20:07 -0700 Subject: [PATCH 139/170] Clarify anti-quorum recovery and old TLog retirement --- fdbserver/clustercontroller/ClusterController.h | 2 ++ fdbserver/clustercontroller/ClusterRecovery.cpp | 4 +++- fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h | 4 +++- 3 files changed, 8 insertions(+), 2 deletions(-) diff --git a/fdbserver/clustercontroller/ClusterController.h b/fdbserver/clustercontroller/ClusterController.h index 5cd1751fee1..77e0d0bae1c 100644 --- a/fdbserver/clustercontroller/ClusterController.h +++ b/fdbserver/clustercontroller/ClusterController.h @@ -2458,6 +2458,8 @@ class ClusterControllerData { std::vector backup_workers; std::set backup_addresses; + // Old TLog roles retire after terminal recovery; their exclusion does not require another recovery. + // Excluded current TLogs still need a replacement transaction system. for (auto& logSet : dbi.logSystemConfig.tLogs) { for (auto& it : logSet.tLogs) { auto tlogWorker = id_worker.find(it.interf().filteredLocality.processId()); diff --git a/fdbserver/clustercontroller/ClusterRecovery.cpp b/fdbserver/clustercontroller/ClusterRecovery.cpp index 7799550a0e7..d7b0f7194cc 100644 --- a/fdbserver/clustercontroller/ClusterRecovery.cpp +++ b/fdbserver/clustercontroller/ClusterRecovery.cpp @@ -531,7 +531,9 @@ Future trackTlogRecovery(Reference self, bool allLogs = newState.tLogs.size() == configuration.expectedLogSets(!self->primaryDcId.empty() ? self->primaryDcId[0] : Optional()); - // Anti-quorum recovery must still permit removing a lost region while its old history remains durable. + // Anti-quorum permits STORAGE_RECOVERED before remote catch-up, so a lost region can be removed. + // Until catch-up or reconfiguration makes old history unnecessary, retain it in coordinator state and + // defer FULLY_RECOVERED and old-role retirement; initialization alone does not prove durable catch-up. bool storageRecovered = newState.oldTLogData.empty() || self->logSystem->storageRecovered(); bool finalUpdate = newState.oldTLogData.empty() && allLogs; TraceEvent("TrackTLogRecovery") diff --git a/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h b/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h index 1d7e8ecd466..b031e45117b 100644 --- a/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h +++ b/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h @@ -368,9 +368,11 @@ struct LogSystem : ReferenceCounted { // Convert LogSystem to DBCoreState and override input newState as return value void toCoreState(DBCoreState& newState) const; - // The storage/backup recovery policy can be satisfied before remote logs have copied their old prefix. + // The storage/backup recovery policy can be satisfied before remote logs have copied their old generations' data. + // This permits STORAGE_RECOVERED under anti-quorum, but does not make old log history safe to discard. bool storageRecovered() const; bool remoteStorageRecovered() const; + // Waits for durable remote TLog progress through the current local start version, not remote storage recovery. // Requires every expected current log set. The returned future is shared by this recovery. Future onRemoteLogPrefixDurable(); From d9623304edd58b5f9c253e1b2446346945a4d439 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Mon, 21 Sep 2026 22:22:20 -0700 Subject: [PATCH 140/170] Avoid recreating removed storage interfaces during hot-shard monitoring --- fdbserver/ratekeeper/Ratekeeper.cpp | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/fdbserver/ratekeeper/Ratekeeper.cpp b/fdbserver/ratekeeper/Ratekeeper.cpp index 8254dcfb2b5..eae4dbd374f 100644 --- a/fdbserver/ratekeeper/Ratekeeper.cpp +++ b/fdbserver/ratekeeper/Ratekeeper.cpp @@ -292,12 +292,18 @@ Future Ratekeeper::monitorHotShards(Reference const } UID ssi = ssHighWriteQueue.get(); + auto interface = storageServerInterfaces.find(ssi); + // The selected server may have left since the last rate update. + if (interface == storageServerInterfaces.end()) { + CODE_PROBE(true, "Hot shard storage server removed before monitoring"); + continue; + } SetThrottledShardRequest setReq; // TraceEvent(SevDebug, "SendGetHotShardsRequest"); try { GetHotShardsRequest getReq; - GetHotShardsReply reply = co_await storageServerInterfaces[ssi].getHotShards.getReply(getReq); + GetHotShardsReply reply = co_await interface->second.getHotShards.getReply(getReq); // Backup's restore range can't be throttled, otherwise restore would fail, // i.e., "ApplyMutationsError". From ad55df6436fbf44d73655e3704680f36a90c854a Mon Sep 17 00:00:00 2001 From: Michael Stack Date: Tue, 22 Sep 2026 11:46:38 -0700 Subject: [PATCH 141/170] Fix integer overflow in memory tracker frame-pointer walk bounds check (BulkDumpingS3WithChaos failure) (#14104) * Fix integer overflow in memory tracker frame-pointer walk bounds check captureFramesFP guarded each frame-pointer dereference with `a + 16 > hi`. That addition wraps when `a` is within 16 bytes of the top of the address space, so a garbage saved-FP value of 0xfffffffffffffff0 passed the upper bound test, passed the alignment test, and passed the monotonic `next <= fp` test -- nothing is above it -- and then faulted reading fp[1] at 0xfffffffffffffff8. Express the bound as a subtraction instead. The tracker samples allocations from the global operator new/new[], so any workload can reach this. It reproduced on roughly 1% of simulation runs of a bulk-dump test; 100,000 runs of that test are clean with this change, which puts the 95% upper bound on the residual rate near 3e-5. Also floor the walk at the live frame rather than at gStackLow. For the initial thread pthread_attr_getstack reports the mapped top paired with an RLIMIT_STACK-sized length, so gStackLow sits megabytes below the first byte the kernel has mapped -- a ~9.9 MiB unmapped window under a 10 MiB stack limit -- and the guard accepted all of it. This is an unproven fix for a latent second defect rather than the cause of the crash above: the captured fault address fell outside the reported stack range. Taking the higher of gStackLow and the live frame also makes a frame pointer on an unrelated stack fail the bounds test immediately instead of being walked against this thread's range. Register the crash handler with SA_SIGINFO and record the fault address, the reported and mapped stack extents, and RLIMIT_STACK on the Crash trace event. Without SA_SIGINFO si_addr was discarded, so segfault reports carried no fault address at all; recovering it is what identified the overflow. * Shorten comments and drop a dead bounds clause Review feedback: the comments were overlong and read as generated prose. Cut them to what is not already evident from the code -- why gStackLow is not a safe floor, the overflow the subtraction avoids, why the stack extents are sampled at registration rather than in the handler, and what FaultInReportedStack distinguishes. Also drop `hi - 16 < lo` from the precondition. When it holds, the loop's own bounds test breaks on the first iteration and returns 0 anyway. --- flow/MemoryTracker.cpp | 19 +++++----- flow/Platform.cpp | 82 ++++++++++++++++++++++++++++++++++++++++-- 2 files changed, 91 insertions(+), 10 deletions(-) diff --git a/flow/MemoryTracker.cpp b/flow/MemoryTracker.cpp index 7b4027f5331..26b99c33e9b 100644 --- a/flow/MemoryTracker.cpp +++ b/flow/MemoryTracker.cpp @@ -149,17 +149,20 @@ __attribute__((no_instrument_function, noinline)) int captureFramesFP(void** out initStackBoundsForThread(); } void** fp = static_cast(__builtin_frame_address(0)); - // Fallback for threads where pthread_getattr_np failed: ±8 MB around - // the initial frame. - // Caveat: this may need to be constrained more tightly to deal with - // smaller stacks. - uintptr_t lo = gStackLow ? gStackLow : reinterpret_cast(fp); - uintptr_t hi = gStackHigh ? gStackHigh : reinterpret_cast(fp) + (8u << 20); + uintptr_t base = reinterpret_cast(fp); + // gStackLow spans all of RLIMIT_STACK, mostly unmapped for the main thread; the walk only ascends. + uintptr_t lo = std::max(gStackLow, base); + // Fallback for threads where pthread_getattr_np failed: 8 MB above the initial frame. + uintptr_t hi = gStackHigh ? gStackHigh : base + (8u << 20); + if (hi < 16) { + return 0; + } int n = 0; while (fp && n < max) { uintptr_t a = reinterpret_cast(fp); - // Reject out-of-stack or misaligned fp before dereferencing. - if (a < lo || a + 16 > hi) { + // Reject out-of-stack or misaligned fp before dereferencing. Subtraction, not + // `a + 16 > hi`: that addition wraps for an fp near the top of the address space. + if (a < lo || a > hi - 16) { break; } if (a & (sizeof(void*) - 1)) { diff --git a/flow/Platform.cpp b/flow/Platform.cpp index 2fbdafdfc41..58236e4489a 100644 --- a/flow/Platform.cpp +++ b/flow/Platform.cpp @@ -3721,6 +3721,57 @@ void registerCrashHandlerCallback(void (*f)()) { g_crashHandlerCallbacks.push_back(f); } +#ifdef __linux__ +// Sampled at registration time, not in the handler: reading /proc/self/maps needs +// malloc, and the handler can run with the allocator's locks held. +uintptr_t g_reportedStackLow = 0; +uintptr_t g_reportedStackHigh = 0; +uintptr_t g_mainStackMappedLow = 0; +uintptr_t g_mainStackMappedHigh = 0; +uint64_t g_stackRlimit = 0; + +// Global because crashHandler keeps its plain sa_handler signature. +uintptr_t g_faultAddress = 0; +bool g_faultAddressValid = false; + +void sampleMainStackExtent() { + struct rlimit rl; + if (getrlimit(RLIMIT_STACK, &rl) == 0) { + g_stackRlimit = rl.rlim_cur; + } + + // Must match initStackBoundsForThread in flow/MemoryTracker.cpp. + pthread_attr_t attr; + if (pthread_getattr_np(pthread_self(), &attr) == 0) { + void* base = nullptr; + size_t size = 0; + if (pthread_attr_getstack(&attr, &base, &size) == 0) { + g_reportedStackLow = reinterpret_cast(base); + g_reportedStackHigh = g_reportedStackLow + size; + } + pthread_attr_destroy(&attr); + } + + FILE* maps = fopen("/proc/self/maps", "r"); + if (maps == nullptr) { + return; + } + char line[512]; + while (fgets(line, sizeof(line), maps) != nullptr) { + if (strstr(line, "[stack]") == nullptr) { + continue; + } + unsigned long low = 0, high = 0; + if (sscanf(line, "%lx-%lx", &low, &high) == 2) { + g_mainStackMappedLow = low; + g_mainStackMappedHigh = high; + } + break; + } + fclose(maps); +} +#endif + // The crashHandler function is registered to handle signals before the process terminates. // Basic information about the crash is printed/traced, and stdout and trace events are flushed. void crashHandler(int sig) { @@ -3740,6 +3791,9 @@ void crashHandler(int sig) { fprintf(error ? stderr : stdout, "SIGNAL: %s (%d)\n", strsignal(sig), sig); if (error) { + if (g_faultAddressValid) { + fprintf(stderr, "FaultAddress: 0x%zx\n", (size_t)g_faultAddress); + } fprintf(stderr, "Trace: %s\n", backtrace.c_str()); } @@ -3747,6 +3801,18 @@ void crashHandler(int sig) { { TraceEvent te(error ? SevError : SevInfo, error ? "Crash" : "ProcessTerminated"); te.detail("Signal", sig).detail("Name", strsignal(sig)).detail("Trace", backtrace); + if (g_faultAddressValid) { + // FaultInReportedStack separates a wild pointer from stack bounds that reach + // below what is mapped. + te.detail("FaultAddress", format("0x%zx", (size_t)g_faultAddress)) + .detail("ReportedStackLow", format("0x%zx", (size_t)g_reportedStackLow)) + .detail("ReportedStackHigh", format("0x%zx", (size_t)g_reportedStackHigh)) + .detail("MainStackMappedLow", format("0x%zx", (size_t)g_mainStackMappedLow)) + .detail("MainStackMappedHigh", format("0x%zx", (size_t)g_mainStackMappedHigh)) + .detail("StackRlimit", g_stackRlimit) + .detail("FaultInReportedStack", + g_faultAddress >= g_reportedStackLow && g_faultAddress < g_reportedStackHigh); + } if (error) { te.setErrorKind(ErrorKind::BugDetected); } @@ -3781,14 +3847,26 @@ void crashHandler(int sig) { #endif } +#ifdef __linux__ +void crashHandlerSigInfo(int sig, siginfo_t* info, void*) { + if (info != nullptr && (sig == SIGSEGV || sig == SIGBUS)) { + g_faultAddress = reinterpret_cast(info->si_addr); + g_faultAddressValid = true; + } + crashHandler(sig); +} +#endif + void registerCrashHandler() { #ifdef __linux__ + sampleMainStackExtent(); + // For these otherwise fatal errors, attempt to log a trace of // what was happening and then exit struct sigaction action; - action.sa_handler = crashHandler; + action.sa_sigaction = crashHandlerSigInfo; sigfillset(&action.sa_mask); - action.sa_flags = 0; + action.sa_flags = SA_SIGINFO; // deliver siginfo_t so SIGSEGV reports its fault address sigaction(SIGILL, &action, nullptr); sigaction(SIGFPE, &action, nullptr); From 14714a9380ccc95715139d8d54828a756754050a Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 22 Sep 2026 12:27:12 -0700 Subject: [PATCH 142/170] Make non-converting constructors explicit --- bindings/c/test/fdb_api.hpp | 8 +++--- bindings/c/test/mako/admin_server.hpp | 2 +- bindings/c/test/mako/ddsketch.hpp | 2 +- bindings/c/test/mako/future.hpp | 2 +- bindings/c/test/mako/stats.hpp | 2 +- bindings/c/test/mako/time.hpp | 4 +-- bindings/c/test/mako/utils.hpp | 4 +-- bindings/c/test/unit/fdb_api.hpp | 26 +++++++++---------- .../fdbclient/DataDistributionConfig.h | 3 ++- fdbserver/SimulatedCluster.cpp | 2 +- .../checkpoint/RocksDBCheckpointUtils.cpp | 2 +- fdbserver/clustercontroller/Status.cpp | 2 +- .../include/fdbserver/core/BulkDumpUtil.h | 2 +- .../include/fdbserver/core/WorkerInterface.h | 10 +++---- .../datadistributor/DataDistribution.cpp | 2 +- fdbserver/datadistributor/DataDistribution.h | 14 +++++----- .../kvstore/KeyValueStoreShardedRocksDB.cpp | 4 +-- fdbserver/kvstore/VersionedBTree.cpp | 14 +++++----- fdbserver/sequencer/ResolutionBalancer.h | 2 +- .../include/fdbserver/tester/workloads.h | 8 +++--- fdbserver/tlog/TestTLogServer.h | 2 +- fdbserver/worker/MetricLogger.cpp | 2 +- fdbserver/workloads/AsyncFile.h | 2 +- fdbserver/workloads/UDPWorkload.cpp | 2 +- flow/include/flow/genericactors.h | 6 ++--- 25 files changed, 65 insertions(+), 64 deletions(-) diff --git a/bindings/c/test/fdb_api.hpp b/bindings/c/test/fdb_api.hpp index f2f9e4235fe..89cbff5fed9 100644 --- a/bindings/c/test/fdb_api.hpp +++ b/bindings/c/test/fdb_api.hpp @@ -345,7 +345,7 @@ class Result { friend class Transaction; std::shared_ptr r; - Result(native::FDBResult* result) { + explicit Result(native::FDBResult* result) { if (result) r = std::shared_ptr(result, &native::fdb_result_destroy); } @@ -377,7 +377,7 @@ class Future { friend std::hash; std::shared_ptr f; - Future(native::FDBFuture* future) { + explicit(false) Future(native::FDBFuture* future) { if (future) f = std::shared_ptr(future, &native::fdb_future_destroy); } @@ -473,7 +473,7 @@ class TypedFuture : public Future { using Future::get; using Future::getNothrow; using Future::then; - TypedFuture(const Future& f) noexcept : Future(f) {} + explicit TypedFuture(const Future& f) noexcept : Future(f) {} public: using ContainedType = typename VarTraits::Type; @@ -686,7 +686,7 @@ class Database : public IDatabaseOps { public: Database(const Database&) noexcept = default; Database& operator=(const Database&) noexcept = default; - Database(const std::string& cluster_file_path) : db(nullptr) { + explicit Database(const std::string& cluster_file_path) : db(nullptr) { auto db_raw = static_cast(nullptr); if (auto err = Error(native::fdb_create_database(cluster_file_path.c_str(), &db_raw))) throwError(fmt::format("Failed to create database with '{}': ", cluster_file_path), err); diff --git a/bindings/c/test/mako/admin_server.hpp b/bindings/c/test/mako/admin_server.hpp index b9b819fb6ca..e5ba4bdd9f3 100644 --- a/bindings/c/test/mako/admin_server.hpp +++ b/bindings/c/test/mako/admin_server.hpp @@ -95,7 +95,7 @@ class AdminServer { } public: - AdminServer(const Arguments& args) + explicit AdminServer(const Arguments& args) : args(args), server_pid(-1), pipe_to_server(boost::process::pipe()), pipe_to_client(boost::process::pipe()) { start(); } diff --git a/bindings/c/test/mako/ddsketch.hpp b/bindings/c/test/mako/ddsketch.hpp index 72cf4738a08..0c429c32e7d 100644 --- a/bindings/c/test/mako/ddsketch.hpp +++ b/bindings/c/test/mako/ddsketch.hpp @@ -220,7 +220,7 @@ class DDSketch : public DDSketchBase, T> { template class DDSketchSlow : public DDSketchBase, T> { public: - DDSketchSlow(double errorGuarantee = 0.1) + explicit DDSketchSlow(double errorGuarantee = 0.1) : DDSketchBase, T>(errorGuarantee), gamma((1.0 + errorGuarantee) / (1.0 - errorGuarantee)), logGamma(log(gamma)) { offset = getIndex(1.0 / DDSketchBase, T>::EPS) + 5; diff --git a/bindings/c/test/mako/future.hpp b/bindings/c/test/mako/future.hpp index 99b1956eb8a..97461b3d73d 100644 --- a/bindings/c/test/mako/future.hpp +++ b/bindings/c/test/mako/future.hpp @@ -36,7 +36,7 @@ enum class FutureRC { OK, RETRY, ABORT }; struct LogContext { static constexpr const bool do_log = true; - LogContext(std::string_view step) noexcept : step(step), transaction_timeout_expected(false) {} + explicit LogContext(std::string_view step) noexcept : step(step), transaction_timeout_expected(false) {} LogContext(std::string_view step, bool transaction_timeout_expected) noexcept : step(step), transaction_timeout_expected(transaction_timeout_expected) {} std::string_view step; diff --git a/bindings/c/test/mako/stats.hpp b/bindings/c/test/mako/stats.hpp index 6348da9515b..42963b48f1b 100644 --- a/bindings/c/test/mako/stats.hpp +++ b/bindings/c/test/mako/stats.hpp @@ -311,7 +311,7 @@ class CPUUtilizationTimer { TimerKind kind; public: - CPUUtilizationTimer(TimerKind kind) : kind(kind) {} + explicit CPUUtilizationTimer(TimerKind kind) : kind(kind) {} void start() { timepoint_start = steady_clock::now(); cpu_time_start = (kind == THREAD) ? getProcessorTimeThread() : getProcessorTimeProcess(); diff --git a/bindings/c/test/mako/time.hpp b/bindings/c/test/mako/time.hpp index d962421ac12..f45ee86d9d4 100644 --- a/bindings/c/test/mako/time.hpp +++ b/bindings/c/test/mako/time.hpp @@ -53,8 +53,8 @@ class Stopwatch { public: Stopwatch() noexcept : p1(), p2() {} - Stopwatch(StartAtCtor) noexcept { start(); } - Stopwatch(timepoint_t start_time) noexcept : p1(start_time), p2() {} + explicit Stopwatch(StartAtCtor) noexcept { start(); } + explicit Stopwatch(timepoint_t start_time) noexcept : p1(start_time), p2() {} Stopwatch(const Stopwatch&) noexcept = default; Stopwatch& operator=(const Stopwatch&) noexcept = default; timepoint_t getStart() const noexcept { return p1; } diff --git a/bindings/c/test/mako/utils.hpp b/bindings/c/test/mako/utils.hpp index 80ee8dfc261..dbec5a1f5d0 100644 --- a/bindings/c/test/mako/utils.hpp +++ b/bindings/c/test/mako/utils.hpp @@ -163,7 +163,7 @@ class ExitGuard { std::decay_t fn; public: - ExitGuard(Func&& fn) : fn(std::forward(fn)) {} + explicit ExitGuard(Func&& fn) : fn(std::forward(fn)) {} ~ExitGuard() { fn(); } }; @@ -174,7 +174,7 @@ class FailGuard { std::decay_t fn; public: - FailGuard(Func&& fn) : fn(std::forward(fn)) {} + explicit FailGuard(Func&& fn) : fn(std::forward(fn)) {} ~FailGuard() { if (std::uncaught_exceptions()) { diff --git a/bindings/c/test/unit/fdb_api.hpp b/bindings/c/test/unit/fdb_api.hpp index 335b62dc62f..cafccdb335f 100644 --- a/bindings/c/test/unit/fdb_api.hpp +++ b/bindings/c/test/unit/fdb_api.hpp @@ -73,7 +73,7 @@ class Future { // } protected: - Future(FDBFuture* f) : future_(f) {} + explicit Future(FDBFuture* f) : future_(f) {} FDBFuture* future_; }; @@ -86,7 +86,7 @@ class Int64Future : public Future { private: friend class Transaction; friend class Database; - Int64Future(FDBFuture* f) : Future(f) {} + explicit Int64Future(FDBFuture* f) : Future(f) {} }; class DoubleFuture : public Future { @@ -98,7 +98,7 @@ class DoubleFuture : public Future { private: friend class Transaction; friend class Database; - DoubleFuture(FDBFuture* f) : Future(f) {} + explicit DoubleFuture(FDBFuture* f) : Future(f) {} }; class KeyFuture : public Future { @@ -110,7 +110,7 @@ class KeyFuture : public Future { private: friend class Transaction; friend class Database; - KeyFuture(FDBFuture* f) : Future(f) {} + explicit KeyFuture(FDBFuture* f) : Future(f) {} }; class ValueFuture : public Future { @@ -121,7 +121,7 @@ class ValueFuture : public Future { private: friend class Transaction; - ValueFuture(FDBFuture* f) : Future(f) {} + explicit ValueFuture(FDBFuture* f) : Future(f) {} }; class StringArrayFuture : public Future { @@ -133,7 +133,7 @@ class StringArrayFuture : public Future { private: friend class Transaction; - StringArrayFuture(FDBFuture* f) : Future(f) {} + explicit StringArrayFuture(FDBFuture* f) : Future(f) {} }; class KeyValueArrayFuture : public Future { @@ -145,7 +145,7 @@ class KeyValueArrayFuture : public Future { private: friend class Transaction; - KeyValueArrayFuture(FDBFuture* f) : Future(f) {} + explicit KeyValueArrayFuture(FDBFuture* f) : Future(f) {} }; class MappedKeyValueArrayFuture : public Future { @@ -157,7 +157,7 @@ class MappedKeyValueArrayFuture : public Future { private: friend class Transaction; - MappedKeyValueArrayFuture(FDBFuture* f) : Future(f) {} + explicit MappedKeyValueArrayFuture(FDBFuture* f) : Future(f) {} }; class KeyRangeArrayFuture : public Future { @@ -169,14 +169,14 @@ class KeyRangeArrayFuture : public Future { private: friend class Transaction; - KeyRangeArrayFuture(FDBFuture* f) : Future(f) {} + explicit KeyRangeArrayFuture(FDBFuture* f) : Future(f) {} }; class EmptyFuture : public Future { private: friend class Transaction; friend class Database; - EmptyFuture(FDBFuture* f) : Future(f) {} + explicit EmptyFuture(FDBFuture* f) : Future(f) {} }; class Result { @@ -184,7 +184,7 @@ class Result { virtual ~Result() = 0; protected: - Result(FDBResult* r) : result_(r) {} + explicit Result(FDBResult* r) : result_(r) {} FDBResult* result_; }; @@ -197,7 +197,7 @@ class KeyValueArrayResult : public Result { private: friend class Transaction; - KeyValueArrayResult(FDBResult* r) : Result(r) {} + explicit KeyValueArrayResult(FDBResult* r) : Result(r) {} }; // Wrapper around FDBDatabase, providing database-level API @@ -222,7 +222,7 @@ class Database final { class Transaction final { public: // Given an FDBDatabase, initializes a new transaction. - Transaction(FDBDatabase* db); + explicit Transaction(FDBDatabase* db); ~Transaction(); // Wrapper around fdb_transaction_reset. diff --git a/fdbclient/include/fdbclient/DataDistributionConfig.h b/fdbclient/include/fdbclient/DataDistributionConfig.h index 0176cb8c40b..f6d6509de76 100644 --- a/fdbclient/include/fdbclient/DataDistributionConfig.h +++ b/fdbclient/include/fdbclient/DataDistributionConfig.h @@ -37,7 +37,8 @@ struct DDRangeConfig { constexpr static FileIdentifier file_identifier = 9193856; - explicit(false) DDRangeConfig(Optional replicationFactor = {}, Optional teamID = {}) + DDRangeConfig() = default; + explicit DDRangeConfig(Optional replicationFactor, Optional teamID = {}) : replicationFactor(replicationFactor), teamID(teamID) {} Optional replicationFactor; diff --git a/fdbserver/SimulatedCluster.cpp b/fdbserver/SimulatedCluster.cpp index 42073d6de31..04c83ea11ef 100644 --- a/fdbserver/SimulatedCluster.cpp +++ b/fdbserver/SimulatedCluster.cpp @@ -608,7 +608,7 @@ class TestConfig : public BasicTestConfig { } TestConfig() = default; - explicit(false) TestConfig(const BasicTestConfig& config) : BasicTestConfig(config) {} + explicit TestConfig(const BasicTestConfig& config) : BasicTestConfig(config) {} }; template diff --git a/fdbserver/checkpoint/RocksDBCheckpointUtils.cpp b/fdbserver/checkpoint/RocksDBCheckpointUtils.cpp index 83dbc625f35..d0b87f2ef47 100644 --- a/fdbserver/checkpoint/RocksDBCheckpointUtils.cpp +++ b/fdbserver/checkpoint/RocksDBCheckpointUtils.cpp @@ -892,7 +892,7 @@ RangeResult RocksDBSstFileReader::getRange(const KeyRange& range) { class RocksDBCheckpointByteSampleReader : public ICheckpointByteSampleReader { public: - explicit(false) RocksDBCheckpointByteSampleReader(const CheckpointMetaData& checkpoint); + explicit RocksDBCheckpointByteSampleReader(const CheckpointMetaData& checkpoint); ~RocksDBCheckpointByteSampleReader() override = default; KeyValue next() override; diff --git a/fdbserver/clustercontroller/Status.cpp b/fdbserver/clustercontroller/Status.cpp index cd91c939c5c..820d60a5826 100644 --- a/fdbserver/clustercontroller/Status.cpp +++ b/fdbserver/clustercontroller/Status.cpp @@ -178,7 +178,7 @@ class StatusCounter { public: StatusCounter() : hz(0), roughness(0), counter(0) {} StatusCounter(double hz, double roughness, int64_t counter) : hz(hz), roughness(roughness), counter(counter) {} - explicit(false) StatusCounter(const std::string& parsableText) { parseText(parsableText); } + explicit StatusCounter(const std::string& parsableText) { parseText(parsableText); } StatusCounter& parseText(const std::string& parsableText) { sscanf(parsableText.c_str(), "%lf %lf %" SCNd64 "", &hz, &roughness, &counter); diff --git a/fdbserver/core/include/fdbserver/core/BulkDumpUtil.h b/fdbserver/core/include/fdbserver/core/BulkDumpUtil.h index 1725f25d8e4..b5c767a7bd8 100644 --- a/fdbserver/core/include/fdbserver/core/BulkDumpUtil.h +++ b/fdbserver/core/include/fdbserver/core/BulkDumpUtil.h @@ -104,7 +104,7 @@ Future uploadBulkDumpFileSet(BulkLoadTransportMethod transportMethod, class ParallelismLimitor { public: - explicit(false) ParallelismLimitor(int maxParallelism) : maxParallelism(maxParallelism) {} + explicit ParallelismLimitor(int maxParallelism) : maxParallelism(maxParallelism) {} inline void decrementTaskCounter() { ASSERT(numRunningTasks.get() <= maxParallelism); diff --git a/fdbserver/core/include/fdbserver/core/WorkerInterface.h b/fdbserver/core/include/fdbserver/core/WorkerInterface.h index 428ad321268..989e33d32d3 100644 --- a/fdbserver/core/include/fdbserver/core/WorkerInterface.h +++ b/fdbserver/core/include/fdbserver/core/WorkerInterface.h @@ -79,7 +79,7 @@ struct WorkerInterface { Optional grpcAddress() const { return clientInterface.grpcAddress; } WorkerInterface() = default; - explicit(false) WorkerInterface(const LocalityData& locality) : locality(locality) {} + explicit WorkerInterface(const LocalityData& locality) : locality(locality) {} void initEndpoints() { clientInterface.initEndpoints(); @@ -573,7 +573,7 @@ struct GetEncryptionAtRestModeResponse { uint32_t mode; GetEncryptionAtRestModeResponse() : mode(EncryptionAtRestModeDeprecated::Mode::DISABLED) {} - explicit(false) GetEncryptionAtRestModeResponse(uint32_t m) : mode(m) {} + explicit GetEncryptionAtRestModeResponse(uint32_t m) : mode(m) {} template void serialize(Ar& ar) { @@ -587,7 +587,7 @@ struct GetEncryptionAtRestModeRequest { ReplyPromise reply; GetEncryptionAtRestModeRequest() = default; - explicit(false) GetEncryptionAtRestModeRequest(UID tId) : tlogId(tId) {} + explicit GetEncryptionAtRestModeRequest(UID tId) : tlogId(tId) {} template void serialize(Ar& ar) { @@ -938,7 +938,7 @@ struct ExecuteRequest { Arena arena; StringRef execPayload; - explicit(false) ExecuteRequest(StringRef execPayload) : execPayload(execPayload) {} + explicit ExecuteRequest(StringRef execPayload) : execPayload(execPayload) {} ExecuteRequest() : execPayload() {} @@ -1059,7 +1059,7 @@ struct DiskStoreRequest { bool includePartialStores; ReplyPromise>> reply; - explicit(false) DiskStoreRequest(bool includePartialStores = false) : includePartialStores(includePartialStores) {} + explicit DiskStoreRequest(bool includePartialStores = false) : includePartialStores(includePartialStores) {} template void serialize(Ar& ar) { diff --git a/fdbserver/datadistributor/DataDistribution.cpp b/fdbserver/datadistributor/DataDistribution.cpp index 9365b10bbfa..7b8da096bda 100644 --- a/fdbserver/datadistributor/DataDistribution.cpp +++ b/fdbserver/datadistributor/DataDistribution.cpp @@ -137,7 +137,7 @@ enum class DDAuditContext : uint8_t { }; struct DDAudit { - explicit(false) DDAudit(AuditStorageState coreState) + explicit DDAudit(AuditStorageState coreState) : coreState(coreState), actors(true), foundError(false), auditStorageAnyChildFailed(false), retryCount(0), cancelled(false), overallCompleteDoAuditCount(0), overallIssuedDoAuditCount(0), overallSkippedDoAuditCount(0), remainingBudgetForAuditTasks(SERVER_KNOBS->CONCURRENT_AUDIT_TASK_COUNT_MAX), context(DDAuditContext::INVALID) {} diff --git a/fdbserver/datadistributor/DataDistribution.h b/fdbserver/datadistributor/DataDistribution.h index 268e90d92c3..40345b12056 100644 --- a/fdbserver/datadistributor/DataDistribution.h +++ b/fdbserver/datadistributor/DataDistribution.h @@ -162,7 +162,7 @@ struct GetMetricsRequest { KeyRange keys; Promise reply; GetMetricsRequest() = default; - explicit(false) GetMetricsRequest(KeyRange const& keys) : keys(keys) {} + explicit GetMetricsRequest(KeyRange const& keys) : keys(keys) {} }; struct GetTopKMetricsReply { @@ -188,10 +188,10 @@ struct GetTopKMetricsRequest { double maxReadLoadPerKSecond = 0, minReadLoadPerKSecond = 0; // all returned shards won't exceed this read load GetTopKMetricsRequest() = default; - explicit(false) GetTopKMetricsRequest(std::vector const& keys, - int topK = 1, - double maxReadLoadPerKSecond = std::numeric_limits::max(), - double minReadLoadPerKSecond = 0) + explicit GetTopKMetricsRequest(std::vector const& keys, + int topK = 1, + double maxReadLoadPerKSecond = std::numeric_limits::max(), + double minReadLoadPerKSecond = 0) : topK(topK), keys(keys), maxReadLoadPerKSecond(maxReadLoadPerKSecond), minReadLoadPerKSecond(minReadLoadPerKSecond) { ASSERT_GE(topK, 1); @@ -252,7 +252,7 @@ FDB_BOOLEAN_PARAM(MoveKeyRangeOutPhysicalShard); class PhysicalShardCollection : public ReferenceCounted { public: PhysicalShardCollection() : lastTransitionStartTime(now()), requireTransition(false) {} - explicit(false) PhysicalShardCollection(Reference db) + explicit PhysicalShardCollection(Reference db) : txnProcessor(db), lastTransitionStartTime(now()), requireTransition(false) {} enum class PhysicalShardCreationTime { DDInit, DDRelocator }; @@ -604,7 +604,7 @@ inline bool bulkDumpIsEnabled(int bulkDumpModeValue) { class BulkLoadTaskCollection : public ReferenceCounted { public: - explicit(false) BulkLoadTaskCollection(UID ddId) : ddId(ddId) { + explicit BulkLoadTaskCollection(UID ddId) : ddId(ddId) { bulkLoadTaskMap.insert(allKeys, Optional()); } diff --git a/fdbserver/kvstore/KeyValueStoreShardedRocksDB.cpp b/fdbserver/kvstore/KeyValueStoreShardedRocksDB.cpp index b8b805ea6bd..ee03399e270 100644 --- a/fdbserver/kvstore/KeyValueStoreShardedRocksDB.cpp +++ b/fdbserver/kvstore/KeyValueStoreShardedRocksDB.cpp @@ -362,7 +362,7 @@ class CompactOnRangeDeletionCollectorFactory : public rocksdb::TablePropertiesCo // A factory of a table property collector that marks a SST file as need-compaction when the number of range // deletions exceeds the threshold. // @param numRangeDeletionsAllowed, triggers compaction range deletion count exceeds numRangeDeletionsAllowed. - explicit(false) CompactOnRangeDeletionCollectorFactory(uint64_t numRangeDeletionsAllowed) + explicit CompactOnRangeDeletionCollectorFactory(uint64_t numRangeDeletionsAllowed) : threshold(numRangeDeletionsAllowed), numFilesMarkedForCompaction(0) {} ~CompactOnRangeDeletionCollectorFactory() override = default; @@ -2515,7 +2515,7 @@ struct ShardedRocksDBKeyValueStore : IKeyValueStore { struct DeleteVisitor : public rocksdb::WriteBatch::Handler { std::vector>* deletes; - explicit(false) DeleteVisitor(std::vector>* deletes) : deletes(deletes) { + explicit DeleteVisitor(std::vector>* deletes) : deletes(deletes) { ASSERT(deletes); } diff --git a/fdbserver/kvstore/VersionedBTree.cpp b/fdbserver/kvstore/VersionedBTree.cpp index 1a819a732d3..bffe295da01 100644 --- a/fdbserver/kvstore/VersionedBTree.cpp +++ b/fdbserver/kvstore/VersionedBTree.cpp @@ -1658,7 +1658,7 @@ class ObjectCache : NonCopyable { // must eventually give them back with moveIn() or remove them with reclaim(). class Evictor : NonCopyable { public: - explicit(false) Evictor(int64_t sizeLimit = 0) : sizeLimit(sizeLimit) {} + explicit Evictor(int64_t sizeLimit = 0) : sizeLimit(sizeLimit) {} // Evictors are normally singletons, either one per real process or one per virtual process in simulation static Evictor* getEvictor() { @@ -1791,7 +1791,7 @@ class ObjectCache : NonCopyable { int64_t movedOutCount = 0; }; - explicit(false) ObjectCache(Evictor* evictor = nullptr) : pEvictor(evictor) { + explicit ObjectCache(Evictor* evictor = nullptr) : pEvictor(evictor) { if (pEvictor == nullptr) { pEvictor = Evictor::getEvictor(); } @@ -2061,9 +2061,9 @@ class DWALPager final : public IPager2 { struct RemappedPage { enum Type { NONE = 'N', REMAP = 'R', FREE = 'F', DETACH = 'D' }; - explicit(false) RemappedPage(Version v = invalidVersion, - LogicalPageID o = invalidLogicalPageID, - LogicalPageID n = invalidLogicalPageID) + explicit RemappedPage(Version v = invalidVersion, + LogicalPageID o = invalidLogicalPageID, + LogicalPageID n = invalidLogicalPageID) : version(v), originalPageID(o), newPageID(n) {} Version version; @@ -5333,7 +5333,7 @@ class VersionedBTree { // Clear SingleKeyMutation() : op(MutationRef::ClearRange) {} // Set - explicit(false) SingleKeyMutation(Value val) : op(MutationRef::SetValue), value(val) {} + explicit SingleKeyMutation(Value val) : op(MutationRef::SetValue), value(val) {} // Atomic Op SingleKeyMutation(MutationRef::Type op, Value val) : op(op), value(val) {} @@ -11172,7 +11172,7 @@ struct KVSource { // TODO there is probably a better way to do this Prefix extraRangePrefix; - explicit(false) KVSource(const std::vector& desc, int numPrefixes = 0) : desc(desc) { + explicit KVSource(const std::vector& desc, int numPrefixes = 0) : desc(desc) { if (numPrefixes == 0) { numPrefixes = 1; for (auto& p : desc) { diff --git a/fdbserver/sequencer/ResolutionBalancer.h b/fdbserver/sequencer/ResolutionBalancer.h index 12a3a4e25ae..bfff88fe5f2 100644 --- a/fdbserver/sequencer/ResolutionBalancer.h +++ b/fdbserver/sequencer/ResolutionBalancer.h @@ -41,7 +41,7 @@ struct ResolutionBalancer { std::vector resolvers; AsyncTrigger triggerResolution; - explicit(false) ResolutionBalancer(Version* version) : pVersion(version) {} + explicit ResolutionBalancer(Version* version) : pVersion(version) {} Future resolutionBalancing(); diff --git a/fdbserver/tester/include/fdbserver/tester/workloads.h b/fdbserver/tester/include/fdbserver/tester/workloads.h index ca227251b08..d275e573cea 100644 --- a/fdbserver/tester/include/fdbserver/tester/workloads.h +++ b/fdbserver/tester/include/fdbserver/tester/workloads.h @@ -108,7 +108,7 @@ struct TestWorkloadImpl : Workload { static_assert(std::is_same_v, "Workload must not override TestWorkload::description"); - explicit(false) TestWorkloadImpl(WorkloadContext const& wcx) : Workload(wcx) {} + explicit TestWorkloadImpl(WorkloadContext const& wcx) : Workload(wcx) {} template requires(E) TestWorkloadImpl(WorkloadContext const& wcx, NoOptions o) : Workload(wcx, o) {} @@ -120,7 +120,7 @@ struct CompoundWorkload; class DeterministicRandom; struct FailureInjectionWorkload : TestWorkload { - explicit(false) FailureInjectionWorkload(WorkloadContext const&); + explicit FailureInjectionWorkload(WorkloadContext const&); ~FailureInjectionWorkload() override = default; virtual void initFailureInjectionMode(DeterministicRandom& random); virtual bool shouldInject(DeterministicRandom& random, const WorkloadRequest& work, const unsigned count) const; @@ -154,7 +154,7 @@ struct CompoundWorkload : TestWorkload { std::vector> workloads; std::vector> failureInjection; - explicit(false) CompoundWorkload(WorkloadContext& wcx); + explicit CompoundWorkload(WorkloadContext& wcx); CompoundWorkload* add(Reference&& w); void addFailureInjection(WorkloadRequest& work); bool shouldInjectFailure(DeterministicRandom& random, @@ -254,7 +254,7 @@ struct WorkloadFactory : IWorkloadFactory { "Each workload must have a Workload::NAME member"); using WorkloadType = TestWorkloadImpl; bool runInUntrustedClient; - explicit(false) WorkloadFactory(UntrustedMode runInUntrustedClient = UntrustedMode::False) + explicit WorkloadFactory(UntrustedMode runInUntrustedClient = UntrustedMode::False) : runInUntrustedClient(runInUntrustedClient) { auto& f = factories(); std::string name = WorkloadType::NAME; diff --git a/fdbserver/tlog/TestTLogServer.h b/fdbserver/tlog/TestTLogServer.h index bc179a8954d..e89af65cb38 100644 --- a/fdbserver/tlog/TestTLogServer.h +++ b/fdbserver/tlog/TestTLogServer.h @@ -102,7 +102,7 @@ struct TLogTestContext : NonCopyable, public ReferenceCounted { static Future peekCommitMessages(TLogTestContext* pTLogTestContext, uint16_t logGroupID, uint32_t tag); - explicit(false) TLogTestContext(TestTLogOptions& tLogOptions) : tLogOptions(tLogOptions), epoch(1) {} + explicit TLogTestContext(TestTLogOptions& tLogOptions) : tLogOptions(tLogOptions), epoch(1) {} // paramaters std::string diskQueueBasename; diff --git a/fdbserver/worker/MetricLogger.cpp b/fdbserver/worker/MetricLogger.cpp index aa8127534c0..ca2a66c59a0 100644 --- a/fdbserver/worker/MetricLogger.cpp +++ b/fdbserver/worker/MetricLogger.cpp @@ -45,7 +45,7 @@ namespace { struct MetricsRule { - explicit(false) MetricsRule(bool enabled = false, int minLevel = 0, StringRef const& name = StringRef()) + explicit MetricsRule(bool enabled = false, int minLevel = 0, StringRef const& name = StringRef()) : namePattern(name), enabled(enabled), minLevel(minLevel) {} Standalone typePattern; diff --git a/fdbserver/workloads/AsyncFile.h b/fdbserver/workloads/AsyncFile.h index b85cf928bd3..b8ff9389875 100644 --- a/fdbserver/workloads/AsyncFile.h +++ b/fdbserver/workloads/AsyncFile.h @@ -67,7 +67,7 @@ struct AsyncFileWorkload : TestWorkload { std::string path; - explicit(false) AsyncFileWorkload(WorkloadContext const&); + explicit AsyncFileWorkload(WorkloadContext const&); ~AsyncFileWorkload() override = default; // Allocates a buffer of a given size. If necessary, the buffer will be aligned to 4K diff --git a/fdbserver/workloads/UDPWorkload.cpp b/fdbserver/workloads/UDPWorkload.cpp index 28b99593a0b..bb5b9cb19e8 100644 --- a/fdbserver/workloads/UDPWorkload.cpp +++ b/fdbserver/workloads/UDPWorkload.cpp @@ -50,7 +50,7 @@ struct UDPWorkload : TestWorkload { std::unordered_map sent, received, acked, successes; PromiseStream toAck; - explicit(false) UDPWorkload(WorkloadContext const& wcx) : TestWorkload(wcx) { + explicit UDPWorkload(WorkloadContext const& wcx) : TestWorkload(wcx) { keyPrefix = getOption(options, "keyPrefix"_sr, "/udp/"_sr); runFor = getOption(options, "runFor"_sr, 60.0); minPort = getOption(options, "minPort"_sr, 5000); diff --git a/flow/include/flow/genericactors.h b/flow/include/flow/genericactors.h index 443babad286..fe143e86454 100644 --- a/flow/include/flow/genericactors.h +++ b/flow/include/flow/genericactors.h @@ -2027,7 +2027,7 @@ struct FlowLock : NonCopyable, public ReferenceCounted { FlowLock* lock; int64_t remaining; Releaser() : lock(0), remaining(0) {} - explicit(false) Releaser(FlowLock& lock, int64_t amount = 1) : lock(&lock), remaining(amount) {} + explicit Releaser(FlowLock& lock, int64_t amount = 1) : lock(&lock), remaining(amount) {} Releaser(Releaser&& r) noexcept : lock(r.lock), remaining(r.remaining) { r.remaining = 0; } void operator=(Releaser&& r) { if (remaining) @@ -2402,9 +2402,9 @@ class AndFuture { AndFuture& operator=(AndFuture const& f) = default; AndFuture& operator=(AndFuture&& f) noexcept = default; - explicit(false) AndFuture(Future const& f) : futureCount(1), futures{ f } {} + explicit AndFuture(Future const& f) : futureCount(1), futures{ f } {} - explicit(false) AndFuture(Error const& e) : futureCount(1), futures{ Future(e) } {} + explicit AndFuture(Error const& e) : futureCount(1), futures{ Future(e) } {} operator Future() { return getFuture(); } From dd80972103458c00abb931495861aee425878629 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 22 Sep 2026 13:17:05 -0700 Subject: [PATCH 143/170] Suppress string-view length false positives in C API wrapper --- bindings/c/test/fdb_api.hpp | 23 ++++++++++++++++++++++- 1 file changed, 22 insertions(+), 1 deletion(-) diff --git a/bindings/c/test/fdb_api.hpp b/bindings/c/test/fdb_api.hpp index 89cbff5fed9..d9ef6659171 100644 --- a/bindings/c/test/fdb_api.hpp +++ b/bindings/c/test/fdb_api.hpp @@ -139,6 +139,8 @@ CharsRef toCharsRef(const std::optional>& s) noexcept { [[maybe_unused]] constexpr const bool OverflowCheck = false; +// clang-tidy does not recognize size() through intSize(); call-site suppressions mark +// byte buffers whose lengths are passed explicitly to the C API or KeySelector. inline int intSize(size_t size) { if constexpr (OverflowCheck) { if (size > static_cast(std::numeric_limits::max())) @@ -278,6 +280,7 @@ inline int maxApiVersion() { namespace network { inline Error setOptionNothrow(FDBNetworkOption option, BytesRef str) noexcept { + // NOLINTNEXTLINE(bugprone-suspicious-stringview-data-usage) return Error(native::fdb_network_set_option(option, str.data(), intSize(str.size()))); } @@ -500,18 +503,22 @@ struct KeySelector { namespace key_select { inline KeySelector firstGreaterThan(KeyRef key, int offset = 0) { + // NOLINTNEXTLINE(bugprone-suspicious-stringview-data-usage) return KeySelector{ FDB_KEYSEL_FIRST_GREATER_THAN(key.data(), intSize(key.size())) + offset }; } inline KeySelector firstGreaterOrEqual(KeyRef key, int offset = 0) { + // NOLINTNEXTLINE(bugprone-suspicious-stringview-data-usage) return KeySelector{ FDB_KEYSEL_FIRST_GREATER_OR_EQUAL(key.data(), intSize(key.size())) + offset }; } inline KeySelector lastLessThan(KeyRef key, int offset = 0) { + // NOLINTNEXTLINE(bugprone-suspicious-stringview-data-usage) return KeySelector{ FDB_KEYSEL_LAST_LESS_THAN(key.data(), intSize(key.size())) + offset }; } inline KeySelector lastLessOrEqual(KeyRef key, int offset = 0) { + // NOLINTNEXTLINE(bugprone-suspicious-stringview-data-usage) return KeySelector{ FDB_KEYSEL_LAST_LESS_OR_EQUAL(key.data(), intSize(key.size())) + offset }; } @@ -549,6 +556,7 @@ class Transaction { } Error setOptionNothrow(FDBTransactionOption option, BytesRef str) noexcept { + // NOLINTNEXTLINE(bugprone-suspicious-stringview-data-usage) return Error(native::fdb_transaction_set_option(tr.get(), option, str.data(), intSize(str.size()))); } @@ -600,6 +608,7 @@ class Transaction { } TypedFuture get(KeyRef key, bool snapshot) { + // NOLINTNEXTLINE(bugprone-suspicious-stringview-data-usage) return native::fdb_transaction_get(tr.get(), key.data(), intSize(key.size()), snapshot); } @@ -631,6 +640,7 @@ class Transaction { } TypedFuture watch(KeyRef key) { + // NOLINTNEXTLINE(bugprone-suspicious-stringview-data-usage) return native::fdb_transaction_watch(tr.get(), key.data(), intSize(key.size())); } @@ -643,26 +653,36 @@ class Transaction { void cancel() { return native::fdb_transaction_cancel(tr.get()); } void set(KeyRef key, ValueRef value) { + // NOLINTNEXTLINE(bugprone-suspicious-stringview-data-usage) native::fdb_transaction_set(tr.get(), key.data(), intSize(key.size()), value.data(), intSize(value.size())); } void atomicOp(KeyRef key, ValueRef param, FDBMutationType operationType) { + // NOLINTBEGIN(bugprone-suspicious-stringview-data-usage) native::fdb_transaction_atomic_op( tr.get(), key.data(), intSize(key.size()), param.data(), intSize(param.size()), operationType); + // NOLINTEND(bugprone-suspicious-stringview-data-usage) } - void clear(KeyRef key) { native::fdb_transaction_clear(tr.get(), key.data(), intSize(key.size())); } + void clear(KeyRef key) { + // NOLINTNEXTLINE(bugprone-suspicious-stringview-data-usage) + native::fdb_transaction_clear(tr.get(), key.data(), intSize(key.size())); + } void clearRange(KeyRef begin, KeyRef end) { + // NOLINTBEGIN(bugprone-suspicious-stringview-data-usage) native::fdb_transaction_clear_range( tr.get(), begin.data(), intSize(begin.size()), end.data(), intSize(end.size())); + // NOLINTEND(bugprone-suspicious-stringview-data-usage) } void addConflictRange(KeyRef begin, KeyRef end, FDBConflictRangeType rangeType) { + // NOLINTBEGIN(bugprone-suspicious-stringview-data-usage) if (auto err = Error(native::fdb_transaction_add_conflict_range( tr.get(), begin.data(), intSize(begin.size()), end.data(), intSize(end.size()), rangeType))) { throwError("fdb_transaction_add_conflict_range returned error: ", err); } + // NOLINTEND(bugprone-suspicious-stringview-data-usage) } void addReadConflictRange(KeyRef begin, KeyRef end) { addConflictRange(begin, end, FDB_CONFLICT_RANGE_TYPE_READ); } @@ -708,6 +728,7 @@ class Database : public IDatabaseOps { } Error setOptionNothrow(FDBDatabaseOption option, BytesRef str) noexcept { + // NOLINTNEXTLINE(bugprone-suspicious-stringview-data-usage) return Error(native::fdb_database_set_option(db.get(), option, str.data(), intSize(str.size()))); } From 68ec177be4f9e0adf54ceee950ff055f114e6a38 Mon Sep 17 00:00:00 2001 From: Pierce Lopez Date: Tue, 22 Sep 2026 13:49:45 -0400 Subject: [PATCH 144/170] fix bindingtester2 fdbcli "json status" error handling The contrib/local_cluster/ code that bindingtester2 uses was not showing fdbcli stderr, nor catching timeouts correctly. --- contrib/local_cluster/lib/fdb_process.py | 30 ++++++++++++++++++++---- 1 file changed, 25 insertions(+), 5 deletions(-) diff --git a/contrib/local_cluster/lib/fdb_process.py b/contrib/local_cluster/lib/fdb_process.py index 9d91d31f9ab..33267698814 100644 --- a/contrib/local_cluster/lib/fdb_process.py +++ b/contrib/local_cluster/lib/fdb_process.py @@ -15,7 +15,7 @@ FDBSERVER_TIMEOUT: float = 180.0 -FDBCLI_TIMEOUT: float = 180.0 +FDBCLI_TIMEOUT: float = 30.0 FDBCLI_RETRY_TIME: float = 1.0 @@ -37,7 +37,7 @@ def __init__(self, strerror: str, filename: str, *args, **kwargs): @property def filename(self) -> str: """Name of the file""" - return self._strerror + return self._filename @property def strerror(self) -> str: @@ -207,10 +207,30 @@ async def get_server_status(cluster_file: str) -> Union[Dict, None]: cluster_file=cluster_file, commands="status json" ).run() try: - output = await asyncio.wait_for(fdbcli_process.stdout.read(-1), FDBCLI_TIMEOUT) + stdout, stderr = await asyncio.wait_for( + fdbcli_process.communicate(), FDBCLI_TIMEOUT + ) + except asyncio.TimeoutError: + logger.warning("Timed out waiting for fdbcli [status json]") + fdbcli_process.kill() await fdbcli_process.wait() - return json.loads(output.decode()) - except TimeoutError: + return None + + if fdbcli_process.returncode != 0: + logger.warning( + f"fdbcli [status json] exited with {fdbcli_process.returncode}, " + f"stderr: {stderr.decode(errors='replace').strip()[:1024]!r}" + ) + return None + + try: + return json.loads(stdout.decode()) + except json.JSONDecodeError: + logger.warning( + f"fdbcli [status json] emitted non-JSON output: " + f"{stdout.decode(errors='replace')[:1024]!r}, " + f"stderr: {stderr.decode(errors='replace').strip()[:1024]!r}" + ) return None From 4c5705f0f8b278b4a05f588e9b9a5e9637658c0f Mon Sep 17 00:00:00 2001 From: Pierce Lopez Date: Tue, 22 Sep 2026 13:51:35 -0400 Subject: [PATCH 145/170] fix bindingtester2 joshua start script needs to unset FDB_NETWORK_OPTION_EXTERNAL_CLIENT_DIRECTORY since it may be set in the joshua container image (for joshua-coordinator) --- contrib/Joshua/scripts/binding_test_start.sh | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/contrib/Joshua/scripts/binding_test_start.sh b/contrib/Joshua/scripts/binding_test_start.sh index be4f84a3d8c..2ecb837fb6c 100755 --- a/contrib/Joshua/scripts/binding_test_start.sh +++ b/contrib/Joshua/scripts/binding_test_start.sh @@ -3,5 +3,12 @@ set -e set -o pipefail +# The Joshua agent image sets these to point at an external client directory that +# may hold libfdb_c versions too old for the binaries under test, which makes the +# multi-version client fail to load (api_function_missing, 2204). The tests here +# use the client shipped in the package, so drop them (see also bindingTest.sh). +unset FDB_NETWORK_OPTION_EXTERNAL_CLIENT_DIRECTORY +unset FDB_NETWORK_OPTION_EXTERNAL_CLIENT_LIBRARY + # It is necessary to tee to output.log in case timeout happens python3 ./binding_test.py --stop-at-failure 10 --fdbserver-path $(pwd)/fdbserver --fdbcli-path $(pwd)/fdbcli --libfdb-path $(pwd) --num-ops 1000 --num-hca-ops 100 --concurrency 5 --test-timeout 60 --random 2>&1 | tee output.log From b695453aa4f9174f266c8503a488e80dc401609c Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 22 Sep 2026 15:44:30 -0700 Subject: [PATCH 146/170] Use live team health for cluster replication severity --- .../ClusterHealthIFactor.cpp | 32 +++++-- .../ClusterHealthMonitorTesting.cpp | 78 ++++++++++++---- .../datadistributor/DDTeamCollection.cpp | 90 +++++++++++++++++++ fdbserver/datadistributor/DDTeamCollection.h | 2 + .../datadistributor/DataDistribution.cpp | 1 + tests/fast/DDPipelineSaturation.toml | 1 + 6 files changed, 179 insertions(+), 25 deletions(-) diff --git a/fdbserver/clustercontroller/ClusterHealthIFactor.cpp b/fdbserver/clustercontroller/ClusterHealthIFactor.cpp index de593105438..8609e472b73 100644 --- a/fdbserver/clustercontroller/ClusterHealthIFactor.cpp +++ b/fdbserver/clustercontroller/ClusterHealthIFactor.cpp @@ -21,6 +21,7 @@ #include #include +#include "fdbserver/core/Knobs.h" #include "fdbserver/core/RecoveryState.h" #include "flow/Trace.h" @@ -134,6 +135,15 @@ std::string_view StorageReplicationFactor::getName() const { Future StorageReplicationFactor::fetchLevel(Reference workerEventProvider, TrackCodeProbes trackCodeProbes) { + auto teamEventsAndErrors = co_await workerEventProvider->getLatestDataDistributorEvents("TotalDataInFlight"); + if (!teamEventsAndErrors.present()) { + co_return Level::METRICS_MISSING; + } + WorkerEvents teamEvents = filterEmptyEvents(teamEventsAndErrors.get().first); + if (teamEvents.empty()) { + co_return Level::METRICS_MISSING; + } + auto eventsAndErrors = co_await workerEventProvider->getLatestDataDistributorEvents("MovingData"); if (!eventsAndErrors.present()) { co_return Level::METRICS_MISSING; @@ -146,9 +156,18 @@ Future StorageReplicationFactor::fetchLevel(Reference StorageReplicationFactor::fetchLevel(Reference 0 || inFlight > 0) { queuedOrInFlightRepairMoves += priorityTeamUnhealthy + priorityTeam2Left + priorityTeam1Left + priorityTeam0Left; } } - if (zeroReplicaTeams > 0) { + if (highestTeamPriority >= SERVER_KNOBS->PRIORITY_TEAM_0_LEFT) { CODE_PROBE(trackCodeProbes, "ClusterHealth StorageReplicationFactor returns OUTAGE"); co_return Level::OUTAGE; } - if (oneReplicaTeams > 0 && workerEventProvider->shouldTreatStorageTeamOneReplicaLeftAsCritical()) { + if (highestTeamPriority >= SERVER_KNOBS->PRIORITY_TEAM_1_LEFT && + workerEventProvider->shouldTreatStorageTeamOneReplicaLeftAsCritical()) { CODE_PROBE(trackCodeProbes, "ClusterHealth StorageReplicationFactor returns CRITICAL_INTERVENTION_REQUIRED"); co_return Level::CRITICAL_INTERVENTION_REQUIRED; } - if (queuedOrInFlightRepairMoves > 0) { + if (highestTeamPriority >= SERVER_KNOBS->PRIORITY_TEAM_UNHEALTHY || queuedOrInFlightRepairMoves > 0) { CODE_PROBE(trackCodeProbes, "ClusterHealth StorageReplicationFactor returns SELF_HEALING"); co_return Level::SELF_HEALING; } diff --git a/fdbserver/clustercontroller/ClusterHealthMonitorTesting.cpp b/fdbserver/clustercontroller/ClusterHealthMonitorTesting.cpp index 296de920996..bf54aedcc21 100644 --- a/fdbserver/clustercontroller/ClusterHealthMonitorTesting.cpp +++ b/fdbserver/clustercontroller/ClusterHealthMonitorTesting.cpp @@ -20,6 +20,7 @@ #include +#include "fdbserver/core/Knobs.h" #include "fdbserver/core/RecoveryState.h" #include "flow/UnitTest.h" @@ -270,43 +271,84 @@ TEST_CASE("/fdbserver/clustercontroller/ClusterHealthMonitor/TLogSpaceFactor") { TEST_CASE("/fdbserver/clustercontroller/ClusterHealthMonitor/StorageReplicationFactor") { StorageReplicationFactor factor; auto provider = makeReference(); - Level level; - + auto teamMetrics = [](int priority) { + TraceEventFields fields; + fields.addField("HighestTeamPriority", std::to_string(priority)); + return fields; + }; + provider->setLatestEvents("TotalDataInFlight", + makeLatestWorkerEvents(teamMetrics(SERVER_KNOBS->PRIORITY_TEAM_HEALTHY))); provider->setLatestEvents("MovingData", makeLatestWorkerEvents(MovingDataMetricsBuilder().build())); - level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); + Level level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); ASSERT_EQ(level, Level::HEALTHY); + // Repair activity remains SelfHealing, even after the teams themselves recover. provider->setLatestEvents( "MovingData", makeLatestWorkerEvents(MovingDataMetricsBuilder().inQueue(1).priorityTeamUnhealthy(1).build())); level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); ASSERT_EQ(level, Level::SELF_HEALING); - provider->setLatestEvents( - "MovingData", makeLatestWorkerEvents(MovingDataMetricsBuilder().inFlight(1).priorityTeam1Left(1).build())); + provider->setLatestEvents("TotalDataInFlight", + makeLatestWorkerEvents(teamMetrics(SERVER_KNOBS->PRIORITY_TEAM_1_LEFT))); + provider->setLatestEvents("MovingData", makeLatestWorkerEvents(MovingDataMetricsBuilder().build())); level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); ASSERT_EQ(level, Level::SELF_HEALING); - provider->setStorageTeamOneReplicaLeftIsCritical(true); - level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); - ASSERT_EQ(level, Level::CRITICAL_INTERVENTION_REQUIRED); + // Admission and completion can change the queue's priority counts without changing team health. + // In particular, a full relocation pipeline can temporarily contain no zero-replica relocations. + for (int priority : { SERVER_KNOBS->PRIORITY_TEAM_0_LEFT, SERVER_KNOBS->PRIORITY_TEAM_1_LEFT }) { + provider->setLatestEvents("TotalDataInFlight", makeLatestWorkerEvents(teamMetrics(priority))); + for (int zeroReplicaMoves : { 2, 0, 1 }) { + provider->setLatestEvents("MovingData", + makeLatestWorkerEvents(MovingDataMetricsBuilder() + .inQueue(900) + .inFlight(100) + .priorityTeam1Left(700) + .priorityTeam0Left(zeroReplicaMoves) + .build())); + level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); + ASSERT_EQ(level, + priority == SERVER_KNOBS->PRIORITY_TEAM_0_LEFT ? Level::OUTAGE + : Level::CRITICAL_INTERVENTION_REQUIRED); + } + provider->setLatestEvents("MovingData", makeLatestWorkerEvents(MovingDataMetricsBuilder().build())); + level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); + ASSERT_EQ(level, + priority == SERVER_KNOBS->PRIORITY_TEAM_0_LEFT ? Level::OUTAGE + : Level::CRITICAL_INTERVENTION_REQUIRED); + } + + // Old severe relocations must not keep a recovered team at Outage or Critical. + provider->setLatestEvents("TotalDataInFlight", + makeLatestWorkerEvents(teamMetrics(SERVER_KNOBS->PRIORITY_TEAM_HEALTHY))); provider->setLatestEvents( "MovingData", makeLatestWorkerEvents(MovingDataMetricsBuilder().inQueue(1).priorityTeam0Left(1).build())); level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); - ASSERT_EQ(level, Level::OUTAGE); + ASSERT_EQ(level, Level::SELF_HEALING); - WorkerEvents staleWorkerEvents; - staleWorkerEvents.emplace(NetworkAddress(IPAddress(0x01010101), 1), - MovingDataMetricsBuilder().inQueue(1).priorityTeam0Left(1).build()); - staleWorkerEvents.emplace(NetworkAddress(IPAddress(0x02020202), 2), MovingDataMetricsBuilder().build()); - provider->setLatestEvents("MovingData", makeLatestWorkerEvents(std::move(staleWorkerEvents))); - provider->setLatestDataDistributorEvents( - "MovingData", - makeLatestWorkerEvents(NetworkAddress(IPAddress(0x02020202), 2), MovingDataMetricsBuilder().build())); + // Both summaries must come from the current DD, not a former DD worker's cached events. + provider->setLatestEvents("TotalDataInFlight", + makeLatestWorkerEvents(teamMetrics(SERVER_KNOBS->PRIORITY_TEAM_0_LEFT))); + provider->setLatestDataDistributorEvents("TotalDataInFlight", + makeLatestWorkerEvents(teamMetrics(SERVER_KNOBS->PRIORITY_TEAM_HEALTHY))); + provider->setLatestDataDistributorEvents("MovingData", makeLatestWorkerEvents(MovingDataMetricsBuilder().build())); level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); ASSERT_EQ(level, Level::HEALTHY); - provider->setLatestEvents("MovingData", LatestWorkerEvents()); + // Uninitialized, old-format, empty, and unavailable team summaries must not fabricate health. + TraceEventFields oldFormat; + oldFormat.addField("HighestPriority", "0"); + for (auto fields : { teamMetrics(-1), oldFormat, TraceEventFields() }) { + provider->setLatestDataDistributorEvents("TotalDataInFlight", makeLatestWorkerEvents(fields)); + level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); + ASSERT_EQ(level, Level::METRICS_MISSING); + } + provider->setLatestDataDistributorEvents("TotalDataInFlight", LatestWorkerEvents()); + level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); + ASSERT_EQ(level, Level::METRICS_MISSING); + provider->setLatestDataDistributorEvents("TotalDataInFlight", + makeLatestWorkerEvents(teamMetrics(SERVER_KNOBS->PRIORITY_TEAM_HEALTHY))); provider->setLatestDataDistributorEvents("MovingData", LatestWorkerEvents()); level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); ASSERT_EQ(level, Level::METRICS_MISSING); diff --git a/fdbserver/datadistributor/DDTeamCollection.cpp b/fdbserver/datadistributor/DDTeamCollection.cpp index 25211170fc7..4e64fc916f5 100644 --- a/fdbserver/datadistributor/DDTeamCollection.cpp +++ b/fdbserver/datadistributor/DDTeamCollection.cpp @@ -3690,6 +3690,7 @@ class DDTeamCollectionImpl { .detail("StorageTeamSize", self->configuration.storageTeamSize) .detail("ZeroHealthy", self->zeroOptimalTeams.get()) .detail("HighestPriority", highestPriority) + .detail("HighestTeamPriority", self->getHighestTeamPriority()) .trackLatest(self->primary ? "TotalDataInFlight" : "TotalDataInFlightRemote"); // This trace event's trackLatest // lifetime is controlled by @@ -4638,6 +4639,31 @@ void DDTeamCollection::resetLocalitySet() { } } +int DDTeamCollection::getHighestTeamPriority() const { + if (teamCollections.empty()) { + return -1; + } + int highestPriority = 0; + for (const auto* collection : teamCollections) { + if (collection == nullptr || !collection->initialFailureReactionDelay.isReady()) { + return -1; + } + // Team health is updated independently of relocation admission and completion. Include both regions + // and conservatively keep counting degraded teams until their trackers are retired. + int collectionPriority = -1; + for (const auto& [priority, count] : collection->priority_teams) { + if (count > 0) { + collectionPriority = std::max(collectionPriority, priority); + } + } + if (collectionPriority < 0) { + return -1; + } + highestPriority = std::max(highestPriority, collectionPriority); + } + return highestPriority; +} + bool DDTeamCollection::satisfiesPolicy(const std::vector>& team, int amount) const { std::vector forcedEntries, resultEntries; if (amount == -1) { @@ -7607,6 +7633,66 @@ class DDTeamCollectionUnitTest { co_await delay(0); } + static Future TeamTracker_HighestTeamPriority() { + auto policy = makeReference(3, "zoneid", makeReference()); + auto primary = testTeamCollection(3, policy, 3); + auto remote = testTeamCollection(3, policy, 3); + ASSERT_EQ(primary->getHighestTeamPriority(), -1); + primary->teamCollections = { primary.get(), remote.get() }; + remote->teamCollections = primary->teamCollections; + remote->primary = false; + ASSERT_EQ(primary->getHighestTeamPriority(), -1); + for (auto* collection : primary->teamCollections) { + collection->initialFailureReactionDelay = Future(Void()); + collection->pipelineFull->set(true); + collection->addTeam({ collection->server_info[UID(1, 0)], + collection->server_info[UID(2, 0)], + collection->server_info[UID(3, 0)] }, + IsInitialTeam::True, + IsRedundantTeam::False, + 0.05); + } + co_await delay(0.01); + ASSERT_EQ(primary->getHighestTeamPriority(), SERVER_KNOBS->PRIORITY_TEAM_HEALTHY); + remote->initialFailureReactionDelay = Never(); + ASSERT_EQ(primary->getHighestTeamPriority(), -1); + remote->initialFailureReactionDelay = Future(Void()); + + auto setFailed = [](DDTeamCollection* collection, UID uid, IsFailed failed) { + collection->server_status.set(uid, + ServerStatus(failed, + IsUndesired::False, + IsWiggling::False, + collection->server_info[uid]->getLastKnownInterface().locality)); + }; + // Real failures must update the aggregate even while the relocation pipeline is full. + setFailed(remote.get(), UID(1, 0), IsFailed::True); + setFailed(remote.get(), UID(2, 0), IsFailed::True); + co_await delay(0.01); + ASSERT_EQ(primary->getHighestTeamPriority(), SERVER_KNOBS->PRIORITY_TEAM_1_LEFT); + for (int id = 1; id <= 3; ++id) { + setFailed(primary.get(), UID(id, 0), IsFailed::True); + } + co_await delay(0.01); + ASSERT_EQ(primary->getHighestTeamPriority(), SERVER_KNOBS->PRIORITY_TEAM_0_LEFT); + ASSERT_EQ(remote->getHighestTeamPriority(), SERVER_KNOBS->PRIORITY_TEAM_0_LEFT); + for (int id = 1; id <= 3; ++id) { + setFailed(primary.get(), UID(id, 0), IsFailed::False); + } + co_await delay(0.01); + // Zero-count entries left behind by recovered teams must not affect the aggregate. + ASSERT_EQ(primary->priority_teams[SERVER_KNOBS->PRIORITY_TEAM_0_LEFT], 0); + ASSERT_EQ(primary->getHighestTeamPriority(), SERVER_KNOBS->PRIORITY_TEAM_1_LEFT); + setFailed(remote.get(), UID(1, 0), IsFailed::False); + setFailed(remote.get(), UID(2, 0), IsFailed::False); + co_await delay(0.01); + ASSERT_EQ(primary->getHighestTeamPriority(), SERVER_KNOBS->PRIORITY_TEAM_HEALTHY); + primary->teamCollections = { primary.get() }; + ASSERT_EQ(primary->getHighestTeamPriority(), SERVER_KNOBS->PRIORITY_TEAM_HEALTHY); + primary->teamCollections.push_back(nullptr); + ASSERT_EQ(primary->getHighestTeamPriority(), -1); + } + static Future TeamTracker_RetriesMergedShardForUndesiredServer() { constexpr double checkTeamDelay = 0.05; @@ -8003,3 +8089,7 @@ TEST_CASE("/DataDistribution/TeamTracker/RetriesMergedShardForUndesiredServer") TEST_CASE("/DataDistribution/TeamTracker/RechecksHealthyZone") { co_await DDTeamCollectionUnitTest::TeamTracker_RechecksHealthyZone(); } + +TEST_CASE("/DataDistribution/TeamTracker/HighestTeamPriority") { + co_await DDTeamCollectionUnitTest::TeamTracker_HighestTeamPriority(); +} diff --git a/fdbserver/datadistributor/DDTeamCollection.h b/fdbserver/datadistributor/DDTeamCollection.h index c0a85957623..6a1e933065d 100644 --- a/fdbserver/datadistributor/DDTeamCollection.h +++ b/fdbserver/datadistributor/DDTeamCollection.h @@ -234,6 +234,8 @@ class DDTeamCollection : public ReferenceCounted { std::vector allServers; int64_t unhealthyServers; std::map priority_teams; + // Across all team collections; -1 until each collection has initialized its team health. + int getHighestTeamPriority() const; std::map> tss_info_by_pair; std::map> server_and_tss_info; // TODO could replace this with an efficient way to do a // read-only concatenation of 2 data structures? diff --git a/fdbserver/datadistributor/DataDistribution.cpp b/fdbserver/datadistributor/DataDistribution.cpp index 7b8da096bda..6f44afe8c62 100644 --- a/fdbserver/datadistributor/DataDistribution.cpp +++ b/fdbserver/datadistributor/DataDistribution.cpp @@ -714,6 +714,7 @@ struct DataDistributor : NonCopyable, ReferenceCounted { .detail("TotalBytes", 0) .detail("UnhealthyServers", 0) .detail("HighestPriority", 0) + .detail("HighestTeamPriority", -1) .trackLatest(self->totalDataInFlightEventHolder->trackingKey); TraceEvent("TotalDataInFlight", self->ddId) .detail("Primary", false) diff --git a/tests/fast/DDPipelineSaturation.toml b/tests/fast/DDPipelineSaturation.toml index 21989dde4cc..cc74b5039a0 100644 --- a/tests/fast/DDPipelineSaturation.toml +++ b/tests/fast/DDPipelineSaturation.toml @@ -20,6 +20,7 @@ storageEngineExcludeTypes = [5] [[knobs]] +cluster_health_metric_enable = true dd_max_pipeline_moves = 20 min_shard_bytes = 10000 shard_bytes_per_sqrt_bytes = 0 From fbb12b33778008803c40389ff8f0ad9c6c460e50 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 22 Sep 2026 16:11:08 -0700 Subject: [PATCH 147/170] Remove team-priority aggregation unit test --- .../datadistributor/DDTeamCollection.cpp | 64 ------------------- 1 file changed, 64 deletions(-) diff --git a/fdbserver/datadistributor/DDTeamCollection.cpp b/fdbserver/datadistributor/DDTeamCollection.cpp index 4e64fc916f5..916417872ee 100644 --- a/fdbserver/datadistributor/DDTeamCollection.cpp +++ b/fdbserver/datadistributor/DDTeamCollection.cpp @@ -7633,66 +7633,6 @@ class DDTeamCollectionUnitTest { co_await delay(0); } - static Future TeamTracker_HighestTeamPriority() { - auto policy = makeReference(3, "zoneid", makeReference()); - auto primary = testTeamCollection(3, policy, 3); - auto remote = testTeamCollection(3, policy, 3); - ASSERT_EQ(primary->getHighestTeamPriority(), -1); - primary->teamCollections = { primary.get(), remote.get() }; - remote->teamCollections = primary->teamCollections; - remote->primary = false; - ASSERT_EQ(primary->getHighestTeamPriority(), -1); - for (auto* collection : primary->teamCollections) { - collection->initialFailureReactionDelay = Future(Void()); - collection->pipelineFull->set(true); - collection->addTeam({ collection->server_info[UID(1, 0)], - collection->server_info[UID(2, 0)], - collection->server_info[UID(3, 0)] }, - IsInitialTeam::True, - IsRedundantTeam::False, - 0.05); - } - co_await delay(0.01); - ASSERT_EQ(primary->getHighestTeamPriority(), SERVER_KNOBS->PRIORITY_TEAM_HEALTHY); - remote->initialFailureReactionDelay = Never(); - ASSERT_EQ(primary->getHighestTeamPriority(), -1); - remote->initialFailureReactionDelay = Future(Void()); - - auto setFailed = [](DDTeamCollection* collection, UID uid, IsFailed failed) { - collection->server_status.set(uid, - ServerStatus(failed, - IsUndesired::False, - IsWiggling::False, - collection->server_info[uid]->getLastKnownInterface().locality)); - }; - // Real failures must update the aggregate even while the relocation pipeline is full. - setFailed(remote.get(), UID(1, 0), IsFailed::True); - setFailed(remote.get(), UID(2, 0), IsFailed::True); - co_await delay(0.01); - ASSERT_EQ(primary->getHighestTeamPriority(), SERVER_KNOBS->PRIORITY_TEAM_1_LEFT); - for (int id = 1; id <= 3; ++id) { - setFailed(primary.get(), UID(id, 0), IsFailed::True); - } - co_await delay(0.01); - ASSERT_EQ(primary->getHighestTeamPriority(), SERVER_KNOBS->PRIORITY_TEAM_0_LEFT); - ASSERT_EQ(remote->getHighestTeamPriority(), SERVER_KNOBS->PRIORITY_TEAM_0_LEFT); - for (int id = 1; id <= 3; ++id) { - setFailed(primary.get(), UID(id, 0), IsFailed::False); - } - co_await delay(0.01); - // Zero-count entries left behind by recovered teams must not affect the aggregate. - ASSERT_EQ(primary->priority_teams[SERVER_KNOBS->PRIORITY_TEAM_0_LEFT], 0); - ASSERT_EQ(primary->getHighestTeamPriority(), SERVER_KNOBS->PRIORITY_TEAM_1_LEFT); - setFailed(remote.get(), UID(1, 0), IsFailed::False); - setFailed(remote.get(), UID(2, 0), IsFailed::False); - co_await delay(0.01); - ASSERT_EQ(primary->getHighestTeamPriority(), SERVER_KNOBS->PRIORITY_TEAM_HEALTHY); - primary->teamCollections = { primary.get() }; - ASSERT_EQ(primary->getHighestTeamPriority(), SERVER_KNOBS->PRIORITY_TEAM_HEALTHY); - primary->teamCollections.push_back(nullptr); - ASSERT_EQ(primary->getHighestTeamPriority(), -1); - } - static Future TeamTracker_RetriesMergedShardForUndesiredServer() { constexpr double checkTeamDelay = 0.05; @@ -8089,7 +8029,3 @@ TEST_CASE("/DataDistribution/TeamTracker/RetriesMergedShardForUndesiredServer") TEST_CASE("/DataDistribution/TeamTracker/RechecksHealthyZone") { co_await DDTeamCollectionUnitTest::TeamTracker_RechecksHealthyZone(); } - -TEST_CASE("/DataDistribution/TeamTracker/HighestTeamPriority") { - co_await DDTeamCollectionUnitTest::TeamTracker_HighestTeamPriority(); -} From d171ebef9751babb9fbbcf0c193a23d36babb1d0 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 22 Sep 2026 16:11:59 -0700 Subject: [PATCH 148/170] Keep cluster health override out of pipeline test configuration --- tests/fast/DDPipelineSaturation.toml | 1 - 1 file changed, 1 deletion(-) diff --git a/tests/fast/DDPipelineSaturation.toml b/tests/fast/DDPipelineSaturation.toml index cc74b5039a0..21989dde4cc 100644 --- a/tests/fast/DDPipelineSaturation.toml +++ b/tests/fast/DDPipelineSaturation.toml @@ -20,7 +20,6 @@ storageEngineExcludeTypes = [5] [[knobs]] -cluster_health_metric_enable = true dd_max_pipeline_moves = 20 min_shard_bytes = 10000 shard_bytes_per_sqrt_bytes = 0 From 5807a46abe93b5e96fe4e0e47703cca429712434 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 22 Sep 2026 16:12:54 -0700 Subject: [PATCH 149/170] Simplify native CDC buffering and retag regression coverage --- design/cdc.md | 6 +- fdbserver/cdcproxy/CDCProxy.cpp | 256 ++++++++++++------ .../include/fdbserver/cdcproxy/CDCProxyTest.h | 75 ----- fdbserver/workloads/CMakeLists.txt | 2 +- fdbserver/workloads/NativeCdcEndToEnd.cpp | 60 ++-- tests/fast/NativeCdcRetaggingMemoryBound.toml | 2 +- 6 files changed, 196 insertions(+), 205 deletions(-) delete mode 100644 fdbserver/cdcproxy/include/fdbserver/cdcproxy/CDCProxyTest.h diff --git a/design/cdc.md b/design/cdc.md index 9202ebae915..0bddd1aa5fd 100644 --- a/design/cdc.md +++ b/design/cdc.md @@ -982,9 +982,9 @@ writes, verifies replay after proxy replacement and transaction-system recovery, and checks shared-tag retention and returning to an old tag. It covers the reader/cleanup contract without a load-driven move policy. `NativeCdcRetaggingMemoryBound` verifies a pending retag with a 4.5 KiB proxy -budget and competing reader reservations, then acknowledges and checks history -finalization and retired cleanup. Neither fixture qualifies simultaneous -mixed-version processes or throughput under balancing. +budget, verifies both streams deliver before acknowledgement, then acknowledges +and checks history finalization and retired cleanup. Neither fixture qualifies +simultaneous mixed-version processes or throughput under balancing. The shared-tag workload forces streams to share routing tags and verifies both range filtering and acknowledgement coordination. In particular, removing one diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index a9880e2387f..470e5643f5c 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -33,7 +33,6 @@ #include "NativeCdcInternal.h" #include "fdbserver/core/NativeCdcMetadata.h" #include "fdbserver/cdcproxy/CDCProxy.h" -#include "fdbserver/cdcproxy/CDCProxyTest.h" #include "fdbserver/core/Knobs.h" #include "fdbserver/core/LogProtocolMessage.h" #include "fdbserver/core/OTELSpanContextMessage.h" @@ -583,14 +582,14 @@ class CDCProxy { CDCCommittedPrefix const& prefix, int64_t preferredBufferedBatch, int64_t hardBufferedBatchLimit); - Future materializeBufferSelection(Reference tag, - Reference cursor, - Version throughVersion, - CDCBufferSelection const& selection, - int64_t rawPeekReservation, - FlowLock::Releaser& reservation, - int64_t bufferLimit, - Future invalidated); + CDCBufferTagPassResult materializeBufferSelection(Reference tag, + Reference cursor, + Version throughVersion, + CDCBufferSelection const& selection, + int64_t rawPeekReservation, + FlowLock::Releaser& reservation, + int64_t bufferLimit, + Future invalidated); Future rotateContendedPeek(); Future bufferTagPass(Reference tag, Version begin, Prefetch prefetch); Future bufferTagCursor(Reference tag, @@ -976,6 +975,7 @@ struct CDCStreamMetadataUpdate { CDCStreamMetadataUpdate reconcileBufferedStreamMetadata(Reference stream, CDCStreamReadState const& metadata) { CDCStreamMetadataUpdate update; + ASSERT(stream->mutations.empty() || stream->mutations.back().version <= stream->metadataReadVersion); stream->minVersion = std::max(stream->minVersion, metadata.minVersion); std::vector previousIntervals; if (metadata.readVersion >= stream->metadataReadVersion) { @@ -1010,30 +1010,14 @@ CDCStreamMetadataUpdate reconcileBufferedStreamMetadata(ReferenceminVersion - 1, interval.end - 1)); } - for (auto buffered = stream->mutations.begin(); buffered != stream->mutations.end();) { - const Version version = buffered->version; - if (!update.historyChanged && version >= stream->minVersion) { - break; - } - const bool sameTag = - !update.historyChanged || - std::any_of(stream->tagIntervals.begin(), stream->tagIntervals.end(), [&](const auto& interval) { - return interval.begin <= version && version < interval.end && - std::any_of(previousIntervals.begin(), previousIntervals.end(), [&](const auto& previous) { - return previous.tag == interval.tag && previous.begin <= version && version < previous.end; - }); - }); - if (version < stream->minVersion || !sameTag) { - update.releasedBytes += estimatedCDCConsumeVersionBytes(*buffered); - buffered = stream->mutations.erase(buffered); - } else { - ++buffered; - } + // Buffered mutations were admitted by an earlier metadata snapshot. A later cutover cannot change their + // routing, and history cleanup only removes acknowledged intervals, so only the acknowledged prefix expires. + while (!stream->mutations.empty() && stream->mutations.front().version < stream->minVersion) { + update.releasedBytes += estimatedCDCConsumeVersionBytes(stream->mutations.front()); + stream->mutations.pop_front(); } stream->bufferedBytes -= update.releasedBytes; ASSERT_GE(stream->bufferedBytes, 0); - // A pre-cutover tag may have speculatively buffered farther than its newly discovered end. No reply can have - // delivered that tail: replies are bounded by the metadata snapshot that established their routing history. stream->bufferedThrough = contiguousStreamBufferedThrough(*stream); stream->initialized = true; return update; @@ -1461,14 +1445,14 @@ Future CDCProxy::rotateContendedPeek() { co_await delay(SERVER_KNOBS->BLOCKING_PEEK_TIMEOUT); } -Future CDCProxy::materializeBufferSelection(Reference tag, - Reference cursor, - Version throughVersion, - CDCBufferSelection const& selection, - int64_t rawPeekReservation, - FlowLock::Releaser& reservation, - int64_t bufferLimit, - Future invalidated) { +CDCBufferTagPassResult CDCProxy::materializeBufferSelection(Reference tag, + Reference cursor, + Version throughVersion, + CDCBufferSelection const& selection, + int64_t rawPeekReservation, + FlowLock::Releaser& reservation, + int64_t bufferLimit, + Future invalidated) { const int64_t materializationReservation = reservation.remaining - rawPeekReservation; ASSERT_GE(materializationReservation, 0); if (selection.selectedBytes <= materializationReservation) { @@ -1476,25 +1460,18 @@ Future CDCProxy::materializeBufferSelection(ReferenceholdExpansion(id, tag->tag, reservation.remaining)) { - co_await delay(0.01); - } - } // Two readers can exhaust the budget with initial reservations and then both wait for an expansion. // Drop this cursor and reservation before reacquiring the full amount in one request. tag->nextPassReservation = rawPeekReservation + selection.selectedBytes; ASSERT_LE(tag->nextPassReservation, bufferLimit); - co_return CDCBufferTagPassResult::RETRY; + return CDCBufferTagPassResult::RETRY; } if (!tag->active) { - co_return CDCBufferTagPassResult::STOP; + return CDCBufferTagPassResult::STOP; } - // A metadata refresh can widen a stream's validated read window. Discard an estimate made before that refresh, - // including when capacity and the refresh become ready together. + // Materialize only while the selected read window is still valid. if (invalidated.isReady()) { - co_return CDCBufferTagPassResult::RETRY; + return CDCBufferTagPassResult::RETRY; } std::unordered_map batches = @@ -1503,11 +1480,10 @@ Future CDCProxy::materializeBufferSelection(Reference CDCProxy::materializeBufferSelection(Reference CDCProxy::bufferTagPass(Reference tag, @@ -1666,7 +1642,7 @@ Future CDCProxy::bufferTagCursor(Reference CDCProxy::consumeReply(Reference stre auto buffered = co_await race(waitForBufferedVersion(stream, begin), delay(SERVER_KNOBS->CDC_PROXY_CONSUME_POLL_TIMEOUT)); if (buffered.index() == 1) { - if (auto test = CDCProxyMaterializationTest::get()) { - test->recordLeaseExpiry(id); - } CODE_PROBE(true, "CDC proxy expires an idle consume lease"); CDCConsumeReply reply; reply.lastConsumedVersion = cursor.lastConsumedVersion; @@ -2523,12 +2496,18 @@ class CDCPrefetchTestCursor final : public IReplayPeekCursor, public ReferenceCo bool fetched = false; bool done = false; bool containsMutation; + int mutationCount; + int consumedMutations = 0; int fetches = 0; Promise fetchStarted; public: - explicit CDCPrefetchTestCursor(Future ready, bool containsMutation = true, Version version = 100) - : ready(ready), messageVersion(version), position(version), containsMutation(containsMutation) { + explicit CDCPrefetchTestCursor(Future ready, + bool containsMutation = true, + Version version = 100, + int mutationCount = 1) + : ready(ready), messageVersion(version), position(version), containsMutation(containsMutation), + mutationCount(mutationCount) { BinaryWriter writer(AssumeVersion(g_network->protocolVersion())); writer << MutationRef(MutationRef::SetValue, "k"_sr, "value"_sr); payload = writer.toValue(); @@ -2545,6 +2524,10 @@ class CDCPrefetchTestCursor final : public IReplayPeekCursor, public ReferenceCo StringRef getMessage() override { return payload; } StringRef getMessageWithTags() override { return payload; } void nextMessage() override { + if (++consumedMutations < mutationCount) { + setProtocolVersion(input.get().protocolVersion()); + return; + } done = true; position = LogMessageVersion(messageVersion + 1); } @@ -2565,20 +2548,23 @@ class CDCPrefetchTestCursor final : public IReplayPeekCursor, public ReferenceCo Version popped() const override { return 0; } Version getMinKnownCommittedVersion() const override { return messageVersion; } int64_t getMaxRetainedReplyCount() const override { return 1; } - void setReplyByteLimit(int limitBytes) override { ASSERT_GT(limitBytes, payload.size()); } + void setReplyByteLimit(int limitBytes) override { ASSERT_GT(limitBytes, int64_t(payload.size()) * mutationCount); } Optional getPrimaryPeekLocation() const override { return {}; } Optional getCurrentPeekLocation() const override { return {}; } Version getMaxKnownVersion() const override { return messageVersion; } Reference cloneNoMore() override { - auto clone = makeReference(Void(), containsMutation, messageVersion); + auto clone = makeReference(Void(), containsMutation, messageVersion, mutationCount); clone->position = position; + clone->consumedMutations = consumedMutations; clone->fetched = fetched; clone->done = done; return clone; } void advanceTo(LogMessageVersion next) override { if (next > position) { - nextMessage(); + consumedMutations = mutationCount; + done = true; + position = LogMessageVersion(messageVersion + 1); } } void addref() override { ReferenceCounted::addref(); } @@ -2908,11 +2894,10 @@ class CDCProxyPrefetchTest { selection.selectedStreamIds.insert(1); selection.selectedBytes = passLimit.preferredBufferedBytes + 1; auto cursor = makeReference(Void()); - auto work = test.proxy.materializeBufferSelection( + const auto result = test.proxy.materializeBufferSelection( test.tag, cursor, 100, selection, passLimit.rawReplyBytes, reservation, limit, Never()); // Waiting for an incremental reservation would deadlock against the other reader's held capacity. - ASSERT(work.isReady()); - ASSERT(work.get() == CDCBufferTagPassResult::RETRY); + ASSERT(result == CDCBufferTagPassResult::RETRY); ASSERT_EQ(test.tag->nextPassReservation, passLimit.rawReplyBytes + selection.selectedBytes); ASSERT_EQ(test.proxy.bufferLock.waiters(), 0); ASSERT_EQ(test.proxy.bufferLock.activePermits(), limit); @@ -2921,19 +2906,112 @@ class CDCProxyPrefetchTest { co_return; } - static Future interrupted(int action) { + static Future competingExpandedReservations() { + CDCProxyPrefetchTest test; + const int64_t limit = SERVER_KNOBS->CDC_PROXY_BUFFER_BYTES; + const int peekBytes = std::min(SERVER_KNOBS->MAXIMUM_PEEK_BYTES, limit / 2); + const auto passLimit = calculateBufferPassLimits(limit, peekBytes, 1).get(); + if (2 * passLimit.reservationBytes > limit) { + co_return; // This topology permits only one initial reader reservation. + } + const MutationRef mutation(MutationRef::SetValue, "k"_sr, "value"_sr); + const int64_t mutationBytes = mutation.expectedSize() + sizeof(MutationRef); + const int mutationCount = + std::max(1, (peekBytes - int64_t(sizeof(VersionedMutationsRef))) / mutationBytes + 1); + const int64_t batchBytes = sizeof(VersionedMutationsRef) + mutationCount * mutationBytes; + BinaryWriter writer(AssumeVersion(g_network->protocolVersion())); + writer << mutation; + if (int64_t(writer.toValue().size()) * mutationCount >= peekBytes || + 2 * batchBytes + passLimit.rawReplyBytes >= 2 * passLimit.reservationBytes) { + co_return; // Tiny budgets cannot fit both expanded batches and the second reader's raw reply. + } + ASSERT_GT(batchBytes, passLimit.preferredBufferedBytes); + // Reserve unrelated capacity so the two initial readers exhaust the effective budget under any knob size. + const int64_t ballastBytes = limit - 2 * passLimit.reservationBytes; + co_await test.proxy.bufferLock.take(TaskPriority::TLogPeekReply, ballastBytes); + FlowLock::Releaser ballast(test.proxy.bufferLock, ballastBytes); + + auto first = test.addStream(1); + auto second = test.addStream(2); + auto secondTag = makeReference(Tag(tagLocalityCDC, 1)); + test.tag->streamIds.erase(2); + second->tagIntervals.front().tag = secondTag->tag; + secondTag->streamIds.insert(2); + test.proxy.tags[secondTag->tag] = secondTag; + first->readAhead.cancel(); + second->readAhead.cancel(); + auto firstDemand = test.proxy.waitForBufferedVersion(first, 100); + auto secondDemand = test.proxy.waitForBufferedVersion(second, 100); + Promise firstReady; + Promise secondReady; + auto firstCursor = makeReference(firstReady.getFuture(), true, 100, mutationCount); + auto secondCursor = makeReference(secondReady.getFuture(), true, 100, mutationCount); + auto firstFetch = firstCursor->onFetchStarted(); + auto secondFetch = secondCursor->onFetchStarted(); + auto firstPass = test.proxy.bufferTagCursor(test.tag, 100, firstCursor, Never(), Prefetch::False); + auto secondPass = test.proxy.bufferTagCursor(secondTag, 100, secondCursor, Never(), Prefetch::False); + co_await timeoutError(firstFetch && secondFetch, 5.0); + ASSERT(!firstPass.isReady() && !secondPass.isReady()); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), limit); + ASSERT_EQ(test.proxy.bufferedBytes, 0); + + firstReady.send(Void()); + ASSERT(co_await timeoutError(firstPass, 5.0) == CDCBufferTagPassResult::RETRY); + ASSERT_EQ(test.tag->nextPassReservation, passLimit.rawReplyBytes + batchBytes); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), ballastBytes + passLimit.reservationBytes); + auto firstRetryCursor = makeReference(Void(), true, 100, mutationCount); + auto firstRetry = test.proxy.bufferTagCursor(test.tag, 100, firstRetryCursor, Never(), Prefetch::False); + ASSERT(!firstRetry.isReady()); + ASSERT_EQ(firstRetryCursor->fetchCount(), 0); + ASSERT_GT(test.proxy.bufferLock.waiters(), 0); + secondReady.send(Void()); + ASSERT(co_await timeoutError(secondPass, 5.0) == CDCBufferTagPassResult::RETRY); + ASSERT_EQ(secondTag->nextPassReservation, passLimit.rawReplyBytes + batchBytes); + ASSERT(co_await timeoutError(firstRetry, 5.0) == CDCBufferTagPassResult::RETRY); + auto secondRetryCursor = makeReference(Void(), true, 100, mutationCount); + ASSERT(co_await timeoutError( + test.proxy.bufferTagCursor(secondTag, 100, secondRetryCursor, Never(), Prefetch::False), 5.0) == + CDCBufferTagPassResult::RETRY); + co_await timeoutError(firstDemand && secondDemand, 5.0); + ASSERT_EQ(firstRetryCursor->fetchCount(), 1); + ASSERT_EQ(secondRetryCursor->fetchCount(), 1); + for (auto stream : { first, second }) { + ASSERT_EQ(stream->bufferedThrough, 100); + ASSERT_EQ(stream->minVersion, 1); + ASSERT_EQ(stream->mutations.size(), 1); + ASSERT_EQ(stream->mutations.front().mutations.size(), mutationCount); + ASSERT_EQ(stream->bufferedBytes, batchBytes); + } + ASSERT_EQ(test.proxy.bufferedBytes, 2 * batchBytes); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), ballastBytes + test.proxy.bufferedBytes); + ASSERT_LE(test.proxy.peakActivePermits, limit); + test.proxy.clearBufferedMutations(first); + test.proxy.clearBufferedMutations(second); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), ballastBytes); + ballast.release(ballast.remaining); + ASSERT_EQ(test.proxy.bufferLock.activePermits(), 0); + co_return; + } + + static Future interrupted(int action, Prefetch prefetch = Prefetch::True) { CDCProxyPrefetchTest test; auto stream = test.addStream(1); + Future demand; + if (!prefetch) { + stream->readAhead.cancel(); + demand = test.proxy.waitForBufferedVersion(stream, 100); + ASSERT_EQ(stream->readDemand, 1); + } Promise ready; Promise generationChanged; auto cursor = makeReference(ready.getFuture()); auto fetchStarted = cursor->onFetchStarted(); - auto work = test.proxy.bufferTagCursor(test.tag, 100, cursor, generationChanged.getFuture(), Prefetch::True); + auto work = test.proxy.bufferTagCursor(test.tag, 100, cursor, generationChanged.getFuture(), prefetch); auto start = co_await race(fetchStarted, work); ASSERT_EQ(start.index(), 0); co_await delay(0); ASSERT_EQ(cursor->fetchCount(), 1); - ASSERT(stream->readAhead.claimedBy(test.tag.getPtr())); + ASSERT_EQ(stream->readAhead.claimedBy(test.tag.getPtr()), bool(prefetch)); if (action == 0) { test.tag->refresh.trigger(); } else if (action == 1) { @@ -2946,15 +3024,20 @@ class CDCProxyPrefetchTest { } else { // No consumer comes back: the one speculative peek expires without granting another credit. ASSERT_EQ(action, 4); + ASSERT(prefetch); } if (action != 3) { - co_await work; + co_await timeoutError(work, SERVER_KNOBS->BLOCKING_PEEK_TIMEOUT + 5.0); } ASSERT_EQ(stream->bufferedThrough, 99); ASSERT(stream->mutations.empty()); ASSERT(!stream->readAhead.claimedBy(test.tag.getPtr())); ASSERT(!stream->readAhead.armedFor(99)); ASSERT_EQ(test.proxy.bufferLock.activePermits(), 0); + if (!prefetch) { + demand.cancel(); + ASSERT_EQ(stream->readDemand, 0); + } co_return; } @@ -3047,12 +3130,12 @@ TEST_CASE("/NativeCDC/PrefetchDeclinesUnavailableCapacity") { TEST_CASE("/NativeCDC/PrefetchDeclinesQueuedCapacity") { return CDCProxyPrefetchTest::capacity(true); } -TEST_CASE("/NativeCDC/PrefetchDeclinesExtraCapacity") { - return CDCProxyPrefetchTest::extraCapacity(); -} TEST_CASE("/NativeCDC/DemandRetriesExpandedReservation") { return CDCProxyPrefetchTest::extraCapacity(); } +TEST_CASE("/NativeCDC/CompetingExpandedReservations") { + return CDCProxyPrefetchTest::competingExpandedReservations(); +} TEST_CASE("/NativeCDC/PrefetchRefreshCancels") { return CDCProxyPrefetchTest::interrupted(0); } @@ -3068,6 +3151,18 @@ TEST_CASE("/NativeCDC/PrefetchActorCancellationReleases") { TEST_CASE("/NativeCDC/PrefetchIdleDeadline") { return CDCProxyPrefetchTest::interrupted(4); } +TEST_CASE("/NativeCDC/DemandRefreshCancels") { + return CDCProxyPrefetchTest::interrupted(0, Prefetch::False); +} +TEST_CASE("/NativeCDC/DemandGenerationChangeCancels") { + return CDCProxyPrefetchTest::interrupted(1, Prefetch::False); +} +TEST_CASE("/NativeCDC/DemandReplacementCancels") { + return CDCProxyPrefetchTest::interrupted(2, Prefetch::False); +} +TEST_CASE("/NativeCDC/DemandActorCancellationReleases") { + return CDCProxyPrefetchTest::interrupted(3, Prefetch::False); +} TEST_CASE("/NativeCDC/PrefetchEmptyDoesNotRetry") { return CDCProxyPrefetchTest::empty(); } @@ -3445,8 +3540,8 @@ TEST_CASE("/NativeCDC/ProxyHistoryReconciliation") { initial.readVersion = 150; initial.tagAssignments = { { 90, oldTag } }; ASSERT(reconcileBufferedStreamMetadata(stream, initial).historyChanged); - stream->tagIntervals.front().bufferedThrough = 240; - stream->bufferedThrough = 240; + stream->tagIntervals.front().bufferedThrough = 150; + stream->bufferedThrough = 150; const auto addBufferedVersion = [&](Version version) { auto& buffered = stream->mutations.emplace_back(); buffered.version = version; @@ -3454,9 +3549,6 @@ TEST_CASE("/NativeCDC/ProxyHistoryReconciliation") { stream->bufferedBytes += estimatedCDCConsumeVersionBytes(buffered); }; addBufferedVersion(149); - addBufferedVersion(200); - addBufferedVersion(220); - const int64_t originalBytes = stream->bufferedBytes; const int64_t preservedBytes = estimatedCDCConsumeVersionBytes(stream->mutations.front()); CDCStreamReadState migrating = initial; @@ -3464,19 +3556,19 @@ TEST_CASE("/NativeCDC/ProxyHistoryReconciliation") { migrating.tagAssignments.emplace_back(200, targetTag); const CDCStreamMetadataUpdate migrated = reconcileBufferedStreamMetadata(stream, migrating); ASSERT(migrated.historyChanged); - ASSERT_EQ(migrated.releasedBytes, originalBytes - preservedBytes); + ASSERT_EQ(migrated.releasedBytes, 0); ASSERT_EQ(stream->bufferedBytes, preservedBytes); ASSERT_EQ(stream->mutations.size(), 1); ASSERT_EQ(stream->mutations.front().version, 149); ASSERT_EQ(stream->tagIntervals.size(), 2); ASSERT_EQ(stream->tagIntervals[0].end, 200); - ASSERT_EQ(stream->tagIntervals[0].bufferedThrough, 199); + ASSERT_EQ(stream->tagIntervals[0].bufferedThrough, 150); ASSERT_EQ(stream->tagIntervals[1].begin, 200); ASSERT_EQ(stream->tagIntervals[1].bufferedThrough, 199); - ASSERT_EQ(stream->bufferedThrough, 199); + ASSERT_EQ(stream->bufferedThrough, 150); - stream->tagIntervals[1].bufferedThrough = 250; - stream->bufferedThrough = 250; + ASSERT(advanceStreamTagBufferedThrough(stream, oldTag, 199)); + ASSERT(!advanceStreamTagBufferedThrough(stream, targetTag, 250)); addBufferedVersion(230); addBufferedVersion(250); CDCStreamReadState finalized = migrating; diff --git a/fdbserver/cdcproxy/include/fdbserver/cdcproxy/CDCProxyTest.h b/fdbserver/cdcproxy/include/fdbserver/cdcproxy/CDCProxyTest.h deleted file mode 100644 index 1fc55f9d6f0..00000000000 --- a/fdbserver/cdcproxy/include/fdbserver/cdcproxy/CDCProxyTest.h +++ /dev/null @@ -1,75 +0,0 @@ -/* - * CDCProxyTest.h - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -#pragma once - -#include -#include "fdbclient/FDBTypes.h" -#include "flow/flow.h" - -// Simulation-only barrier at the point where two real tag readers need to expand their reservations. -class CDCProxyMaterializationTest : public ReferenceCounted { - UID owner; - std::map reservations; - bool released = false; - int expiredLeases = 0; - inline static Reference installed; - -public: - CDCProxyMaterializationTest(UID owner, Tag first, Tag second) : owner(owner) { - ASSERT_NE(first, second); - reservations.emplace(first, 0); - reservations.emplace(second, 0); - } - - static Reference get() { - return g_network->isSimulated() ? installed : Reference(); - } - static void install(Reference test) { - ASSERT(g_network->isSimulated()); - ASSERT(!installed); - installed = test; - } - static void uninstall() { - if (installed) { - installed->released = true; - installed.clear(); - } - } - bool holdExpansion(UID proxy, Tag tag, int64_t reserved) { - if (proxy != owner || released || !reservations.contains(tag)) { - return false; - } - reservations.at(tag) = reserved; - return true; - } - bool bothReadersHeld() const { return reservations.begin()->second > 0 && reservations.rbegin()->second > 0; } - int64_t heldBytes() const { return reservations.begin()->second + reservations.rbegin()->second; } - void release() { - ASSERT(bothReadersHeld()); - released = true; - } - void recordLeaseExpiry(UID proxy) { - if (proxy == owner) { - ++expiredLeases; - } - } - int leaseExpiries() const { return expiredLeases; } -}; diff --git a/fdbserver/workloads/CMakeLists.txt b/fdbserver/workloads/CMakeLists.txt index ad489461eb6..dcd55b71919 100644 --- a/fdbserver/workloads/CMakeLists.txt +++ b/fdbserver/workloads/CMakeLists.txt @@ -13,10 +13,10 @@ target_include_directories(fdbserver_workloads PRIVATE ${CMAKE_CURRENT_SOURCE_DIR} ${CMAKE_SOURCE_DIR}/fdbclient) target_link_libraries(fdbserver_workloads PRIVATE - fdbserver_cdcproxy fdbserver_clustercontroller fdbserver_consistencyscan fdbserver_core + fdbserver_logsystem fdbserver_checkpoint fdbserver_kvstore fdbserver_worker diff --git a/fdbserver/workloads/NativeCdcEndToEnd.cpp b/fdbserver/workloads/NativeCdcEndToEnd.cpp index 4df9405ae9f..e24a96f881a 100644 --- a/fdbserver/workloads/NativeCdcEndToEnd.cpp +++ b/fdbserver/workloads/NativeCdcEndToEnd.cpp @@ -33,7 +33,6 @@ #include "fdbclient/DatabaseContext.h" #include "fdbclient/NativeCdc.h" #include "fdbclient/SystemData.h" -#include "fdbserver/cdcproxy/CDCProxyTest.h" #include "fdbserver/clustercontroller/NativeCdcProxyBalancer.h" #include "fdbserver/core/Knobs.h" #include "fdbserver/core/RecoveryState.h" @@ -43,7 +42,6 @@ #include "fdbserver/tester/workloads.h" #include "fdbrpc/simulator.h" #include "flow/DeterministicRandom.h" -#include "flow/ScopeExit.h" // Exercises native CDC by registering overlapping streams, writing mutations, consuming and acknowledging them, // and checking delivery, retention, assignment publication, failure recovery, and drain behavior. Test options @@ -585,17 +583,20 @@ class NativeCdcEndToEndWorkload : public TestWorkload { co_return committed; } - Future consumeContendedRetag(int index, - Reference ledger, - Version through, - Reference> firstBatches, - Future releaseAcknowledgements) { + Future consumeRetagWithAckPause(int index, + Reference ledger, + Version through, + Reference> firstBatches, + Future releaseAcknowledgements) { bool first = true; while (streams[index].consumer->position().lastConsumedVersion < through) { const CDCConsumeReply reply = co_await streams[index].consumer->consume(); ledger->observe(reply); ledger->verifyThrough(reply.lastConsumedVersion); - if (first && !reply.mutations.empty()) { + if (first) { + if (reply.mutations.empty()) { + continue; + } first = false; firstBatches->set(firstBatches->get() + 1); co_await releaseAcknowledgements; @@ -656,46 +657,20 @@ class NativeCdcEndToEndWorkload : public TestWorkload { ASSERT_LT(before, pending.state.assignment.version); ASSERT_GE(after, pending.state.assignment.version); - auto barrier = makeReference(original.state.proxyId, oldTag, newTag); - CDCProxyMaterializationTest::install(barrier); - ScopeExit removeBarrier([] { CDCProxyMaterializationTest::uninstall(); }); + ASSERT_LT(operationTimeout, SERVER_KNOBS->CDC_PROXY_CONSUME_POLL_TIMEOUT); + const double consumeStarted = now(); Promise releaseAcknowledgements; auto firstBatches = makeReference>(0); std::vector> consumers{ - consumeContendedRetag(0, moving, after, firstBatches, releaseAcknowledgements.getFuture()), - consumeContendedRetag(1, active, after, firstBatches, releaseAcknowledgements.getFuture()) + consumeRetagWithAckPause(0, moving, after, firstBatches, releaseAcknowledgements.getFuture()), + consumeRetagWithAckPause(1, active, after, firstBatches, releaseAcknowledgements.getFuture()) }; - const double deadline = now() + operationTimeout; - while (!barrier->bothReadersHeld()) { - for (const auto& consumer : consumers) { - if (consumer.isReady()) { - consumer.get(); - ASSERT(false); - } - } - ASSERT_LT(now(), deadline); - co_await delay(0.01); - } - auto status = co_await getRetagBufferStatus(cx, original.state.proxyId); - ASSERT_EQ(status.activePermits, barrier->heldBytes()); - ASSERT_EQ(status.activePermits, status.bufferLimit); - ASSERT_EQ(status.bufferedBytes, 0); - ASSERT_EQ(firstBatches->get(), 0); - ASSERT((co_await readRetagSnapshot(cx, 0)).state.pending); - TraceEvent("NativeCdcRetagContendedReaders") - .detail("Cutover", pending.state.assignment.version) - .detail("ActivePermits", status.activePermits) - .detail("BufferLimit", status.bufferLimit); - - const double releasedAt = now(); - barrier->release(); // Both real readers must deliver before either acknowledges, and well before a consume lease can expire. while (firstBatches->get() < 2) { - co_await timeoutError(firstBatches->onChange(), std::max(0.0, releasedAt + operationTimeout - now())); + co_await timeoutError(firstBatches->onChange(), std::max(0.0, consumeStarted + operationTimeout - now())); } - ASSERT_EQ(barrier->leaseExpiries(), 0); - ASSERT_LT(now() - releasedAt, SERVER_KNOBS->CDC_PROXY_CONSUME_POLL_TIMEOUT); - status = co_await getRetagBufferStatus(cx, original.state.proxyId); + ASSERT_LT(now() - consumeStarted, SERVER_KNOBS->CDC_PROXY_CONSUME_POLL_TIMEOUT); + auto status = co_await getRetagBufferStatus(cx, original.state.proxyId); ASSERT_GT(status.bufferedBytes, 0); const int64_t oldBytes = co_await timeoutError(retainedTagBytes(oldTag, before, before), operationTimeout); const int64_t newBytes = @@ -720,14 +695,13 @@ class NativeCdcEndToEndWorkload : public TestWorkload { .detail("PeakActivePermits", status.peakActivePermits); releaseAcknowledgements.send(Void()); co_await timeoutError(waitForAll(consumers), operationTimeout); - ASSERT_EQ(barrier->leaseExpiries(), 0); co_await waitForCanonicalRetag(cx, 0, pending.state.assignment); Reference logs = makeLogSystemConsumerFromServerDBInfo(UID(), dbInfo->get()); co_await timeoutError(logs->waitForPopped(pending.state.assignment.version, oldTag), operationTimeout); co_await timeoutError(logs->waitForPopped(after + 1, newTag), operationTimeout); status = co_await getRetagBufferStatus(cx, original.state.proxyId); ASSERT_EQ(status.bufferedBytes, 0); - CODE_PROBE(true, "Native CDC retagging progresses under competing expanded reservations without lease expiry"); + CODE_PROBE(true, "Native CDC retagging delivers both streams before acknowledgement within a bounded budget"); CODE_PROBE(true, "Native CDC retains both retag histories during an acknowledgement pause then drains"); for (const auto& stream : streams) { co_await removeNativeCdcStreamClient(cx, stream.name); diff --git a/tests/fast/NativeCdcRetaggingMemoryBound.toml b/tests/fast/NativeCdcRetaggingMemoryBound.toml index 1740a8a1df3..db22b44485b 100644 --- a/tests/fast/NativeCdcRetaggingMemoryBound.toml +++ b/tests/fast/NativeCdcRetaggingMemoryBound.toml @@ -14,7 +14,7 @@ native_cdc_retag_cleanup_interval = 0.25 cdc_proxy_buffer_bytes = 4608 maximum_peek_bytes = 1152 cdc_proxy_consume_reply_bytes = 4096 -# Progress has a much shorter deadline, so expiry cannot release a contended reservation. +# Both readers must deliver before acknowledgement, well before a consume lease can expire. cdc_proxy_consume_poll_timeout = 600.0 cdc_proxy_pop_scan_interval = 600.0 From 12d80935a77571cd5100d0e6a6ca750469bf1274 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 22 Sep 2026 16:23:20 -0700 Subject: [PATCH 150/170] Avoid copying cluster health test event fields --- fdbserver/clustercontroller/ClusterHealthMonitorTesting.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fdbserver/clustercontroller/ClusterHealthMonitorTesting.cpp b/fdbserver/clustercontroller/ClusterHealthMonitorTesting.cpp index bf54aedcc21..2d53233dcae 100644 --- a/fdbserver/clustercontroller/ClusterHealthMonitorTesting.cpp +++ b/fdbserver/clustercontroller/ClusterHealthMonitorTesting.cpp @@ -339,7 +339,7 @@ TEST_CASE("/fdbserver/clustercontroller/ClusterHealthMonitor/StorageReplicationF // Uninitialized, old-format, empty, and unavailable team summaries must not fabricate health. TraceEventFields oldFormat; oldFormat.addField("HighestPriority", "0"); - for (auto fields : { teamMetrics(-1), oldFormat, TraceEventFields() }) { + for (const auto& fields : { teamMetrics(-1), oldFormat, TraceEventFields() }) { provider->setLatestDataDistributorEvents("TotalDataInFlight", makeLatestWorkerEvents(fields)); level = co_await factor.fetchLevel(provider, TrackCodeProbes::False); ASSERT_EQ(level, Level::METRICS_MISSING); From 6015c91441f641e3636a6cf9fedbc696d636e167 Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 22 Sep 2026 16:42:03 -0700 Subject: [PATCH 151/170] Fix range-loop copy in CDC buffering test --- fdbserver/cdcproxy/CDCProxy.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fdbserver/cdcproxy/CDCProxy.cpp b/fdbserver/cdcproxy/CDCProxy.cpp index 470e5643f5c..0ac3cb132c2 100644 --- a/fdbserver/cdcproxy/CDCProxy.cpp +++ b/fdbserver/cdcproxy/CDCProxy.cpp @@ -2975,7 +2975,7 @@ class CDCProxyPrefetchTest { co_await timeoutError(firstDemand && secondDemand, 5.0); ASSERT_EQ(firstRetryCursor->fetchCount(), 1); ASSERT_EQ(secondRetryCursor->fetchCount(), 1); - for (auto stream : { first, second }) { + for (const auto& stream : { first, second }) { ASSERT_EQ(stream->bufferedThrough, 100); ASSERT_EQ(stream->minVersion, 1); ASSERT_EQ(stream->mutations.size(), 1); From bb794028f2b53f45f4fe965eba82b99ea9a4650f Mon Sep 17 00:00:00 2001 From: Trevor Clinkenbeard Date: Tue, 22 Sep 2026 18:13:11 -0700 Subject: [PATCH 152/170] Buggify native CDC retag cleanup interval --- fdbserver/core/ServerKnobs.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fdbserver/core/ServerKnobs.cpp b/fdbserver/core/ServerKnobs.cpp index b692dd02959..5ab5d42e811 100644 --- a/fdbserver/core/ServerKnobs.cpp +++ b/fdbserver/core/ServerKnobs.cpp @@ -191,7 +191,7 @@ void ServerKnobs::initialize(Randomize randomize, ClientKnobs* clientKnobs, IsSi init( CDC_PROXY_POP_SCAN_INTERVAL, 5.0 ); if( randomize && buggify() ) CDC_PROXY_POP_SCAN_INTERVAL = 0.1; init( CDC_PROXY_REBALANCE_ENABLED, false ); if( randomize && buggify() ) CDC_PROXY_REBALANCE_ENABLED = true; init( CDC_PROXY_REBALANCE_INTERVAL, 60.0 ); if( randomize && buggify() ) CDC_PROXY_REBALANCE_INTERVAL = deterministicRandom()->randomInt(10, 121); - init( NATIVE_CDC_RETAG_CLEANUP_INTERVAL, 30.0 ); + init( NATIVE_CDC_RETAG_CLEANUP_INTERVAL, 30.0 ); if( randomize && buggify() ) NATIVE_CDC_RETAG_CLEANUP_INTERVAL = 0.1; init( APPLY_MUTATION_BYTES, 1e6 ); init( BUGGIFY_RECOVER_MEMORY_LIMIT, 1e6 ); init( BUGGIFY_WORKER_REMOVED_MAX_LAG, 30 ); From 6d3feb38490ae6b9833dad008c420d7ee74d7fe5 Mon Sep 17 00:00:00 2001 From: Mark Shabanov Date: Fri, 26 Jun 2026 16:24:39 +0300 Subject: [PATCH 153/170] Fixed unability to run more than one commit proxy in small configurations --- .../clustercontroller/ClusterController.cpp | 16 +- .../clustercontroller/ClusterController.h | 230 +++++++++++------- .../clustercontroller/ClusterRecovery.cpp | 4 +- 3 files changed, 150 insertions(+), 100 deletions(-) diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index 22f49fe07fb..f6f84456d07 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -325,7 +325,7 @@ Future recruitFailedLogRouters(ClusterControllerData* cluster, !db->recoveryData->remoteDcIds.empty() ? db->recoveryData->remoteDcIds[0] : Optional(); // Use getWorkersForRoleInDatacenter to get workers for all log routers at once - std::map>, int> id_used; + ClusterControllerData::WorkerUsages id_used; cluster->updateKnownIds(&id_used); std::vector workers = @@ -1006,7 +1006,7 @@ void checkOutstandingStorageRequests(ClusterControllerData* self) { // Finds and returns a new process for role WorkerDetails findNewProcessForSingleton(ClusterControllerData* self, const recruitment::ClusterRole role, - std::map>, int>& id_used) { + ClusterControllerData::WorkerUsages& id_used) { // find new process in cluster for role WorkerDetails newWorker = self->getWorkerForRoleInDatacenter( @@ -1019,7 +1019,7 @@ WorkerDetails findNewProcessForSingleton(ClusterControllerData* self, } // acknowledge that the pid is now potentially used by this role as well - id_used[newWorker.interf.locality.processId()]++; + id_used[newWorker.interf.locality.processId()].addRole(role); return newWorker; } @@ -1157,14 +1157,14 @@ void checkBetterSingletons(ClusterControllerData* self) { } // note: this map doesn't consider pids used by existing singletons - std::map>, int> id_used = self->getUsedIds(); + ClusterControllerData::WorkerUsages id_used = self->getUsedIds(); // We prefer spreading out other roles more than separating singletons on their own process // so we artificially amplify the pid count for the processes used by non-singleton roles. // In other words, we make the processes used for other roles less desirable to be used // by singletons as well. for (auto& it : id_used) { - it.second *= PID_USED_AMP_FOR_NON_SINGLETON; + it.second.multiplier *= PID_USED_AMP_FOR_NON_SINGLETON; } // Try to find a new process for each singleton. @@ -2963,7 +2963,7 @@ Future startDataDistributor(ClusterControllerData* self, double waitTime) co_return; } - std::map>, int> idUsed = self->getUsedIds(); + auto idUsed = self->getUsedIds(); WorkerFitnessInfo ddWorker = self->getWorkerForRoleInDatacenter(self->clusterControllerDcId, recruitment::DataDistributor, recruitment::NeverAssign, @@ -3063,7 +3063,7 @@ Future startRatekeeper(ClusterControllerData* self, double waitTime) { co_return; } - std::map>, int> id_used = self->getUsedIds(); + ClusterControllerData::WorkerUsages id_used = self->getUsedIds(); WorkerFitnessInfo rkWorker = self->getWorkerForRoleInDatacenter(self->clusterControllerDcId, recruitment::Ratekeeper, recruitment::NeverAssign, @@ -3158,7 +3158,7 @@ Future startConsistencyScan(ClusterControllerData* self) { co_return; } - std::map>, int> id_used = self->getUsedIds(); + auto id_used = self->getUsedIds(); WorkerFitnessInfo csWorker = self->getWorkerForRoleInDatacenter(self->clusterControllerDcId, recruitment::ConsistencyScan, recruitment::NeverAssign, diff --git a/fdbserver/clustercontroller/ClusterController.h b/fdbserver/clustercontroller/ClusterController.h index 77e0d0bae1c..df7b11da85a 100644 --- a/fdbserver/clustercontroller/ClusterController.h +++ b/fdbserver/clustercontroller/ClusterController.h @@ -291,6 +291,43 @@ class ClusterControllerData { } }; + struct WorkerUsage { + std::bitset roles; + unsigned weight = 0; + unsigned multiplier = 1; + + static unsigned roleWeight(recruitment::ClusterRole role) { + // TO DO: Introduce knobs for different weights of different + // roles + return 1; + } + + void addRole(recruitment::ClusterRole role) { + if (!roles.test(role)) { + roles.set(role); + weight += roleWeight(role); + } + } + + unsigned getWeight() const { return weight * multiplier; } + + unsigned getUniqueHash() const { return roles.to_ulong(); } + + std::string toString() const { + std::string roleCodes; + + for (unsigned r = 0; r < recruitment::NoRole; r++) + if (roles.test((recruitment::ClusterRole)r)) { + if (!roleCodes.empty()) + roleCodes.append(","); + roleCodes.append(Role::get((recruitment::ClusterRole)r).abbreviation); + } + return roleCodes; + } + }; + + using WorkerUsages = std::map>, WorkerUsage>; + bool workerAvailable(WorkerInfo const& worker, bool checkStable) const { return worker.verified && ((now() - startTime < 2 * FLOW_KNOBS->SERVER_REQUEST_INTERVAL) || (IFailureMonitor::failureMonitor() @@ -586,7 +623,7 @@ class ClusterControllerData { // It attempts to evenly recruit processes from across data_halls or datacenters std::vector getWorkersForTlogsComplex(DatabaseConfiguration const& conf, int32_t desired, - std::map>, int>& id_used, + WorkerUsages& id_used, StringRef field, int minFields, int minPerField, @@ -594,7 +631,7 @@ class ClusterControllerData { bool checkStable, const std::set>& dcIds, const std::vector& exclusionWorkerIds) { - std::map, std::vector> fitness_workers; + std::map, std::vector> fitness_workers; // Go through all the workers to list all the workers that can be recruited. for (const auto& [worker_process_id, worker_info] : id_worker) { @@ -661,8 +698,10 @@ class ClusterControllerData { continue; } - fitness_workers[std::make_tuple( - fitness, id_used[worker_process_id], isLongLivedStateless(worker_process_id))] + fitness_workers[std::make_tuple(fitness, + id_used[worker_process_id].getWeight(), + isLongLivedStateless(worker_process_id), + id_used[worker_process_id].getUniqueHash())] .push_back(worker_details); } @@ -771,7 +810,7 @@ class ClusterControllerData { } for (auto& result : resultSet) { - id_used[result.interf.locality.processId()]++; + id_used[result.interf.locality.processId()].addRole(recruitment::TLog); } return std::vector(resultSet.begin(), resultSet.end()); @@ -780,7 +819,7 @@ class ClusterControllerData { // Attempt to recruit TLogs without degraded processes and see if it improves the configuration std::vector getWorkersForTlogsComplex(DatabaseConfiguration const& conf, int32_t desired, - std::map>, int>& id_used, + WorkerUsages& id_used, StringRef field, int minFields, int minPerField, @@ -788,7 +827,7 @@ class ClusterControllerData { const std::set>& dcIds, const std::vector& exclusionWorkerIds) { desired = std::max(desired, minFields * minPerField); - std::map>, int> withDegradedUsed = id_used; + auto withDegradedUsed = id_used; auto withDegraded = getWorkersForTlogsComplex(conf, desired, withDegradedUsed, @@ -816,7 +855,7 @@ class ClusterControllerData { } try { - std::map>, int> withoutDegradedUsed = id_used; + auto withoutDegradedUsed = id_used; auto withoutDegraded = getWorkersForTlogsComplex(conf, desired, withoutDegradedUsed, @@ -850,11 +889,12 @@ class ClusterControllerData { std::vector getWorkersForTlogsSimple(DatabaseConfiguration const& conf, int32_t required, int32_t desired, - std::map>, int>& id_used, + WorkerUsages& id_used, bool checkStable, const std::set>& dcIds, const std::vector& exclusionWorkerIds) { - std::map, std::vector> fitness_workers; + std::map, std::vector> + fitness_workers; // Go through all the workers to list all the workers that can be recruited. for (const auto& [worker_process_id, worker_info] : id_worker) { @@ -924,10 +964,11 @@ class ClusterControllerData { } fitness_workers[std::make_tuple(fitness, - id_used[worker_process_id], + id_used[worker_process_id].getWeight(), worker_details.degraded, isLongLivedStateless(worker_process_id), - inCCDC)] + inCCDC, + id_used[worker_process_id].getUniqueHash())] .push_back(worker_details); } @@ -983,7 +1024,7 @@ class ClusterControllerData { ASSERT(resultSet.size() >= required && resultSet.size() <= desired); for (auto& result : resultSet) { - id_used[result.interf.locality.processId()]++; + id_used[result.interf.locality.processId()].addRole(recruitment::TLog); } return std::vector(resultSet.begin(), resultSet.end()); @@ -1006,11 +1047,12 @@ class ClusterControllerData { int32_t required, int32_t desired, Reference const& policy, - std::map>, int>& id_used, + WorkerUsages& id_used, bool checkStable = false, const std::set>& dcIds = std::set>(), const std::vector& exclusionWorkerIds = {}) { - std::map, std::vector> fitness_workers; + std::map, std::vector> + fitness_workers; std::vector results; Reference logServerSet = makeReference>(); auto* logServerMap = (LocalityMap*)logServerSet.getPtr(); @@ -1085,7 +1127,11 @@ class ClusterControllerData { fitness = std::max(fitness, recruitment::GoodFit); } - fitness_workers[std::make_tuple(fitness, id_used[worker_process_id], worker_details.degraded, inCCDC)] + fitness_workers[std::make_tuple(fitness, + id_used[worker_process_id].getWeight(), + worker_details.degraded, + inCCDC, + id_used[worker_process_id].getUniqueHash())] .push_back(worker_details); } @@ -1135,7 +1181,7 @@ class ClusterControllerData { results.push_back(*object); } for (auto& result : results) { - id_used[result.interf.locality.processId()]++; + id_used[result.interf.locality.processId()].addRole(recruitment::TLog); } return results; } @@ -1209,7 +1255,7 @@ class ClusterControllerData { results.push_back(*object); } for (auto& result : results) { - id_used[result.interf.locality.processId()]++; + id_used[result.interf.locality.processId()].addRole(recruitment::TLog); } return results; } @@ -1234,7 +1280,7 @@ class ClusterControllerData { tLocalities.push_back(object->interf.locality); } for (auto& result : results) { - id_used[result.interf.locality.processId()]++; + id_used[result.interf.locality.processId()].addRole(recruitment::TLog); } TraceEvent("GetTLogTeamDone") .detail("Policy", policy->info()) @@ -1258,7 +1304,7 @@ class ClusterControllerData { int32_t required, int32_t desired, Reference const& policy, - std::map>, int>& id_used, + WorkerUsages& id_used, bool checkStable = false, const std::set>& dcIds = std::set>(), const std::vector& exclusionWorkerIds = {}) { @@ -1270,7 +1316,7 @@ class ClusterControllerData { if (embedded->name() == "Across") { auto* pa2 = (PolicyAcross*)embedded.getPtr(); if (pa2->attributeKey() == "zoneid" && pa2->embeddedPolicyName() == "One") { - std::map>, int> testUsed = id_used; + auto testUsed = id_used; auto workers = getWorkersForTlogsComplex(conf, desired, @@ -1334,7 +1380,7 @@ class ClusterControllerData { useSimple = true; } if (useSimple) { - std::map>, int> testUsed = id_used; + auto testUsed = id_used; auto workers = getWorkersForTlogsSimple(conf, required, desired, id_used, checkStable, dcIds, exclusionWorkerIds); @@ -1377,7 +1423,7 @@ class ClusterControllerData { std::vector getWorkersForSatelliteLogs(const DatabaseConfiguration& conf, const RegionInfo& region, const RegionInfo& remoteRegion, - std::map>, int>& id_used, + WorkerUsages& id_used, bool& satelliteFallback, bool checkStable = false) { int startDC = 0; @@ -1420,7 +1466,7 @@ class ClusterControllerData { // TLogs can be recruited. It does not balance the number of desired TLogs across the satellite and // remote sides. if (remoteDCUsedAsSatellite) { - std::map>, int> tmpIdUsed; + WorkerUsages tmpIdUsed; auto remoteLogs = getWorkersForTlogs(conf, conf.getRemoteTLogReplicationFactor(), conf.getRemoteTLogReplicationFactor(), @@ -1483,10 +1529,11 @@ class ClusterControllerData { recruitment::ClusterRole role, recruitment::Fitness unacceptableFitness, DatabaseConfiguration const& conf, - std::map>, int>& id_used, + WorkerUsages& id_used, std::map>, int> preferredSharing = {}, bool checkStable = false) { - std::map, std::vector> fitness_workers; + std::map, std::vector> + fitness_workers; for (auto& it : id_worker) { auto fitness = recruitment::machineClassFitness(it.second.details.processClass, role); @@ -1498,16 +1545,17 @@ class ClusterControllerData { it.second.details.interf.locality.dcId() == dcId) { auto sharing = preferredSharing.find(it.first); fitness_workers[std::make_tuple(fitness, - id_used[it.first], + id_used[it.first].getWeight(), isLongLivedStateless(it.first), - sharing != preferredSharing.end() ? sharing->second : 1e6)] + sharing != preferredSharing.end() ? sharing->second : 1e6, + id_used[it.first].getUniqueHash())] .push_back(it.second.details); } } if (!fitness_workers.empty()) { auto worker = deterministicRandom()->randomChoice(fitness_workers.begin()->second); - id_used[worker.interf.locality.processId()]++; + id_used[worker.interf.locality.processId()].addRole(role); return WorkerFitnessInfo(worker, std::max(recruitment::GoodFit, std::get<0>(fitness_workers.begin()->first)), std::get<1>(fitness_workers.begin()->first)); @@ -1521,16 +1569,17 @@ class ClusterControllerData { recruitment::ClusterRole role, int amount, DatabaseConfiguration const& conf, - std::map>, int>& id_used, + WorkerUsages& id_used, std::map>, int> preferredSharing = {}, Optional minWorker = Optional(), bool checkStable = false) { struct WorkerFitnessKey { recruitment::Fitness fitness; - int used; + unsigned used; bool unreliableBackup; bool longLivedStateless; int sharing; + unsigned uniqueHash; std::strong_ordering operator<=>(WorkerFitnessKey const&) const = default; }; @@ -1551,17 +1600,18 @@ class ClusterControllerData { !isExcludedDegradedServer(it.second.details.interf.addresses()) && it.second.details.interf.locality.dcId() == dcId && (!minWorker.present() || - (it.second.details.interf.id() != minWorker.get().worker.interf.id() && - (fitness < minWorker.get().fitness || - (fitness == minWorker.get().fitness && id_used[it.first] <= minWorker.get().used))))) { + (it.second.details.interf.id() != minWorker.get().worker.interf.id() && + (fitness < minWorker.get().fitness || + (fitness == minWorker.get().fitness && id_used[it.first].getWeight() <= minWorker.get().used))))) { auto sharing = preferredSharing.find(it.first); fitness_workers[{ fitness, - id_used[it.first], - role == recruitment::Backup && g_network->isSimulated() && + id_used[it.first].getWeight(), + role == recruitment::Backup && g_network->isSimulated() && !g_simulator->getProcessByAddress(it.second.details.interf.address()) ->isReliable(), isLongLivedStateless(it.first), - sharing != preferredSharing.end() ? sharing->second : 1'000'000 }] + sharing != preferredSharing.end() ? sharing->second : 1'000'000, + id_used[it.first].getUniqueHash() }] .push_back(it.second.details); } } @@ -1570,7 +1620,7 @@ class ClusterControllerData { deterministicRandom()->randomShuffle(it.second); for (int i = 0; i < it.second.size(); i++) { results.push_back(it.second[i]); - id_used[it.second[i].interf.locality.processId()]++; + id_used[it.second[i].interf.locality.processId()].addRole(role); if (results.size() == amount) return results; } @@ -1587,7 +1637,7 @@ class ClusterControllerData { recruitment::Fitness worstFit; recruitment::ClusterRole role; int count; - int worstUsed = 1; + unsigned worstUsed = 1; bool degraded = false; RoleFitness(int bestFit, int worstFit, int count, recruitment::ClusterRole role) @@ -1603,7 +1653,7 @@ class ClusterControllerData { RoleFitness(const std::vector& workers, recruitment::ClusterRole role, - const std::map>, int>& id_used) + const WorkerUsages& id_used) : role(role) { // Every recruitment will attempt to recruit the preferred amount through GoodFit, // So a recruitment which only has BestFit is not better than one that has a GoodFit process @@ -1619,7 +1669,7 @@ class ClusterControllerData { TraceEvent(SevError, "UsedNotFound").detail("ProcessId", it.interf.locality.processId().get()); ASSERT(false); } - if (thisUsed->second == 0) { + if ((unsigned)thisUsed->second.getWeight() == 0) { TraceEvent(SevError, "UsedIsZero").detail("ProcessId", it.interf.locality.processId().get()); ASSERT(false); } @@ -1628,9 +1678,9 @@ class ClusterControllerData { if (thisFit > worstFit) { worstFit = thisFit; - worstUsed = thisUsed->second; + worstUsed = thisUsed->second.getWeight(); } else if (thisFit == worstFit) { - worstUsed = std::max(worstUsed, thisUsed->second); + worstUsed = std::max(worstUsed, thisUsed->second.getWeight()); } degraded = degraded || it.degraded; } @@ -1694,15 +1744,15 @@ class ClusterControllerData { return result; } - void updateKnownIds(std::map>, int>* id_used) { - (*id_used)[masterProcessId]++; - (*id_used)[clusterControllerProcessId]++; + void updateKnownIds(WorkerUsages* id_used) { + (*id_used)[masterProcessId].addRole(recruitment::Master); + (*id_used)[clusterControllerProcessId].addRole(recruitment::ClusterController); } RecruitRemoteFromConfigurationReply findRemoteWorkersForConfiguration( RecruitRemoteFromConfigurationRequest const& req) { RecruitRemoteFromConfigurationReply result; - std::map>, int> id_used; + WorkerUsages id_used; updateKnownIds(&id_used); // Primary and satellite TLogs can share the remote DC. Account for their workers when placing log routers so @@ -1710,7 +1760,7 @@ class ClusterControllerData { for (const auto& [processId, worker] : id_worker) { if (std::find(req.exclusionWorkerIds.begin(), req.exclusionWorkerIds.end(), worker.details.interf.id()) != req.exclusionWorkerIds.end()) { - id_used[processId]++; + id_used[processId].addRole(recruitment::TLog); } } @@ -1779,7 +1829,7 @@ class ClusterControllerData { Optional dcId, bool checkGoodRecruitment) { RecruitFromConfigurationReply result; - std::map>, int> id_used; + WorkerUsages id_used; updateKnownIds(&id_used); ASSERT(dcId.present()); @@ -2022,7 +2072,7 @@ class ClusterControllerData { throw no_more_servers(); } else { RecruitFromConfigurationReply result; - std::map>, int> id_used; + WorkerUsages id_used; updateKnownIds(&id_used); auto tlogs = getWorkersForTlogs(req.configuration, req.configuration.tLogReplicationFactor, @@ -2220,17 +2270,18 @@ class ClusterControllerData { } void updateIdUsed(const std::vector& workers, - std::map>, int>& id_used) { + recruitment::ClusterRole role, + WorkerUsages& id_used) { for (auto& it : workers) { - id_used[it.locality.processId()]++; + id_used[it.locality.processId()].addRole(role); } } void compareWorkers(const DatabaseConfiguration& conf, const std::vector& first, - const std::map>, int>& firstUsed, + WorkerUsages& firstUsed, const std::vector& second, - const std::map>, int>& secondUsed, + WorkerUsages& secondUsed, recruitment::ClusterRole role, std::string description) { std::vector firstDetails; @@ -2284,17 +2335,17 @@ class ClusterControllerData { if (!remoteDCUsedAsSatellite) { RecruitFromConfigurationReply compare = findWorkersForConfigurationDispatch(req, false); - std::map>, int> firstUsed; - std::map>, int> secondUsed; + WorkerUsages firstUsed; + WorkerUsages secondUsed; updateKnownIds(&firstUsed); updateKnownIds(&secondUsed); - updateIdUsed(rep.tLogs, firstUsed); - updateIdUsed(compare.tLogs, secondUsed); + updateIdUsed(rep.tLogs, recruitment::TLog, firstUsed); + updateIdUsed(compare.tLogs, recruitment::TLog, secondUsed); compareWorkers( req.configuration, rep.tLogs, firstUsed, compare.tLogs, secondUsed, recruitment::TLog, "TLog"); - updateIdUsed(rep.satelliteTLogs, firstUsed); - updateIdUsed(compare.satelliteTLogs, secondUsed); + updateIdUsed(rep.satelliteTLogs, recruitment::TLog, firstUsed); + updateIdUsed(compare.satelliteTLogs, recruitment::TLog, secondUsed); compareWorkers(req.configuration, rep.satelliteTLogs, firstUsed, @@ -2302,12 +2353,12 @@ class ClusterControllerData { secondUsed, recruitment::TLog, "Satellite"); - updateIdUsed(rep.commitProxies, firstUsed); - updateIdUsed(compare.commitProxies, secondUsed); - updateIdUsed(rep.grvProxies, firstUsed); - updateIdUsed(compare.grvProxies, secondUsed); - updateIdUsed(rep.resolvers, firstUsed); - updateIdUsed(compare.resolvers, secondUsed); + updateIdUsed(rep.commitProxies, recruitment::CommitProxy, firstUsed); + updateIdUsed(compare.commitProxies, recruitment::CommitProxy, secondUsed); + updateIdUsed(rep.grvProxies, recruitment::GrvProxy, firstUsed); + updateIdUsed(compare.grvProxies, recruitment::GrvProxy, secondUsed); + updateIdUsed(rep.resolvers, recruitment::Resolver, firstUsed); + updateIdUsed(compare.resolvers, recruitment::Resolver, secondUsed); compareWorkers(req.configuration, rep.commitProxies, firstUsed, @@ -2329,8 +2380,6 @@ class ClusterControllerData { secondUsed, recruitment::Resolver, "Resolver"); - updateIdUsed(rep.backupWorkers, firstUsed); - updateIdUsed(compare.backupWorkers, secondUsed); compareWorkers(req.configuration, rep.backupWorkers, firstUsed, @@ -2355,7 +2404,7 @@ class ClusterControllerData { } try { - std::map>, int> id_used; + WorkerUsages id_used; getWorkerForRoleInDatacenter( regions[0].dcId, recruitment::ClusterController, recruitment::ExcludeFit, db.config, id_used, {}, true); getWorkerForRoleInDatacenter( @@ -2410,9 +2459,10 @@ class ClusterControllerData { } void updateIdUsed(const std::vector& workers, - std::map>, int>& id_used) { + recruitment::ClusterRole role, + WorkerUsages& id_used) { for (auto& it : workers) { - id_used[it.interf.locality.processId()]++; + id_used[it.interf.locality.processId()].addRole(role); } } @@ -2597,10 +2647,10 @@ class ClusterControllerData { oldMasterFit = std::max(oldMasterFit, recruitment::ExcludeFit); } - std::map>, int> id_used; - std::map>, int> old_id_used; - id_used[clusterControllerProcessId]++; - old_id_used[clusterControllerProcessId]++; + WorkerUsages id_used; + WorkerUsages old_id_used; + id_used[clusterControllerProcessId].addRole(recruitment::ClusterController); + old_id_used[clusterControllerProcessId].addRole(recruitment::ClusterController); WorkerFitnessInfo mworker = getWorkerForRoleInDatacenter( clusterControllerDcId, recruitment::Master, recruitment::NeverAssign, db.config, id_used, {}, true); auto newMasterFit = recruitment::machineClassFitness(mworker.worker.processClass, recruitment::Master); @@ -2608,7 +2658,7 @@ class ClusterControllerData { newMasterFit = std::max(newMasterFit, recruitment::ExcludeFit); } - old_id_used[masterWorker->first]++; + old_id_used[masterWorker->first].addRole(recruitment::Master); if (oldMasterFit < newMasterFit) { TraceEvent("NewRecruitmentIsWorse", id) .detail("OldMasterFit", oldMasterFit) @@ -2648,7 +2698,7 @@ class ClusterControllerData { } // Check tLog fitness - updateIdUsed(tlogs, old_id_used); + updateIdUsed(tlogs, recruitment::TLog, old_id_used); RoleFitness oldTLogFit(tlogs, recruitment::TLog, old_id_used); auto newTLogs = getWorkersForTlogs(db.config, db.config.tLogReplicationFactor, @@ -2673,7 +2723,7 @@ class ClusterControllerData { } } - updateIdUsed(satellite_tlogs, old_id_used); + updateIdUsed(satellite_tlogs, recruitment::TLog, old_id_used); RoleFitness oldSatelliteTLogFit(satellite_tlogs, recruitment::TLog, old_id_used); bool newSatelliteFallback = false; auto newSatelliteTLogs = satellite_tlogs; @@ -2733,7 +2783,7 @@ class ClusterControllerData { return false; } - updateIdUsed(remote_tlogs, old_id_used); + updateIdUsed(remote_tlogs, recruitment::TLog, old_id_used); RoleFitness oldRemoteTLogFit(remote_tlogs, recruitment::TLog, old_id_used); std::vector exclusionWorkerIds; auto fn = [](const WorkerDetails& in) { return in.interf.id(); }; @@ -2757,7 +2807,7 @@ class ClusterControllerData { oldTLogFit.count * std::max(1, db.config.desiredLogRouterCount / std::max(1, oldTLogFit.count)); int newRouterCount = newTLogFit.count * std::max(1, db.config.desiredLogRouterCount / std::max(1, newTLogFit.count)); - updateIdUsed(log_routers, old_id_used); + updateIdUsed(log_routers, recruitment::LogRouter, old_id_used); RoleFitness oldLogRoutersFit(log_routers, recruitment::LogRouter, old_id_used); RoleFitness newLogRoutersFit = oldLogRoutersFit; if (db.config.usableRegions > 1 && dbi.recoveryState == RecoveryState::FULLY_RECOVERED) { @@ -2781,9 +2831,9 @@ class ClusterControllerData { } // Check proxy/grvProxy/resolver fitness - updateIdUsed(commitProxyClasses, old_id_used); - updateIdUsed(grvProxyClasses, old_id_used); - updateIdUsed(resolverClasses, old_id_used); + updateIdUsed(commitProxyClasses, recruitment::CommitProxy, old_id_used); + updateIdUsed(grvProxyClasses, recruitment::GrvProxy, old_id_used); + updateIdUsed(resolverClasses, recruitment::Resolver, old_id_used); RoleFitness oldCommitProxyFit(commitProxyClasses, recruitment::CommitProxy, old_id_used); RoleFitness oldGrvProxyFit(grvProxyClasses, recruitment::GrvProxy, old_id_used); RoleFitness oldResolverFit(resolverClasses, recruitment::Resolver, old_id_used); @@ -2847,7 +2897,7 @@ class ClusterControllerData { RoleFitness newResolverFit(resolvers, recruitment::Resolver, id_used); // Check backup worker fitness - updateIdUsed(backup_workers, old_id_used); + updateIdUsed(backup_workers, recruitment::Backup, old_id_used); RoleFitness oldBackupWorkersFit(backup_workers, recruitment::Backup, old_id_used); const int nBackup = backup_addresses.size(); RoleFitness newBackupWorkersFit(getWorkersForRoleInDatacenter(clusterControllerDcId, @@ -2980,30 +3030,30 @@ class ClusterControllerData { } // Returns a map of for all non-singleton roles - std::map>, int> getUsedIds() { - std::map>, int> idUsed; + WorkerUsages getUsedIds() { + WorkerUsages idUsed; updateKnownIds(&idUsed); auto& dbInfo = db.serverInfo->get(); for (const auto& tlogset : dbInfo.logSystemConfig.tLogs) { for (const auto& tlog : tlogset.tLogs) { if (tlog.present()) { - idUsed[tlog.interf().filteredLocality.processId()]++; + idUsed[tlog.interf().filteredLocality.processId()].addRole(recruitment::TLog); } } } for (const CommitProxyInterface& interf : dbInfo.client.commitProxies) { ASSERT(interf.processId.present()); - idUsed[interf.processId]++; + idUsed[interf.processId].addRole(recruitment::CommitProxy); } for (const GrvProxyInterface& interf : dbInfo.client.grvProxies) { ASSERT(interf.processId.present()); - idUsed[interf.processId]++; + idUsed[interf.processId].addRole(recruitment::GrvProxy); } for (const ResolverInterface& interf : dbInfo.resolvers) { ASSERT(interf.locality.processId().present()); - idUsed[interf.locality.processId()]++; + idUsed[interf.locality.processId()].addRole(recruitment::Resolver); } return idUsed; } diff --git a/fdbserver/clustercontroller/ClusterRecovery.cpp b/fdbserver/clustercontroller/ClusterRecovery.cpp index 6b5034ab198..7c1b16353f2 100644 --- a/fdbserver/clustercontroller/ClusterRecovery.cpp +++ b/fdbserver/clustercontroller/ClusterRecovery.cpp @@ -85,8 +85,8 @@ Future recruitNewMaster(ClusterControllerData* cluster, // We must recruit the master in the same data center as the cluster controller. // This should always be possible, because we can recruit the master on the same process as the cluster // controller. - std::map>, int> id_used; - id_used[cluster->clusterControllerProcessId]++; + ClusterControllerData::WorkerUsages id_used; + id_used[cluster->clusterControllerProcessId].addRole(recruitment::ClusterController); masterWorker = cluster->getWorkerForRoleInDatacenter( cluster->clusterControllerDcId, recruitment::Master, recruitment::NeverAssign, db->config, id_used); if ((recruitment::machineClassFitness(masterWorker.worker.processClass, recruitment::Master) > From c098c1fafb91c777e03f4b824d2fe223095b369e Mon Sep 17 00:00:00 2001 From: Mark Shabanov Date: Tue, 30 Jun 2026 16:10:10 +0300 Subject: [PATCH 154/170] Added proxyColocationOnStateless test --- .../clustercontroller/ClusterController.cpp | 103 ++++++++++++++++++ 1 file changed, 103 insertions(+) diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index f6f84456d07..60a6911d706 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -5207,4 +5207,107 @@ TEST_CASE("/fdbserver/clustercontroller/invalidateExcludedProcessComplaints") { return Void(); } +// Test for the fix described in PR #10411. +// Verifies that in a small cluster (3 stateless, 3 transaction, 3 storage processes), +// the role allocation correctly assigns all 3 commit_proxy roles to the stateless nodes. +// Previously, only 1 commit_proxy would be recruited due to a bug in the candidate +// selection logic within ClusterControllerData::getWorkersForRoleInDatacenter. +// This test ensures that the desired number of proxies (3) is now fully recruited +// when sufficient stateless processes exist. +TEST_CASE("/fdbserver/clustercontroller/proxyColocationOnStateless") { + ClusterControllerData data(ClusterControllerFullInterface(), + LocalityData(), + ServerCoordinators(Reference( + new ClusterConnectionMemoryRecord(ClusterConnectionString()))), + makeReference>>()); + + constexpr int numWorkersPerClass = 3; + + auto makeWorkers = [&](ProcessClass::ClassType classType, StringRef prefix, int count) { + std::vector workers; + for (int i = 0; i < count; i++) { + auto pid = prefix.toString() + std::to_string(i); + WorkerInterface wi; + wi.initEndpoints(); + wi.locality.set(LocalityData::keyZoneId, Standalone(pid + "_zone")); + wi.locality.set(LocalityData::keyProcessId, Standalone(pid)); + data.id_worker[wi.locality.processId()] = + WorkerInfo(Future(), + ReplyPromise(), + 0, + wi, + ProcessClass(classType, ProcessClass::CommandLineSource), + ProcessClass(classType, ProcessClass::CommandLineSource), + ClusterControllerPriorityInfo( + ProcessClass::UnsetFit, false, ClusterControllerPriorityInfo::FitnessUnknown), + false, + true, + Standalone>()); + workers.push_back(wi); + } + return workers; + }; + + auto statelessWorkers = makeWorkers(ProcessClass::StatelessClass, "sl"_sr, numWorkersPerClass); + auto transactionWorkers = makeWorkers(ProcessClass::TransactionClass, "tx"_sr, numWorkersPerClass); + auto storageWorkers = makeWorkers(ProcessClass::StorageClass, "ss"_sr, numWorkersPerClass); + + data.masterProcessId = statelessWorkers[0].locality.processId(); + data.clusterControllerProcessId = statelessWorkers[1].locality.processId(); + data.startTime = 0; + data.gotFullyRecoveredConfig = true; + data.gotProcessClasses = true; + + DatabaseConfiguration config; + config.initialized = true; + config.tLogReplicationFactor = numWorkersPerClass; + config.desiredTLogCount = numWorkersPerClass; + config.commitProxyCount = numWorkersPerClass; + config.grvProxyCount = numWorkersPerClass; + config.resolverCount = numWorkersPerClass; + config.tLogPolicy = makeReference(); + data.db.config = config; + data.db.fullyRecoveredConfig = config; + + // Use findWorkersForConfigurationDispatch — the production code path + // that handles full recruitment: TLogs + two-level proxy/resolver recruiting + RecruitFromConfigurationRequest req; + req.configuration = config; + req.recruitSeedServers = false; + req.maxOldLogRouters = 0; + + auto reply = data.findWorkersForConfigurationDispatch(req, false); + + ASSERT(reply.tLogs.size() == numWorkersPerClass); + ASSERT(reply.commitProxies.size() == numWorkersPerClass); + ASSERT(reply.grvProxies.size() == numWorkersPerClass); + ASSERT(reply.resolvers.size() == numWorkersPerClass); + + // TLogs are on transaction processes + std::set>> txPids; + for (const auto& w : transactionWorkers) { + txPids.insert(w.locality.processId()); + } + for (const auto& tlog : reply.tLogs) { + ASSERT(txPids.contains(tlog.locality.processId())); + } + + // check if proxy/resolvers are on stateless + std::set>> slPids; + for (const auto& w : statelessWorkers) { + slPids.insert(w.locality.processId()); + } + for (const auto& cp : reply.commitProxies) { + ASSERT(slPids.contains(cp.locality.processId())); + } + for (const auto& gp : reply.grvProxies) { + ASSERT(slPids.contains(gp.locality.processId())); + } + for (const auto& rs : reply.resolvers) { + ASSERT(slPids.contains(rs.locality.processId())); + } + + return Void(); +} + } // namespace From a9dcede5fdc3fedaeab172cc685d6ceaeab317c7 Mon Sep 17 00:00:00 2001 From: Mark Shabanov Date: Fri, 31 Jul 2026 17:22:09 +0300 Subject: [PATCH 155/170] drop dead code and fix test timer --- .../clustercontroller/ClusterController.cpp | 2 +- .../clustercontroller/ClusterController.h | 44 +++++++++++-------- 2 files changed, 26 insertions(+), 20 deletions(-) diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index 60a6911d706..7f8b9a8ef6c 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -5254,7 +5254,7 @@ TEST_CASE("/fdbserver/clustercontroller/proxyColocationOnStateless") { data.masterProcessId = statelessWorkers[0].locality.processId(); data.clusterControllerProcessId = statelessWorkers[1].locality.processId(); - data.startTime = 0; + data.startTime = now(); data.gotFullyRecoveredConfig = true; data.gotProcessClasses = true; diff --git a/fdbserver/clustercontroller/ClusterController.h b/fdbserver/clustercontroller/ClusterController.h index df7b11da85a..3014eedcf81 100644 --- a/fdbserver/clustercontroller/ClusterController.h +++ b/fdbserver/clustercontroller/ClusterController.h @@ -316,11 +316,35 @@ class ClusterControllerData { std::string toString() const { std::string roleCodes; + auto appendRoleCode = [&](recruitment::ClusterRole role) { + switch (role) { + case recruitment::Storage: + case recruitment::TLog: + case recruitment::CommitProxy: + case recruitment::GrvProxy: + case recruitment::Master: + case recruitment::Resolver: + case recruitment::LogRouter: + case recruitment::ClusterController: + case recruitment::DataDistributor: + case recruitment::Ratekeeper: + case recruitment::ConsistencyScan: + case recruitment::Backup: + case recruitment::EncryptKeyProxy: + case recruitment::Worker: + roleCodes.append(Role::get(role).abbreviation); + break; + default: + roleCodes.append(format("role%d", role)); + break; + } + }; + for (unsigned r = 0; r < recruitment::NoRole; r++) if (roles.test((recruitment::ClusterRole)r)) { if (!roleCodes.empty()) roleCodes.append(","); - roleCodes.append(Role::get((recruitment::ClusterRole)r).abbreviation); + appendRoleCode((ProcessClass::ClusterRole)r); } return roleCodes; } @@ -1879,13 +1903,6 @@ class ClusterControllerData { dcId, recruitment::Resolver, recruitment::ExcludeFit, req.configuration, id_used, preferredSharing); preferredSharing[first_resolver.worker.interf.locality.processId()] = 2; - // If one of the first process recruitments is forced to share a process, allow all of next recruitments - // to also share a process. - auto maxUsed = std::max({ first_commit_proxy.used, first_grv_proxy.used, first_resolver.used }); - first_commit_proxy.used = maxUsed; - first_grv_proxy.used = maxUsed; - first_resolver.used = maxUsed; - auto commit_proxies = getWorkersForRoleInDatacenter(dcId, recruitment::CommitProxy, req.configuration.getDesiredCommitProxies(), @@ -2136,13 +2153,6 @@ class ClusterControllerData { preferredSharing); preferredSharing[first_resolver.worker.interf.locality.processId()] = 2; - // If one of the first process recruitments is forced to share a process, allow all of next - // recruitments to also share a process. - auto maxUsed = std::max({ first_commit_proxy.used, first_grv_proxy.used, first_resolver.used }); - first_commit_proxy.used = maxUsed; - first_grv_proxy.used = maxUsed; - first_resolver.used = maxUsed; - auto commit_proxies = getWorkersForRoleInDatacenter(dcId, recruitment::CommitProxy, req.configuration.getDesiredCommitProxies(), @@ -2863,10 +2873,6 @@ class ClusterControllerData { preferredSharing, true); preferredSharing[first_resolver.worker.interf.locality.processId()] = 2; - auto maxUsed = std::max({ first_commit_proxy.used, first_grv_proxy.used, first_resolver.used }); - first_commit_proxy.used = maxUsed; - first_grv_proxy.used = maxUsed; - first_resolver.used = maxUsed; auto commit_proxies = getWorkersForRoleInDatacenter(clusterControllerDcId, recruitment::CommitProxy, db.config.getDesiredCommitProxies(), From d96580b490f5c9b469c88ddb866b5209823238bf Mon Sep 17 00:00:00 2001 From: Mark Shabanov Date: Wed, 5 Aug 2026 10:57:43 +0300 Subject: [PATCH 156/170] rebase changes --- .../clustercontroller/ClusterController.cpp | 7 +++++-- fdbserver/clustercontroller/ClusterController.h | 17 +++++++++++------ 2 files changed, 16 insertions(+), 8 deletions(-) diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index 7f8b9a8ef6c..21ef6c66bb4 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -5215,7 +5215,10 @@ TEST_CASE("/fdbserver/clustercontroller/invalidateExcludedProcessComplaints") { // This test ensures that the desired number of proxies (3) is now fully recruited // when sufficient stateless processes exist. TEST_CASE("/fdbserver/clustercontroller/proxyColocationOnStateless") { - ClusterControllerData data(ClusterControllerFullInterface(), + ClusterControllerFullInterface cci; + cci.initEndpoints(); + + ClusterControllerData data(cci, LocalityData(), ServerCoordinators(Reference( new ClusterConnectionMemoryRecord(ClusterConnectionString()))), @@ -5239,7 +5242,7 @@ TEST_CASE("/fdbserver/clustercontroller/proxyColocationOnStateless") { ProcessClass(classType, ProcessClass::CommandLineSource), ProcessClass(classType, ProcessClass::CommandLineSource), ClusterControllerPriorityInfo( - ProcessClass::UnsetFit, false, ClusterControllerPriorityInfo::FitnessUnknown), + recruitment::UnsetFit, false, ClusterControllerPriorityInfo::FitnessUnknown), false, true, Standalone>()); diff --git a/fdbserver/clustercontroller/ClusterController.h b/fdbserver/clustercontroller/ClusterController.h index 3014eedcf81..1c7057b35c0 100644 --- a/fdbserver/clustercontroller/ClusterController.h +++ b/fdbserver/clustercontroller/ClusterController.h @@ -344,7 +344,7 @@ class ClusterControllerData { if (roles.test((recruitment::ClusterRole)r)) { if (!roleCodes.empty()) roleCodes.append(","); - appendRoleCode((ProcessClass::ClusterRole)r); + appendRoleCode((recruitment::ClusterRole)r); } return roleCodes; } @@ -1597,6 +1597,8 @@ class ClusterControllerData { std::map>, int> preferredSharing = {}, Optional minWorker = Optional(), bool checkStable = false) { + // for avoiding recruiting workers with worse fitness + recruitment::Fitness used_fitness = recruitment::NeverAssign; struct WorkerFitnessKey { recruitment::Fitness fitness; unsigned used; @@ -1611,6 +1613,7 @@ class ClusterControllerData { std::map> fitness_workers; std::vector results; if (minWorker.present()) { + used_fitness = minWorker.get().fitness; results.push_back(minWorker.get().worker); } if (amount <= results.size()) { @@ -1623,14 +1626,11 @@ class ClusterControllerData { !conf.isExcludedServer(it.second.details.interf.addresses(), it.second.details.interf.locality) && !isExcludedDegradedServer(it.second.details.interf.addresses()) && it.second.details.interf.locality.dcId() == dcId && - (!minWorker.present() || - (it.second.details.interf.id() != minWorker.get().worker.interf.id() && - (fitness < minWorker.get().fitness || - (fitness == minWorker.get().fitness && id_used[it.first].getWeight() <= minWorker.get().used))))) { + (!minWorker.present() || (it.second.details.interf.id() != minWorker.get().worker.interf.id()))) { auto sharing = preferredSharing.find(it.first); fitness_workers[{ fitness, id_used[it.first].getWeight(), - role == recruitment::Backup && g_network->isSimulated() && + role == recruitment::Backup && g_network->isSimulated() && !g_simulator->getProcessByAddress(it.second.details.interf.address()) ->isReliable(), isLongLivedStateless(it.first), @@ -1641,6 +1641,11 @@ class ClusterControllerData { } for (auto& it : fitness_workers) { + recruitment::Fitness next_fitness = it.first.fitness; + + if (next_fitness > used_fitness) + break; // do not recruit with a greater fitness + used_fitness = next_fitness; deterministicRandom()->randomShuffle(it.second); for (int i = 0; i < it.second.size(); i++) { results.push_back(it.second[i]); From ea786ca0fc9011dc8dface9b191de2d7aed6bf6d Mon Sep 17 00:00:00 2001 From: Mark Shabanov Date: Thu, 6 Aug 2026 13:54:15 +0300 Subject: [PATCH 157/170] extend proxyColocationOnStateless test with WorkerUsage-based distribution and determinism checks --- .../clustercontroller/ClusterController.cpp | 45 +++++++++++++++++++ .../clustercontroller/ClusterController.h | 4 +- 2 files changed, 46 insertions(+), 3 deletions(-) diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index 21ef6c66bb4..bfe6377c850 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -5310,6 +5310,51 @@ TEST_CASE("/fdbserver/clustercontroller/proxyColocationOnStateless") { ASSERT(slPids.contains(rs.locality.processId())); } + // Verify that proxies and resolvers are distributed across different stateless processes, + // not all colocated on a single process. The WorkerUsage tracking via getWeight() + // should prefer less-loaded processes when all candidates have the same fitness. + { + std::set>> distinctPids; + for (const auto& cp : reply.commitProxies) { + ASSERT(distinctPids.insert(cp.locality.processId()).second); + } + distinctPids.clear(); + for (const auto& gp : reply.grvProxies) { + ASSERT(distinctPids.insert(gp.locality.processId()).second); + } + distinctPids.clear(); + for (const auto& rs : reply.resolvers) { + ASSERT(distinctPids.insert(rs.locality.processId()).second); + } + } + + // Verify determinism: a second call with the same inputs must produce the same result. + // The getUniqueHash() from WorkerUsage ensures stable ordering of candidates + // with equal fitness + { + auto reply2 = data.findWorkersForConfigurationDispatch(req, false); + + ASSERT_EQ(reply.tLogs.size(), reply2.tLogs.size()); + for (int i = 0; i < reply.tLogs.size(); ++i) { + ASSERT(reply.tLogs[i].locality.processId() == reply2.tLogs[i].locality.processId()); + } + + ASSERT_EQ(reply.commitProxies.size(), reply2.commitProxies.size()); + for (int i = 0; i < reply.commitProxies.size(); ++i) { + ASSERT(reply.commitProxies[i].locality.processId() == reply2.commitProxies[i].locality.processId()); + } + + ASSERT_EQ(reply.grvProxies.size(), reply2.grvProxies.size()); + for (int i = 0; i < reply.grvProxies.size(); ++i) { + ASSERT(reply.grvProxies[i].locality.processId() == reply2.grvProxies[i].locality.processId()); + } + + ASSERT_EQ(reply.resolvers.size(), reply2.resolvers.size()); + for (int i = 0; i < reply.resolvers.size(); ++i) { + ASSERT(reply.resolvers[i].locality.processId() == reply2.resolvers[i].locality.processId()); + } + } + return Void(); } diff --git a/fdbserver/clustercontroller/ClusterController.h b/fdbserver/clustercontroller/ClusterController.h index 1c7057b35c0..4f5cd85e5c0 100644 --- a/fdbserver/clustercontroller/ClusterController.h +++ b/fdbserver/clustercontroller/ClusterController.h @@ -2473,9 +2473,7 @@ class ClusterControllerData { } } - void updateIdUsed(const std::vector& workers, - recruitment::ClusterRole role, - WorkerUsages& id_used) { + void updateIdUsed(const std::vector& workers, recruitment::ClusterRole role, WorkerUsages& id_used) { for (auto& it : workers) { id_used[it.interf.locality.processId()].addRole(role); } From a945d416eb9a83cda2829b453745c9efe1011e5d Mon Sep 17 00:00:00 2001 From: Mark Shabanov Date: Fri, 28 Aug 2026 13:10:33 +0300 Subject: [PATCH 158/170] Curly braces have been added to pass the clang-tidy check. --- fdbserver/clustercontroller/ClusterController.h | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/fdbserver/clustercontroller/ClusterController.h b/fdbserver/clustercontroller/ClusterController.h index 4f5cd85e5c0..33bd2fdf1e1 100644 --- a/fdbserver/clustercontroller/ClusterController.h +++ b/fdbserver/clustercontroller/ClusterController.h @@ -340,12 +340,13 @@ class ClusterControllerData { } }; - for (unsigned r = 0; r < recruitment::NoRole; r++) + for (unsigned r = 0; r < recruitment::NoRole; r++) { if (roles.test((recruitment::ClusterRole)r)) { if (!roleCodes.empty()) roleCodes.append(","); appendRoleCode((recruitment::ClusterRole)r); } + } return roleCodes; } }; From 0e7aa04c258cbfd7903a0093587f92597774afa2 Mon Sep 17 00:00:00 2001 From: Mark Shabanov Date: Tue, 1 Sep 2026 13:14:14 +0300 Subject: [PATCH 159/170] Fix recruitment under-recruiting across fitness levels and restore backup worker usage in determinism check. Add regression unit tests. --- .../clustercontroller/ClusterController.cpp | 210 ++++++++++++++++-- .../clustercontroller/ClusterController.h | 30 ++- 2 files changed, 209 insertions(+), 31 deletions(-) diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index bfe6377c850..ef8d312555c 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -5246,6 +5246,7 @@ TEST_CASE("/fdbserver/clustercontroller/proxyColocationOnStateless") { false, true, Standalone>()); + data.id_worker[wi.locality.processId()].verified = true; workers.push_back(wi); } return workers; @@ -5273,13 +5274,15 @@ TEST_CASE("/fdbserver/clustercontroller/proxyColocationOnStateless") { data.db.fullyRecoveredConfig = config; // Use findWorkersForConfigurationDispatch — the production code path - // that handles full recruitment: TLogs + two-level proxy/resolver recruiting + // that handles full recruitment: TLogs + two-level proxy/resolver recruiting. + // checkGoodRecruitment=true also exercises the operation_failed gate: since the + // desired counts are reachable here, recruitment must succeed instead of failing. RecruitFromConfigurationRequest req; req.configuration = config; req.recruitSeedServers = false; req.maxOldLogRouters = 0; - auto reply = data.findWorkersForConfigurationDispatch(req, false); + auto reply = data.findWorkersForConfigurationDispatch(req, true); ASSERT(reply.tLogs.size() == numWorkersPerClass); ASSERT(reply.commitProxies.size() == numWorkersPerClass); @@ -5328,31 +5331,200 @@ TEST_CASE("/fdbserver/clustercontroller/proxyColocationOnStateless") { } } - // Verify determinism: a second call with the same inputs must produce the same result. - // The getUniqueHash() from WorkerUsage ensures stable ordering of candidates - // with equal fitness + // Verify determinism: a second recruitment with the same inputs must place every + // role on the same set of processes. Sets are compared instead of positions because + // the order of candidates within a bucket is randomized. { - auto reply2 = data.findWorkersForConfigurationDispatch(req, false); + auto reply2 = data.findWorkersForConfigurationDispatch(req, true); - ASSERT_EQ(reply.tLogs.size(), reply2.tLogs.size()); - for (int i = 0; i < reply.tLogs.size(); ++i) { - ASSERT(reply.tLogs[i].locality.processId() == reply2.tLogs[i].locality.processId()); - } + auto pidSet = [](const auto& interfaces) { + std::set>> pids; + for (const auto& interf : interfaces) { + pids.insert(interf.locality.processId()); + } + return pids; + }; + + ASSERT(pidSet(reply.tLogs) == pidSet(reply2.tLogs)); + ASSERT(pidSet(reply.commitProxies) == pidSet(reply2.commitProxies)); + ASSERT(pidSet(reply.grvProxies) == pidSet(reply2.grvProxies)); + ASSERT(pidSet(reply.resolvers) == pidSet(reply2.resolvers)); + } + + return Void(); +} + +// Regression test: proxy recruitment must fill across fitness levels up to the +// minWorker fitness ceiling. With 2 dedicated commit_proxy processes (BestFit) and +// 3 stateless processes (GoodFit), all 3 desired commit proxies must be recruited; +// the fill loop must not stop after consuming the BestFit bucket. +TEST_CASE("/fdbserver/clustercontroller/proxyRecruitmentAcrossFitnessLevels") { + ClusterControllerFullInterface cci; + cci.initEndpoints(); + + ClusterControllerData data(cci, + LocalityData(), + ServerCoordinators(Reference( + new ClusterConnectionMemoryRecord(ClusterConnectionString()))), + makeReference>>()); - ASSERT_EQ(reply.commitProxies.size(), reply2.commitProxies.size()); - for (int i = 0; i < reply.commitProxies.size(); ++i) { - ASSERT(reply.commitProxies[i].locality.processId() == reply2.commitProxies[i].locality.processId()); + auto makeWorkers = [&](ProcessClass::ClassType classType, StringRef prefix, int count) { + std::vector workers; + for (int i = 0; i < count; i++) { + auto pid = prefix.toString() + std::to_string(i); + WorkerInterface wi; + wi.initEndpoints(); + wi.locality.set(LocalityData::keyZoneId, Standalone(pid + "_zone")); + wi.locality.set(LocalityData::keyProcessId, Standalone(pid)); + data.id_worker[wi.locality.processId()] = + WorkerInfo(Future(), + ReplyPromise(), + 0, + wi, + ProcessClass(classType, ProcessClass::CommandLineSource), + ProcessClass(classType, ProcessClass::CommandLineSource), + ClusterControllerPriorityInfo( + recruitment::UnsetFit, false, ClusterControllerPriorityInfo::FitnessUnknown), + false, + true, + Standalone>()); + data.id_worker[wi.locality.processId()].verified = true; + workers.push_back(wi); } + return workers; + }; + + auto dedicatedWorkers = makeWorkers(ProcessClass::CommitProxyClass, "cp"_sr, 2); + auto statelessWorkers = makeWorkers(ProcessClass::StatelessClass, "sl"_sr, 3); + auto transactionWorkers = makeWorkers(ProcessClass::TransactionClass, "tx"_sr, 3); + + data.masterProcessId = statelessWorkers[0].locality.processId(); + data.clusterControllerProcessId = statelessWorkers[1].locality.processId(); + data.startTime = now(); + data.gotFullyRecoveredConfig = true; + data.gotProcessClasses = true; + + DatabaseConfiguration config; + config.initialized = true; + config.tLogReplicationFactor = 3; + config.desiredTLogCount = 3; + config.commitProxyCount = 3; + config.grvProxyCount = 3; + config.resolverCount = 3; + config.tLogPolicy = makeReference(); + data.db.config = config; + data.db.fullyRecoveredConfig = config; + + RecruitFromConfigurationRequest req; + req.configuration = config; + req.recruitSeedServers = false; + req.maxOldLogRouters = 0; + + // checkGoodRecruitment=true: before the fix this configuration returned only 2 + // commit proxies, which tripped the good-recruitment gate and surfaced as + // operation_failed instead of a successful recruitment. + auto reply = data.findWorkersForConfigurationDispatch(req, true); + + ASSERT_EQ(3, reply.commitProxies.size()); + ASSERT_EQ(3, reply.grvProxies.size()); + ASSERT_EQ(3, reply.resolvers.size()); - ASSERT_EQ(reply.grvProxies.size(), reply2.grvProxies.size()); - for (int i = 0; i < reply.grvProxies.size(); ++i) { - ASSERT(reply.grvProxies[i].locality.processId() == reply2.grvProxies[i].locality.processId()); + // Commit proxies must span both fitness levels: both dedicated (BestFit) + // processes plus one stateless (GoodFit) process. + std::set>> dedicatedPids; + std::set>> statelessPids; + for (const auto& w : dedicatedWorkers) { + dedicatedPids.insert(w.locality.processId()); + } + for (const auto& w : statelessWorkers) { + statelessPids.insert(w.locality.processId()); + } + int dedicatedCount = 0; + int statelessCount = 0; + for (const auto& cp : reply.commitProxies) { + if (dedicatedPids.contains(cp.locality.processId())) { + dedicatedCount++; + } else { + ASSERT(statelessPids.contains(cp.locality.processId())); + statelessCount++; } + } + ASSERT_EQ(2, dedicatedCount); + ASSERT_EQ(1, statelessCount); + + return Void(); +} + +// Regression test: recruitment without a minWorker (log routers, backup workers) +// is not fitness-gated and must fall back to worse fitness buckets to reach the +// requested amount. +TEST_CASE("/fdbserver/clustercontroller/logRouterRecruitmentAcrossFitnessLevels") { + ClusterControllerFullInterface cci; + cci.initEndpoints(); + + ClusterControllerData data(cci, + LocalityData(), + ServerCoordinators(Reference( + new ClusterConnectionMemoryRecord(ClusterConnectionString()))), + makeReference>>()); - ASSERT_EQ(reply.resolvers.size(), reply2.resolvers.size()); - for (int i = 0; i < reply.resolvers.size(); ++i) { - ASSERT(reply.resolvers[i].locality.processId() == reply2.resolvers[i].locality.processId()); + auto makeWorkers = [&](ProcessClass::ClassType classType, StringRef prefix, int count) { + std::vector workers; + for (int i = 0; i < count; i++) { + auto pid = prefix.toString() + std::to_string(i); + WorkerInterface wi; + wi.initEndpoints(); + wi.locality.set(LocalityData::keyZoneId, Standalone(pid + "_zone")); + wi.locality.set(LocalityData::keyProcessId, Standalone(pid)); + data.id_worker[wi.locality.processId()] = + WorkerInfo(Future(), + ReplyPromise(), + 0, + wi, + ProcessClass(classType, ProcessClass::CommandLineSource), + ProcessClass(classType, ProcessClass::CommandLineSource), + ClusterControllerPriorityInfo( + recruitment::UnsetFit, false, ClusterControllerPriorityInfo::FitnessUnknown), + false, + true, + Standalone>()); + data.id_worker[wi.locality.processId()].verified = true; + workers.push_back(wi); } + return workers; + }; + + // For LogRouter: stateless -> GoodFit, transaction -> OkayFit. + auto statelessWorkers = makeWorkers(ProcessClass::StatelessClass, "sl"_sr, 2); + auto transactionWorkers = makeWorkers(ProcessClass::TransactionClass, "tx"_sr, 2); + + data.masterProcessId = statelessWorkers[0].locality.processId(); + data.clusterControllerProcessId = statelessWorkers[1].locality.processId(); + data.startTime = now(); + + DatabaseConfiguration config; + config.initialized = true; + data.db.config = config; + data.db.fullyRecoveredConfig = config; + + ClusterControllerData::WorkerUsages id_used; + data.updateKnownIds(&id_used); + + auto logRouters = data.getWorkersForRoleInDatacenter( + Optional>(), recruitment::LogRouter, 4, config, id_used); + + ASSERT_EQ(4, logRouters.size()); + + // All four processes must be used: both fitness levels. + std::set>> recruitedPids; + for (const auto& w : logRouters) { + recruitedPids.insert(w.interf.locality.processId()); + } + for (const auto& w : statelessWorkers) { + ASSERT(recruitedPids.contains(w.locality.processId())); + } + for (const auto& w : transactionWorkers) { + ASSERT(recruitedPids.contains(w.locality.processId())); } return Void(); diff --git a/fdbserver/clustercontroller/ClusterController.h b/fdbserver/clustercontroller/ClusterController.h index 33bd2fdf1e1..c0b5b978f32 100644 --- a/fdbserver/clustercontroller/ClusterController.h +++ b/fdbserver/clustercontroller/ClusterController.h @@ -297,8 +297,9 @@ class ClusterControllerData { unsigned multiplier = 1; static unsigned roleWeight(recruitment::ClusterRole role) { - // TO DO: Introduce knobs for different weights of different - // roles + // All roles currently contribute equally to the usage weight. + // Per-role weights can be introduced here if different roles + // should affect load differently. return 1; } @@ -311,7 +312,10 @@ class ClusterControllerData { unsigned getWeight() const { return weight * multiplier; } - unsigned getUniqueHash() const { return roles.to_ulong(); } + unsigned getUniqueHash() const { + static_assert(recruitment::NoRole <= 32, "role bitset must fit in an unsigned"); + return roles.to_ulong(); + } std::string toString() const { std::string roleCodes; @@ -1598,8 +1602,11 @@ class ClusterControllerData { std::map>, int> preferredSharing = {}, Optional minWorker = Optional(), bool checkStable = false) { - // for avoiding recruiting workers with worse fitness - recruitment::Fitness used_fitness = recruitment::NeverAssign; + // Do not recruit workers with worse fitness than the already-accepted minWorker. + // The ceiling is fixed for the whole loop: lowering it to the best bucket + // consumed would stop the fill at the first fitness level and prevent reaching + // `amount` from worse-but-acceptable buckets. With no minWorker there is no gate. + recruitment::Fitness used_fitness = minWorker.present() ? minWorker.get().fitness : recruitment::NeverAssign; struct WorkerFitnessKey { recruitment::Fitness fitness; unsigned used; @@ -1614,7 +1621,6 @@ class ClusterControllerData { std::map> fitness_workers; std::vector results; if (minWorker.present()) { - used_fitness = minWorker.get().fitness; results.push_back(minWorker.get().worker); } if (amount <= results.size()) { @@ -1642,11 +1648,9 @@ class ClusterControllerData { } for (auto& it : fitness_workers) { - recruitment::Fitness next_fitness = it.first.fitness; - - if (next_fitness > used_fitness) + if (it.first.fitness > used_fitness) { break; // do not recruit with a greater fitness - used_fitness = next_fitness; + } deterministicRandom()->randomShuffle(it.second); for (int i = 0; i < it.second.size(); i++) { results.push_back(it.second[i]); @@ -1699,7 +1703,7 @@ class ClusterControllerData { TraceEvent(SevError, "UsedNotFound").detail("ProcessId", it.interf.locality.processId().get()); ASSERT(false); } - if ((unsigned)thisUsed->second.getWeight() == 0) { + if (thisUsed->second.getWeight() == 0) { TraceEvent(SevError, "UsedIsZero").detail("ProcessId", it.interf.locality.processId().get()); ASSERT(false); } @@ -1758,7 +1762,7 @@ class ClusterControllerData { degraded == r.degraded; } - std::string toString() const { return format("%d %d %d %d %d", worstFit, worstUsed, count, degraded, bestFit); } + std::string toString() const { return format("%d %u %d %d %d", worstFit, worstUsed, count, degraded, bestFit); } }; std::set>> getDatacenters(DatabaseConfiguration const& conf, @@ -2396,6 +2400,8 @@ class ClusterControllerData { secondUsed, recruitment::Resolver, "Resolver"); + updateIdUsed(rep.backupWorkers, recruitment::Backup, firstUsed); + updateIdUsed(compare.backupWorkers, recruitment::Backup, secondUsed); compareWorkers(req.configuration, rep.backupWorkers, firstUsed, From b125ce37e8acaf3b39fc8caf64f7e5208d8c1976 Mon Sep 17 00:00:00 2001 From: Mark Shabanov Date: Fri, 4 Sep 2026 23:54:05 +0300 Subject: [PATCH 160/170] Restore snapshot semantics of singleton placement amplification --- .../clustercontroller/ClusterController.cpp | 82 ++++++++++++++++++- .../clustercontroller/ClusterController.h | 3 +- 2 files changed, 80 insertions(+), 5 deletions(-) diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index ef8d312555c..264257cc334 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -1095,6 +1095,15 @@ bool isHealthySingleton(ClusterControllerData* self, } } +// Amplify the accumulated non-singleton usage once, so that each singleton placed +// afterwards still costs one unit: one non-singleton role must outweigh all +// singleton placements combined. +static void amplifyNonSingletonUsage(ClusterControllerData::WorkerUsages& id_used) { + for (auto& it : id_used) { + it.second.weight *= PID_USED_AMP_FOR_NON_SINGLETON; + } +} + // Returns a mapping from pid->pidCount for pids std::map>, int> getColocCounts( const std::vector>>& pids) { @@ -1163,9 +1172,7 @@ void checkBetterSingletons(ClusterControllerData* self) { // so we artificially amplify the pid count for the processes used by non-singleton roles. // In other words, we make the processes used for other roles less desirable to be used // by singletons as well. - for (auto& it : id_used) { - it.second.multiplier *= PID_USED_AMP_FOR_NON_SINGLETON; - } + amplifyNonSingletonUsage(id_used); // Try to find a new process for each singleton. WorkerDetails newRKWorker = findNewProcessForSingleton(self, recruitment::Ratekeeper, id_used); @@ -5530,4 +5537,73 @@ TEST_CASE("/fdbserver/clustercontroller/logRouterRecruitmentAcrossFitnessLevels" return Void(); } +// Regression test for the singleton-placement amplification: PID_USED_AMP_FOR_NON_SINGLETON +// must amplify the accumulated non-singleton usage once (snapshot semantics), so each +// singleton placed afterwards still costs one unit. With a one-unit read-time factor the +// first singleton placement on the lighter process would tie it with the busier process +// and subsequent placements could leak onto the busier process. +TEST_CASE("/fdbserver/clustercontroller/singletonPlacementKeepsSnapshotAmplification") { + ClusterControllerData data(ClusterControllerFullInterface(), + LocalityData(), + ServerCoordinators(Reference( + new ClusterConnectionMemoryRecord(ClusterConnectionString()))), + makeReference>>()); + + auto addWorker = [&](StringRef pid) -> Optional> { + WorkerInterface wi; + wi.initEndpoints(); + wi.locality.set(LocalityData::keyZoneId, Standalone(pid.toString() + "_zone")); + wi.locality.set(LocalityData::keyProcessId, Standalone(pid)); + data.id_worker[wi.locality.processId()] = + WorkerInfo(Future(), + ReplyPromise(), + 0, + wi, + ProcessClass(ProcessClass::StatelessClass, ProcessClass::CommandLineSource), + ProcessClass(ProcessClass::StatelessClass, ProcessClass::CommandLineSource), + ClusterControllerPriorityInfo( + recruitment::UnsetFit, false, ClusterControllerPriorityInfo::FitnessUnknown), + false, + true, + Standalone>()); + data.id_worker[wi.locality.processId()].verified = true; + return wi.locality.processId(); + }; + + // Process A carries one non-singleton role, process B carries two. + auto pidA = addWorker("a"_sr); + auto pidB = addWorker("b"_sr); + // Unverified master process so onMasterIsBetter() never redirects placements. + data.masterProcessId = addWorker("m"_sr); + data.id_worker[data.masterProcessId.get()].verified = false; + data.startTime = now(); + data.db.config.initialized = true; + + ClusterControllerData::WorkerUsages id_used; + id_used[pidA].addRole(recruitment::CommitProxy); + id_used[pidB].addRole(recruitment::CommitProxy); + id_used[pidB].addRole(recruitment::GrvProxy); + + amplifyNonSingletonUsage(id_used); + ASSERT_EQ(PID_USED_AMP_FOR_NON_SINGLETON, id_used[pidA].getWeight()); + ASSERT_EQ(2 * PID_USED_AMP_FOR_NON_SINGLETON, id_used[pidB].getWeight()); + + // All three singletons must land on A (100 -> 101 -> 102 -> 103 < 200), and each + // placement must cost exactly one unit: with a read-time factor A would tie B at 200 + // after the first placement. + const recruitment::ClusterRole singletonRoles[] = { recruitment::Ratekeeper, + recruitment::DataDistributor, + recruitment::ConsistencyScan }; + unsigned expectedWeight = PID_USED_AMP_FOR_NON_SINGLETON; + for (const auto& role : singletonRoles) { + WorkerDetails worker = findNewProcessForSingleton(&data, role, id_used); + ASSERT(worker.interf.locality.processId() == pidA); + expectedWeight += 1; + ASSERT_EQ(expectedWeight, id_used[pidA].getWeight()); + } + ASSERT_EQ(2 * PID_USED_AMP_FOR_NON_SINGLETON, id_used[pidB].getWeight()); + + return Void(); +} + } // namespace diff --git a/fdbserver/clustercontroller/ClusterController.h b/fdbserver/clustercontroller/ClusterController.h index c0b5b978f32..5311adc0dff 100644 --- a/fdbserver/clustercontroller/ClusterController.h +++ b/fdbserver/clustercontroller/ClusterController.h @@ -294,7 +294,6 @@ class ClusterControllerData { struct WorkerUsage { std::bitset roles; unsigned weight = 0; - unsigned multiplier = 1; static unsigned roleWeight(recruitment::ClusterRole role) { // All roles currently contribute equally to the usage weight. @@ -310,7 +309,7 @@ class ClusterControllerData { } } - unsigned getWeight() const { return weight * multiplier; } + unsigned getWeight() const { return weight; } unsigned getUniqueHash() const { static_assert(recruitment::NoRole <= 32, "role bitset must fit in an unsigned"); From 26c63ff0a4b102ebdc704814ef8a34938dcae9bf Mon Sep 17 00:00:00 2001 From: Mark Shabanov Date: Sat, 5 Sep 2026 00:13:40 +0300 Subject: [PATCH 161/170] clang-format fix --- .../clustercontroller/ClusterController.cpp | 27 +++++++++---------- 1 file changed, 13 insertions(+), 14 deletions(-) diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index 264257cc334..36967457dfe 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -5246,8 +5246,8 @@ TEST_CASE("/fdbserver/clustercontroller/proxyColocationOnStateless") { ReplyPromise(), 0, wi, - ProcessClass(classType, ProcessClass::CommandLineSource), - ProcessClass(classType, ProcessClass::CommandLineSource), + ProcessClass(ProcessClass::StatelessClass, ProcessClass::CommandLineSource), + ProcessClass(ProcessClass::StatelessClass, ProcessClass::CommandLineSource), ClusterControllerPriorityInfo( recruitment::UnsetFit, false, ClusterControllerPriorityInfo::FitnessUnknown), false, @@ -5554,18 +5554,17 @@ TEST_CASE("/fdbserver/clustercontroller/singletonPlacementKeepsSnapshotAmplifica wi.initEndpoints(); wi.locality.set(LocalityData::keyZoneId, Standalone(pid.toString() + "_zone")); wi.locality.set(LocalityData::keyProcessId, Standalone(pid)); - data.id_worker[wi.locality.processId()] = - WorkerInfo(Future(), - ReplyPromise(), - 0, - wi, - ProcessClass(ProcessClass::StatelessClass, ProcessClass::CommandLineSource), - ProcessClass(ProcessClass::StatelessClass, ProcessClass::CommandLineSource), - ClusterControllerPriorityInfo( - recruitment::UnsetFit, false, ClusterControllerPriorityInfo::FitnessUnknown), - false, - true, - Standalone>()); + data.id_worker[wi.locality.processId()] = WorkerInfo( + Future(), + ReplyPromise(), + 0, + wi, + ProcessClass(ProcessClass::StatelessClass, ProcessClass::CommandLineSource), + ProcessClass(ProcessClass::StatelessClass, ProcessClass::CommandLineSource), + ClusterControllerPriorityInfo(recruitment::UnsetFit, false, ClusterControllerPriorityInfo::FitnessUnknown), + false, + true, + Standalone>()); data.id_worker[wi.locality.processId()].verified = true; return wi.locality.processId(); }; From 241d74bd15a179e8c3c33e728d104356459bdf9f Mon Sep 17 00:00:00 2001 From: Mark Shabanov Date: Tue, 8 Sep 2026 14:07:49 +0300 Subject: [PATCH 162/170] Revert WorkerUsage --- .../clustercontroller/ClusterController.cpp | 530 ++++++------------ .../clustercontroller/ClusterController.h | 289 ++++------ .../clustercontroller/ClusterRecovery.cpp | 4 +- 3 files changed, 287 insertions(+), 536 deletions(-) diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index 36967457dfe..2a25161468a 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -325,7 +325,7 @@ Future recruitFailedLogRouters(ClusterControllerData* cluster, !db->recoveryData->remoteDcIds.empty() ? db->recoveryData->remoteDcIds[0] : Optional(); // Use getWorkersForRoleInDatacenter to get workers for all log routers at once - ClusterControllerData::WorkerUsages id_used; + std::map>, int> id_used; cluster->updateKnownIds(&id_used); std::vector workers = @@ -1006,7 +1006,7 @@ void checkOutstandingStorageRequests(ClusterControllerData* self) { // Finds and returns a new process for role WorkerDetails findNewProcessForSingleton(ClusterControllerData* self, const recruitment::ClusterRole role, - ClusterControllerData::WorkerUsages& id_used) { + std::map>, int>& id_used) { // find new process in cluster for role WorkerDetails newWorker = self->getWorkerForRoleInDatacenter( @@ -1019,7 +1019,7 @@ WorkerDetails findNewProcessForSingleton(ClusterControllerData* self, } // acknowledge that the pid is now potentially used by this role as well - id_used[newWorker.interf.locality.processId()].addRole(role); + id_used[newWorker.interf.locality.processId()]++; return newWorker; } @@ -1095,15 +1095,6 @@ bool isHealthySingleton(ClusterControllerData* self, } } -// Amplify the accumulated non-singleton usage once, so that each singleton placed -// afterwards still costs one unit: one non-singleton role must outweigh all -// singleton placements combined. -static void amplifyNonSingletonUsage(ClusterControllerData::WorkerUsages& id_used) { - for (auto& it : id_used) { - it.second.weight *= PID_USED_AMP_FOR_NON_SINGLETON; - } -} - // Returns a mapping from pid->pidCount for pids std::map>, int> getColocCounts( const std::vector>>& pids) { @@ -1166,13 +1157,15 @@ void checkBetterSingletons(ClusterControllerData* self) { } // note: this map doesn't consider pids used by existing singletons - ClusterControllerData::WorkerUsages id_used = self->getUsedIds(); + std::map>, int> id_used = self->getUsedIds(); // We prefer spreading out other roles more than separating singletons on their own process // so we artificially amplify the pid count for the processes used by non-singleton roles. // In other words, we make the processes used for other roles less desirable to be used // by singletons as well. - amplifyNonSingletonUsage(id_used); + for (auto& it : id_used) { + it.second *= PID_USED_AMP_FOR_NON_SINGLETON; + } // Try to find a new process for each singleton. WorkerDetails newRKWorker = findNewProcessForSingleton(self, recruitment::Ratekeeper, id_used); @@ -2970,7 +2963,7 @@ Future startDataDistributor(ClusterControllerData* self, double waitTime) co_return; } - auto idUsed = self->getUsedIds(); + std::map>, int> idUsed = self->getUsedIds(); WorkerFitnessInfo ddWorker = self->getWorkerForRoleInDatacenter(self->clusterControllerDcId, recruitment::DataDistributor, recruitment::NeverAssign, @@ -3070,7 +3063,7 @@ Future startRatekeeper(ClusterControllerData* self, double waitTime) { co_return; } - ClusterControllerData::WorkerUsages id_used = self->getUsedIds(); + std::map>, int> id_used = self->getUsedIds(); WorkerFitnessInfo rkWorker = self->getWorkerForRoleInDatacenter(self->clusterControllerDcId, recruitment::Ratekeeper, recruitment::NeverAssign, @@ -3165,7 +3158,7 @@ Future startConsistencyScan(ClusterControllerData* self) { co_return; } - auto id_used = self->getUsedIds(); + std::map>, int> id_used = self->getUsedIds(); WorkerFitnessInfo csWorker = self->getWorkerForRoleInDatacenter(self->clusterControllerDcId, recruitment::ConsistencyScan, recruitment::NeverAssign, @@ -5214,394 +5207,213 @@ TEST_CASE("/fdbserver/clustercontroller/invalidateExcludedProcessComplaints") { return Void(); } -// Test for the fix described in PR #10411. -// Verifies that in a small cluster (3 stateless, 3 transaction, 3 storage processes), -// the role allocation correctly assigns all 3 commit_proxy roles to the stateless nodes. -// Previously, only 1 commit_proxy would be recruited due to a bug in the candidate -// selection logic within ClusterControllerData::getWorkersForRoleInDatacenter. -// This test ensures that the desired number of proxies (3) is now fully recruited -// when sufficient stateless processes exist. -TEST_CASE("/fdbserver/clustercontroller/proxyColocationOnStateless") { - ClusterControllerFullInterface cci; - cci.initEndpoints(); +// Adds `count` verified workers of the given process class to `data.id_worker`, each in its +// own zone within datacenter `dcId`, and returns their interfaces. `classType` is honored so +// that per-role recruitment fitness differs across the workers. +static std::vector addRecruitmentTestWorkers(ClusterControllerData& data, + Key const& dcId, + ProcessClass::ClassType classType, + StringRef prefix, + int count) { + std::vector workers; + for (int i = 0; i < count; i++) { + std::string pid = prefix.toString() + std::to_string(i); + LocalityData locality; + locality.set(LocalityData::keyProcessId, Standalone(pid)); + locality.set(LocalityData::keyZoneId, Standalone(pid + "_zone")); + locality.set(LocalityData::keyDcId, dcId); + WorkerInterface worker(locality); + worker.initEndpoints(); + auto& info = data.id_worker[locality.processId()]; + info.verified = true; + info.details.interf = worker; + info.details.processClass = ProcessClass(classType, ProcessClass::CommandLineSource); + info.details.recoveredDiskFiles = true; + workers.push_back(worker); + } + return workers; +} - ClusterControllerData data(cci, - LocalityData(), - ServerCoordinators(Reference( - new ClusterConnectionMemoryRecord(ClusterConnectionString()))), - makeReference>>()); +static ClusterControllerData makeRecruitmentTestData(Key const& dcId) { + LocalityData locality; + locality.set(LocalityData::keyDcId, dcId); + return ClusterControllerData(ClusterControllerFullInterface(), + locality, + ServerCoordinators(Reference( + new ClusterConnectionMemoryRecord(ClusterConnectionString()))), + makeReference>>()); +} - constexpr int numWorkersPerClass = 3; - - auto makeWorkers = [&](ProcessClass::ClassType classType, StringRef prefix, int count) { - std::vector workers; - for (int i = 0; i < count; i++) { - auto pid = prefix.toString() + std::to_string(i); - WorkerInterface wi; - wi.initEndpoints(); - wi.locality.set(LocalityData::keyZoneId, Standalone(pid + "_zone")); - wi.locality.set(LocalityData::keyProcessId, Standalone(pid)); - data.id_worker[wi.locality.processId()] = - WorkerInfo(Future(), - ReplyPromise(), - 0, - wi, - ProcessClass(ProcessClass::StatelessClass, ProcessClass::CommandLineSource), - ProcessClass(ProcessClass::StatelessClass, ProcessClass::CommandLineSource), - ClusterControllerPriorityInfo( - recruitment::UnsetFit, false, ClusterControllerPriorityInfo::FitnessUnknown), - false, - true, - Standalone>()); - data.id_worker[wi.locality.processId()].verified = true; - workers.push_back(wi); - } - return workers; - }; +// Regression test for the original bug: in a small cluster where the desired commit-proxy +// count equals the number of stateless processes, only one proxy was recruited. The candidate +// filter compared each candidate's usage against the first-selected worker's usage and so +// rejected every equally-fit process that already hosted the master or cluster controller. +// All equally-fit processes must now remain eligible. +TEST_CASE("/fdbserver/clustercontroller/proxyRecruitmentFillsEqualFitnessProcesses") { + const Key dcId = "dc1"_sr; + ClusterControllerData data = makeRecruitmentTestData(dcId); - auto statelessWorkers = makeWorkers(ProcessClass::StatelessClass, "sl"_sr, numWorkersPerClass); - auto transactionWorkers = makeWorkers(ProcessClass::TransactionClass, "tx"_sr, numWorkersPerClass); - auto storageWorkers = makeWorkers(ProcessClass::StorageClass, "ss"_sr, numWorkersPerClass); + constexpr int kCount = 3; + auto stateless = addRecruitmentTestWorkers(data, dcId, ProcessClass::StatelessClass, "sl"_sr, kCount); - data.masterProcessId = statelessWorkers[0].locality.processId(); - data.clusterControllerProcessId = statelessWorkers[1].locality.processId(); - data.startTime = now(); - data.gotFullyRecoveredConfig = true; - data.gotProcessClasses = true; + // Two of the three stateless processes already host the master and cluster controller, + // so they start with higher usage than the third. + data.masterProcessId = stateless[0].locality.processId(); + data.clusterControllerProcessId = stateless[1].locality.processId(); DatabaseConfiguration config; config.initialized = true; - config.tLogReplicationFactor = numWorkersPerClass; - config.desiredTLogCount = numWorkersPerClass; - config.commitProxyCount = numWorkersPerClass; - config.grvProxyCount = numWorkersPerClass; - config.resolverCount = numWorkersPerClass; - config.tLogPolicy = makeReference(); - data.db.config = config; - data.db.fullyRecoveredConfig = config; - - // Use findWorkersForConfigurationDispatch — the production code path - // that handles full recruitment: TLogs + two-level proxy/resolver recruiting. - // checkGoodRecruitment=true also exercises the operation_failed gate: since the - // desired counts are reachable here, recruitment must succeed instead of failing. - RecruitFromConfigurationRequest req; - req.configuration = config; - req.recruitSeedServers = false; - req.maxOldLogRouters = 0; - auto reply = data.findWorkersForConfigurationDispatch(req, true); - - ASSERT(reply.tLogs.size() == numWorkersPerClass); - ASSERT(reply.commitProxies.size() == numWorkersPerClass); - ASSERT(reply.grvProxies.size() == numWorkersPerClass); - ASSERT(reply.resolvers.size() == numWorkersPerClass); - - // TLogs are on transaction processes - std::set>> txPids; - for (const auto& w : transactionWorkers) { - txPids.insert(w.locality.processId()); - } - for (const auto& tlog : reply.tLogs) { - ASSERT(txPids.contains(tlog.locality.processId())); - } - - // check if proxy/resolvers are on stateless - std::set>> slPids; - for (const auto& w : statelessWorkers) { - slPids.insert(w.locality.processId()); - } - for (const auto& cp : reply.commitProxies) { - ASSERT(slPids.contains(cp.locality.processId())); - } - for (const auto& gp : reply.grvProxies) { - ASSERT(slPids.contains(gp.locality.processId())); - } - for (const auto& rs : reply.resolvers) { - ASSERT(slPids.contains(rs.locality.processId())); - } - - // Verify that proxies and resolvers are distributed across different stateless processes, - // not all colocated on a single process. The WorkerUsage tracking via getWeight() - // should prefer less-loaded processes when all candidates have the same fitness. - { - std::set>> distinctPids; - for (const auto& cp : reply.commitProxies) { - ASSERT(distinctPids.insert(cp.locality.processId()).second); - } - distinctPids.clear(); - for (const auto& gp : reply.grvProxies) { - ASSERT(distinctPids.insert(gp.locality.processId()).second); - } - distinctPids.clear(); - for (const auto& rs : reply.resolvers) { - ASSERT(distinctPids.insert(rs.locality.processId()).second); - } - } - - // Verify determinism: a second recruitment with the same inputs must place every - // role on the same set of processes. Sets are compared instead of positions because - // the order of candidates within a bucket is randomized. - { - auto reply2 = data.findWorkersForConfigurationDispatch(req, true); + std::map>, int> id_used; + data.updateKnownIds(&id_used); - auto pidSet = [](const auto& interfaces) { - std::set>> pids; - for (const auto& interf : interfaces) { - pids.insert(interf.locality.processId()); - } - return pids; - }; + auto first = + data.getWorkerForRoleInDatacenter(dcId, recruitment::CommitProxy, recruitment::ExcludeFit, config, id_used); + auto proxies = + data.getWorkersForRoleInDatacenter(dcId, recruitment::CommitProxy, kCount, config, id_used, {}, first); - ASSERT(pidSet(reply.tLogs) == pidSet(reply2.tLogs)); - ASSERT(pidSet(reply.commitProxies) == pidSet(reply2.commitProxies)); - ASSERT(pidSet(reply.grvProxies) == pidSet(reply2.grvProxies)); - ASSERT(pidSet(reply.resolvers) == pidSet(reply2.resolvers)); + ASSERT_EQ(proxies.size(), kCount); + std::set>> pids; + for (const auto& w : proxies) { + pids.insert(w.interf.locality.processId()); } - + ASSERT_EQ(pids.size(), kCount); // every proxy on a distinct process return Void(); } -// Regression test: proxy recruitment must fill across fitness levels up to the -// minWorker fitness ceiling. With 2 dedicated commit_proxy processes (BestFit) and -// 3 stateless processes (GoodFit), all 3 desired commit proxies must be recruited; -// the fill loop must not stop after consuming the BestFit bucket. -TEST_CASE("/fdbserver/clustercontroller/proxyRecruitmentAcrossFitnessLevels") { - ClusterControllerFullInterface cci; - cci.initEndpoints(); +// Recruitment must still span fitness levels: dedicated commit-proxy processes (BestFit) are +// preferred, but stateless processes (GoodFit) fill the remainder so the desired count is +// reached instead of stopping at the best-fit class. +TEST_CASE("/fdbserver/clustercontroller/proxyRecruitmentSpansFitnessLevels") { + const Key dcId = "dc1"_sr; + ClusterControllerData data = makeRecruitmentTestData(dcId); - ClusterControllerData data(cci, - LocalityData(), - ServerCoordinators(Reference( - new ClusterConnectionMemoryRecord(ClusterConnectionString()))), - makeReference>>()); - - auto makeWorkers = [&](ProcessClass::ClassType classType, StringRef prefix, int count) { - std::vector workers; - for (int i = 0; i < count; i++) { - auto pid = prefix.toString() + std::to_string(i); - WorkerInterface wi; - wi.initEndpoints(); - wi.locality.set(LocalityData::keyZoneId, Standalone(pid + "_zone")); - wi.locality.set(LocalityData::keyProcessId, Standalone(pid)); - data.id_worker[wi.locality.processId()] = - WorkerInfo(Future(), - ReplyPromise(), - 0, - wi, - ProcessClass(classType, ProcessClass::CommandLineSource), - ProcessClass(classType, ProcessClass::CommandLineSource), - ClusterControllerPriorityInfo( - recruitment::UnsetFit, false, ClusterControllerPriorityInfo::FitnessUnknown), - false, - true, - Standalone>()); - data.id_worker[wi.locality.processId()].verified = true; - workers.push_back(wi); - } - return workers; - }; - - auto dedicatedWorkers = makeWorkers(ProcessClass::CommitProxyClass, "cp"_sr, 2); - auto statelessWorkers = makeWorkers(ProcessClass::StatelessClass, "sl"_sr, 3); - auto transactionWorkers = makeWorkers(ProcessClass::TransactionClass, "tx"_sr, 3); - - data.masterProcessId = statelessWorkers[0].locality.processId(); - data.clusterControllerProcessId = statelessWorkers[1].locality.processId(); - data.startTime = now(); - data.gotFullyRecoveredConfig = true; - data.gotProcessClasses = true; + auto dedicated = addRecruitmentTestWorkers(data, dcId, ProcessClass::CommitProxyClass, "cp"_sr, 2); + auto stateless = addRecruitmentTestWorkers(data, dcId, ProcessClass::StatelessClass, "sl"_sr, 3); DatabaseConfiguration config; config.initialized = true; - config.tLogReplicationFactor = 3; - config.desiredTLogCount = 3; - config.commitProxyCount = 3; - config.grvProxyCount = 3; - config.resolverCount = 3; - config.tLogPolicy = makeReference(); - data.db.config = config; - data.db.fullyRecoveredConfig = config; - RecruitFromConfigurationRequest req; - req.configuration = config; - req.recruitSeedServers = false; - req.maxOldLogRouters = 0; - - // checkGoodRecruitment=true: before the fix this configuration returned only 2 - // commit proxies, which tripped the good-recruitment gate and surfaced as - // operation_failed instead of a successful recruitment. - auto reply = data.findWorkersForConfigurationDispatch(req, true); - - ASSERT_EQ(3, reply.commitProxies.size()); - ASSERT_EQ(3, reply.grvProxies.size()); - ASSERT_EQ(3, reply.resolvers.size()); - - // Commit proxies must span both fitness levels: both dedicated (BestFit) - // processes plus one stateless (GoodFit) process. - std::set>> dedicatedPids; - std::set>> statelessPids; - for (const auto& w : dedicatedWorkers) { + std::map>, int> id_used; + auto first = + data.getWorkerForRoleInDatacenter(dcId, recruitment::CommitProxy, recruitment::ExcludeFit, config, id_used); + auto proxies = data.getWorkersForRoleInDatacenter(dcId, recruitment::CommitProxy, 3, config, id_used, {}, first); + + ASSERT_EQ(proxies.size(), 3); + std::set>> dedicatedPids, statelessPids; + for (const auto& w : dedicated) { dedicatedPids.insert(w.locality.processId()); } - for (const auto& w : statelessWorkers) { + for (const auto& w : stateless) { statelessPids.insert(w.locality.processId()); } - int dedicatedCount = 0; - int statelessCount = 0; - for (const auto& cp : reply.commitProxies) { - if (dedicatedPids.contains(cp.locality.processId())) { + int dedicatedCount = 0, statelessCount = 0; + std::set>> usedPids; + for (const auto& w : proxies) { + auto pid = w.interf.locality.processId(); + usedPids.insert(pid); + if (dedicatedPids.count(pid) == 1) { dedicatedCount++; - } else { - ASSERT(statelessPids.contains(cp.locality.processId())); + } else if (statelessPids.count(pid) == 1) { statelessCount++; } } - ASSERT_EQ(2, dedicatedCount); - ASSERT_EQ(1, statelessCount); - + ASSERT_EQ(usedPids.size(), 3); // all on distinct processes + ASSERT_EQ(dedicatedCount, 2); // both dedicated processes used first + ASSERT_EQ(statelessCount, 1); // remainder filled from stateless return Void(); } -// Regression test: recruitment without a minWorker (log routers, backup workers) -// is not fitness-gated and must fall back to worse fitness buckets to reach the -// requested amount. -TEST_CASE("/fdbserver/clustercontroller/logRouterRecruitmentAcrossFitnessLevels") { - ClusterControllerFullInterface cci; - cci.initEndpoints(); +// Without a minWorker there is no fitness ceiling, so recruitment fills across fitness levels +// from the best available. This locks in that the fix (which gates on the minWorker's fitness) +// does not change the no-minWorker path used by log-router recruitment. +TEST_CASE("/fdbserver/clustercontroller/logRouterRecruitmentWithoutMinWorker") { + const Key dcId = "dc1"_sr; + ClusterControllerData data = makeRecruitmentTestData(dcId); - ClusterControllerData data(cci, - LocalityData(), - ServerCoordinators(Reference( - new ClusterConnectionMemoryRecord(ClusterConnectionString()))), - makeReference>>()); + auto stateless = addRecruitmentTestWorkers(data, dcId, ProcessClass::StatelessClass, "sl"_sr, 2); + auto transaction = addRecruitmentTestWorkers(data, dcId, ProcessClass::TransactionClass, "tx"_sr, 2); - auto makeWorkers = [&](ProcessClass::ClassType classType, StringRef prefix, int count) { - std::vector workers; - for (int i = 0; i < count; i++) { - auto pid = prefix.toString() + std::to_string(i); - WorkerInterface wi; - wi.initEndpoints(); - wi.locality.set(LocalityData::keyZoneId, Standalone(pid + "_zone")); - wi.locality.set(LocalityData::keyProcessId, Standalone(pid)); - data.id_worker[wi.locality.processId()] = - WorkerInfo(Future(), - ReplyPromise(), - 0, - wi, - ProcessClass(classType, ProcessClass::CommandLineSource), - ProcessClass(classType, ProcessClass::CommandLineSource), - ClusterControllerPriorityInfo( - recruitment::UnsetFit, false, ClusterControllerPriorityInfo::FitnessUnknown), - false, - true, - Standalone>()); - data.id_worker[wi.locality.processId()].verified = true; - workers.push_back(wi); - } - return workers; - }; + DatabaseConfiguration config; + config.initialized = true; - // For LogRouter: stateless -> GoodFit, transaction -> OkayFit. - auto statelessWorkers = makeWorkers(ProcessClass::StatelessClass, "sl"_sr, 2); - auto transactionWorkers = makeWorkers(ProcessClass::TransactionClass, "tx"_sr, 2); + std::map>, int> id_used; + auto routers = data.getWorkersForRoleInDatacenter(dcId, recruitment::LogRouter, 4, config, id_used); - data.masterProcessId = statelessWorkers[0].locality.processId(); - data.clusterControllerProcessId = statelessWorkers[1].locality.processId(); - data.startTime = now(); + ASSERT_EQ(routers.size(), 4); + std::set>> pids; + for (const auto& w : routers) { + pids.insert(w.interf.locality.processId()); + } + ASSERT_EQ(pids.size(), 4); // all distinct, spanning both fitness levels + return Void(); +} + +// End-to-end regression test for the original bug through the production recruitment path +// (findWorkersForConfigurationDispatch). In a small cluster whose desired proxy counts equal +// the number of stateless processes, every commit proxy, GRV proxy and resolver must be +// recruited (previously only one commit proxy was), each on a distinct stateless process. +// Process classes are honored: TLogs land on transaction processes, proxies on stateless ones. +TEST_CASE("/fdbserver/clustercontroller/proxyColocationOnStateless") { + const Key dcId = "dc1"_sr; + ClusterControllerData data = makeRecruitmentTestData(dcId); + // Make the good-recruitment gate pass so we assert on the recruitment result itself. + data.goodRecruitmentTime = Void(); + + constexpr int kCount = 3; + auto stateless = addRecruitmentTestWorkers(data, dcId, ProcessClass::StatelessClass, "sl"_sr, kCount); + auto transaction = addRecruitmentTestWorkers(data, dcId, ProcessClass::TransactionClass, "tx"_sr, kCount); + auto storage = addRecruitmentTestWorkers(data, dcId, ProcessClass::StorageClass, "ss"_sr, kCount); + + // Two of the stateless processes already host the master and cluster controller. + data.masterProcessId = stateless[0].locality.processId(); + data.clusterControllerProcessId = stateless[1].locality.processId(); DatabaseConfiguration config; config.initialized = true; + config.usableRegions = 1; + RegionInfo region; + region.dcId = dcId; + config.regions.push_back(region); + config.tLogReplicationFactor = kCount; + config.desiredTLogCount = kCount; + config.commitProxyCount = kCount; + config.grvProxyCount = kCount; + config.resolverCount = kCount; + config.tLogPolicy = makeReference(); data.db.config = config; data.db.fullyRecoveredConfig = config; - ClusterControllerData::WorkerUsages id_used; - data.updateKnownIds(&id_used); - - auto logRouters = data.getWorkersForRoleInDatacenter( - Optional>(), recruitment::LogRouter, 4, config, id_used); - - ASSERT_EQ(4, logRouters.size()); + RecruitFromConfigurationRequest req(config, /*recruitSeedServers=*/false, /*maxOldLogRouters=*/0); + auto result = data.findWorkersForConfigurationDispatch(req, /*checkGoodRecruitment=*/true); - // All four processes must be used: both fitness levels. - std::set>> recruitedPids; - for (const auto& w : logRouters) { - recruitedPids.insert(w.interf.locality.processId()); - } - for (const auto& w : statelessWorkers) { - ASSERT(recruitedPids.contains(w.locality.processId())); + std::set>> statelessPids, transactionPids; + for (const auto& w : stateless) { + statelessPids.insert(w.locality.processId()); } - for (const auto& w : transactionWorkers) { - ASSERT(recruitedPids.contains(w.locality.processId())); + for (const auto& w : transaction) { + transactionPids.insert(w.locality.processId()); } - return Void(); -} - -// Regression test for the singleton-placement amplification: PID_USED_AMP_FOR_NON_SINGLETON -// must amplify the accumulated non-singleton usage once (snapshot semantics), so each -// singleton placed afterwards still costs one unit. With a one-unit read-time factor the -// first singleton placement on the lighter process would tie it with the busier process -// and subsequent placements could leak onto the busier process. -TEST_CASE("/fdbserver/clustercontroller/singletonPlacementKeepsSnapshotAmplification") { - ClusterControllerData data(ClusterControllerFullInterface(), - LocalityData(), - ServerCoordinators(Reference( - new ClusterConnectionMemoryRecord(ClusterConnectionString()))), - makeReference>>()); + // TLogs are recruited on the transaction processes. + ASSERT_EQ(result.tLogs.size(), kCount); + for (const auto& interf : result.tLogs) { + ASSERT(transactionPids.count(interf.locality.processId()) == 1); + } - auto addWorker = [&](StringRef pid) -> Optional> { - WorkerInterface wi; - wi.initEndpoints(); - wi.locality.set(LocalityData::keyZoneId, Standalone(pid.toString() + "_zone")); - wi.locality.set(LocalityData::keyProcessId, Standalone(pid)); - data.id_worker[wi.locality.processId()] = WorkerInfo( - Future(), - ReplyPromise(), - 0, - wi, - ProcessClass(ProcessClass::StatelessClass, ProcessClass::CommandLineSource), - ProcessClass(ProcessClass::StatelessClass, ProcessClass::CommandLineSource), - ClusterControllerPriorityInfo(recruitment::UnsetFit, false, ClusterControllerPriorityInfo::FitnessUnknown), - false, - true, - Standalone>()); - data.id_worker[wi.locality.processId()].verified = true; - return wi.locality.processId(); + // Each proxy/resolver set is fully recruited onto distinct stateless processes. + auto assertDistinctStateless = [&statelessPids, kCount](const std::vector& interfaces) { + ASSERT_EQ(interfaces.size(), kCount); + std::set>> pids; + for (const auto& interf : interfaces) { + pids.insert(interf.locality.processId()); + ASSERT(statelessPids.count(interf.locality.processId()) == 1); + } + ASSERT_EQ(pids.size(), kCount); }; - - // Process A carries one non-singleton role, process B carries two. - auto pidA = addWorker("a"_sr); - auto pidB = addWorker("b"_sr); - // Unverified master process so onMasterIsBetter() never redirects placements. - data.masterProcessId = addWorker("m"_sr); - data.id_worker[data.masterProcessId.get()].verified = false; - data.startTime = now(); - data.db.config.initialized = true; - - ClusterControllerData::WorkerUsages id_used; - id_used[pidA].addRole(recruitment::CommitProxy); - id_used[pidB].addRole(recruitment::CommitProxy); - id_used[pidB].addRole(recruitment::GrvProxy); - - amplifyNonSingletonUsage(id_used); - ASSERT_EQ(PID_USED_AMP_FOR_NON_SINGLETON, id_used[pidA].getWeight()); - ASSERT_EQ(2 * PID_USED_AMP_FOR_NON_SINGLETON, id_used[pidB].getWeight()); - - // All three singletons must land on A (100 -> 101 -> 102 -> 103 < 200), and each - // placement must cost exactly one unit: with a read-time factor A would tie B at 200 - // after the first placement. - const recruitment::ClusterRole singletonRoles[] = { recruitment::Ratekeeper, - recruitment::DataDistributor, - recruitment::ConsistencyScan }; - unsigned expectedWeight = PID_USED_AMP_FOR_NON_SINGLETON; - for (const auto& role : singletonRoles) { - WorkerDetails worker = findNewProcessForSingleton(&data, role, id_used); - ASSERT(worker.interf.locality.processId() == pidA); - expectedWeight += 1; - ASSERT_EQ(expectedWeight, id_used[pidA].getWeight()); - } - ASSERT_EQ(2 * PID_USED_AMP_FOR_NON_SINGLETON, id_used[pidB].getWeight()); - + assertDistinctStateless(result.commitProxies); + assertDistinctStateless(result.grvProxies); + assertDistinctStateless(result.resolvers); return Void(); } diff --git a/fdbserver/clustercontroller/ClusterController.h b/fdbserver/clustercontroller/ClusterController.h index 5311adc0dff..489245887ae 100644 --- a/fdbserver/clustercontroller/ClusterController.h +++ b/fdbserver/clustercontroller/ClusterController.h @@ -291,71 +291,6 @@ class ClusterControllerData { } }; - struct WorkerUsage { - std::bitset roles; - unsigned weight = 0; - - static unsigned roleWeight(recruitment::ClusterRole role) { - // All roles currently contribute equally to the usage weight. - // Per-role weights can be introduced here if different roles - // should affect load differently. - return 1; - } - - void addRole(recruitment::ClusterRole role) { - if (!roles.test(role)) { - roles.set(role); - weight += roleWeight(role); - } - } - - unsigned getWeight() const { return weight; } - - unsigned getUniqueHash() const { - static_assert(recruitment::NoRole <= 32, "role bitset must fit in an unsigned"); - return roles.to_ulong(); - } - - std::string toString() const { - std::string roleCodes; - - auto appendRoleCode = [&](recruitment::ClusterRole role) { - switch (role) { - case recruitment::Storage: - case recruitment::TLog: - case recruitment::CommitProxy: - case recruitment::GrvProxy: - case recruitment::Master: - case recruitment::Resolver: - case recruitment::LogRouter: - case recruitment::ClusterController: - case recruitment::DataDistributor: - case recruitment::Ratekeeper: - case recruitment::ConsistencyScan: - case recruitment::Backup: - case recruitment::EncryptKeyProxy: - case recruitment::Worker: - roleCodes.append(Role::get(role).abbreviation); - break; - default: - roleCodes.append(format("role%d", role)); - break; - } - }; - - for (unsigned r = 0; r < recruitment::NoRole; r++) { - if (roles.test((recruitment::ClusterRole)r)) { - if (!roleCodes.empty()) - roleCodes.append(","); - appendRoleCode((recruitment::ClusterRole)r); - } - } - return roleCodes; - } - }; - - using WorkerUsages = std::map>, WorkerUsage>; - bool workerAvailable(WorkerInfo const& worker, bool checkStable) const { return worker.verified && ((now() - startTime < 2 * FLOW_KNOBS->SERVER_REQUEST_INTERVAL) || (IFailureMonitor::failureMonitor() @@ -651,7 +586,7 @@ class ClusterControllerData { // It attempts to evenly recruit processes from across data_halls or datacenters std::vector getWorkersForTlogsComplex(DatabaseConfiguration const& conf, int32_t desired, - WorkerUsages& id_used, + std::map>, int>& id_used, StringRef field, int minFields, int minPerField, @@ -659,7 +594,7 @@ class ClusterControllerData { bool checkStable, const std::set>& dcIds, const std::vector& exclusionWorkerIds) { - std::map, std::vector> fitness_workers; + std::map, std::vector> fitness_workers; // Go through all the workers to list all the workers that can be recruited. for (const auto& [worker_process_id, worker_info] : id_worker) { @@ -726,10 +661,8 @@ class ClusterControllerData { continue; } - fitness_workers[std::make_tuple(fitness, - id_used[worker_process_id].getWeight(), - isLongLivedStateless(worker_process_id), - id_used[worker_process_id].getUniqueHash())] + fitness_workers[std::make_tuple( + fitness, id_used[worker_process_id], isLongLivedStateless(worker_process_id))] .push_back(worker_details); } @@ -838,7 +771,7 @@ class ClusterControllerData { } for (auto& result : resultSet) { - id_used[result.interf.locality.processId()].addRole(recruitment::TLog); + id_used[result.interf.locality.processId()]++; } return std::vector(resultSet.begin(), resultSet.end()); @@ -847,7 +780,7 @@ class ClusterControllerData { // Attempt to recruit TLogs without degraded processes and see if it improves the configuration std::vector getWorkersForTlogsComplex(DatabaseConfiguration const& conf, int32_t desired, - WorkerUsages& id_used, + std::map>, int>& id_used, StringRef field, int minFields, int minPerField, @@ -855,7 +788,7 @@ class ClusterControllerData { const std::set>& dcIds, const std::vector& exclusionWorkerIds) { desired = std::max(desired, minFields * minPerField); - auto withDegradedUsed = id_used; + std::map>, int> withDegradedUsed = id_used; auto withDegraded = getWorkersForTlogsComplex(conf, desired, withDegradedUsed, @@ -883,7 +816,7 @@ class ClusterControllerData { } try { - auto withoutDegradedUsed = id_used; + std::map>, int> withoutDegradedUsed = id_used; auto withoutDegraded = getWorkersForTlogsComplex(conf, desired, withoutDegradedUsed, @@ -917,12 +850,11 @@ class ClusterControllerData { std::vector getWorkersForTlogsSimple(DatabaseConfiguration const& conf, int32_t required, int32_t desired, - WorkerUsages& id_used, + std::map>, int>& id_used, bool checkStable, const std::set>& dcIds, const std::vector& exclusionWorkerIds) { - std::map, std::vector> - fitness_workers; + std::map, std::vector> fitness_workers; // Go through all the workers to list all the workers that can be recruited. for (const auto& [worker_process_id, worker_info] : id_worker) { @@ -992,11 +924,10 @@ class ClusterControllerData { } fitness_workers[std::make_tuple(fitness, - id_used[worker_process_id].getWeight(), + id_used[worker_process_id], worker_details.degraded, isLongLivedStateless(worker_process_id), - inCCDC, - id_used[worker_process_id].getUniqueHash())] + inCCDC)] .push_back(worker_details); } @@ -1052,7 +983,7 @@ class ClusterControllerData { ASSERT(resultSet.size() >= required && resultSet.size() <= desired); for (auto& result : resultSet) { - id_used[result.interf.locality.processId()].addRole(recruitment::TLog); + id_used[result.interf.locality.processId()]++; } return std::vector(resultSet.begin(), resultSet.end()); @@ -1075,12 +1006,11 @@ class ClusterControllerData { int32_t required, int32_t desired, Reference const& policy, - WorkerUsages& id_used, + std::map>, int>& id_used, bool checkStable = false, const std::set>& dcIds = std::set>(), const std::vector& exclusionWorkerIds = {}) { - std::map, std::vector> - fitness_workers; + std::map, std::vector> fitness_workers; std::vector results; Reference logServerSet = makeReference>(); auto* logServerMap = (LocalityMap*)logServerSet.getPtr(); @@ -1155,11 +1085,7 @@ class ClusterControllerData { fitness = std::max(fitness, recruitment::GoodFit); } - fitness_workers[std::make_tuple(fitness, - id_used[worker_process_id].getWeight(), - worker_details.degraded, - inCCDC, - id_used[worker_process_id].getUniqueHash())] + fitness_workers[std::make_tuple(fitness, id_used[worker_process_id], worker_details.degraded, inCCDC)] .push_back(worker_details); } @@ -1209,7 +1135,7 @@ class ClusterControllerData { results.push_back(*object); } for (auto& result : results) { - id_used[result.interf.locality.processId()].addRole(recruitment::TLog); + id_used[result.interf.locality.processId()]++; } return results; } @@ -1283,7 +1209,7 @@ class ClusterControllerData { results.push_back(*object); } for (auto& result : results) { - id_used[result.interf.locality.processId()].addRole(recruitment::TLog); + id_used[result.interf.locality.processId()]++; } return results; } @@ -1308,7 +1234,7 @@ class ClusterControllerData { tLocalities.push_back(object->interf.locality); } for (auto& result : results) { - id_used[result.interf.locality.processId()].addRole(recruitment::TLog); + id_used[result.interf.locality.processId()]++; } TraceEvent("GetTLogTeamDone") .detail("Policy", policy->info()) @@ -1332,7 +1258,7 @@ class ClusterControllerData { int32_t required, int32_t desired, Reference const& policy, - WorkerUsages& id_used, + std::map>, int>& id_used, bool checkStable = false, const std::set>& dcIds = std::set>(), const std::vector& exclusionWorkerIds = {}) { @@ -1344,7 +1270,7 @@ class ClusterControllerData { if (embedded->name() == "Across") { auto* pa2 = (PolicyAcross*)embedded.getPtr(); if (pa2->attributeKey() == "zoneid" && pa2->embeddedPolicyName() == "One") { - auto testUsed = id_used; + std::map>, int> testUsed = id_used; auto workers = getWorkersForTlogsComplex(conf, desired, @@ -1408,7 +1334,7 @@ class ClusterControllerData { useSimple = true; } if (useSimple) { - auto testUsed = id_used; + std::map>, int> testUsed = id_used; auto workers = getWorkersForTlogsSimple(conf, required, desired, id_used, checkStable, dcIds, exclusionWorkerIds); @@ -1451,7 +1377,7 @@ class ClusterControllerData { std::vector getWorkersForSatelliteLogs(const DatabaseConfiguration& conf, const RegionInfo& region, const RegionInfo& remoteRegion, - WorkerUsages& id_used, + std::map>, int>& id_used, bool& satelliteFallback, bool checkStable = false) { int startDC = 0; @@ -1494,7 +1420,7 @@ class ClusterControllerData { // TLogs can be recruited. It does not balance the number of desired TLogs across the satellite and // remote sides. if (remoteDCUsedAsSatellite) { - WorkerUsages tmpIdUsed; + std::map>, int> tmpIdUsed; auto remoteLogs = getWorkersForTlogs(conf, conf.getRemoteTLogReplicationFactor(), conf.getRemoteTLogReplicationFactor(), @@ -1557,11 +1483,10 @@ class ClusterControllerData { recruitment::ClusterRole role, recruitment::Fitness unacceptableFitness, DatabaseConfiguration const& conf, - WorkerUsages& id_used, + std::map>, int>& id_used, std::map>, int> preferredSharing = {}, bool checkStable = false) { - std::map, std::vector> - fitness_workers; + std::map, std::vector> fitness_workers; for (auto& it : id_worker) { auto fitness = recruitment::machineClassFitness(it.second.details.processClass, role); @@ -1573,17 +1498,16 @@ class ClusterControllerData { it.second.details.interf.locality.dcId() == dcId) { auto sharing = preferredSharing.find(it.first); fitness_workers[std::make_tuple(fitness, - id_used[it.first].getWeight(), + id_used[it.first], isLongLivedStateless(it.first), - sharing != preferredSharing.end() ? sharing->second : 1e6, - id_used[it.first].getUniqueHash())] + sharing != preferredSharing.end() ? sharing->second : 1e6)] .push_back(it.second.details); } } if (!fitness_workers.empty()) { auto worker = deterministicRandom()->randomChoice(fitness_workers.begin()->second); - id_used[worker.interf.locality.processId()].addRole(role); + id_used[worker.interf.locality.processId()]++; return WorkerFitnessInfo(worker, std::max(recruitment::GoodFit, std::get<0>(fitness_workers.begin()->first)), std::get<1>(fitness_workers.begin()->first)); @@ -1597,22 +1521,16 @@ class ClusterControllerData { recruitment::ClusterRole role, int amount, DatabaseConfiguration const& conf, - WorkerUsages& id_used, + std::map>, int>& id_used, std::map>, int> preferredSharing = {}, Optional minWorker = Optional(), bool checkStable = false) { - // Do not recruit workers with worse fitness than the already-accepted minWorker. - // The ceiling is fixed for the whole loop: lowering it to the best bucket - // consumed would stop the fill at the first fitness level and prevent reaching - // `amount` from worse-but-acceptable buckets. With no minWorker there is no gate. - recruitment::Fitness used_fitness = minWorker.present() ? minWorker.get().fitness : recruitment::NeverAssign; struct WorkerFitnessKey { recruitment::Fitness fitness; - unsigned used; + int used; bool unreliableBackup; bool longLivedStateless; int sharing; - unsigned uniqueHash; std::strong_ordering operator<=>(WorkerFitnessKey const&) const = default; }; @@ -1628,32 +1546,35 @@ class ClusterControllerData { for (auto& it : id_worker) { auto fitness = recruitment::machineClassFitness(it.second.details.processClass, role); + // Candidates must not be worse than the already-accepted minWorker. Usage is + // deliberately NOT part of this check: comparing a candidate's usage against + // minWorker's emptied the pool whenever the desired count exceeded the + // least-used equal-fitness processes (e.g. every stateless process when some + // already host master/cluster controller). Spreading across processes is still + // provided by the bucket ordering on `used` in the fill loop below. if (workerAvailable(it.second, checkStable) && !conf.isExcludedServer(it.second.details.interf.addresses(), it.second.details.interf.locality) && !isExcludedDegradedServer(it.second.details.interf.addresses()) && it.second.details.interf.locality.dcId() == dcId && - (!minWorker.present() || (it.second.details.interf.id() != minWorker.get().worker.interf.id()))) { + (!minWorker.present() || (it.second.details.interf.id() != minWorker.get().worker.interf.id() && + fitness <= minWorker.get().fitness))) { auto sharing = preferredSharing.find(it.first); fitness_workers[{ fitness, - id_used[it.first].getWeight(), + id_used[it.first], role == recruitment::Backup && g_network->isSimulated() && !g_simulator->getProcessByAddress(it.second.details.interf.address()) ->isReliable(), isLongLivedStateless(it.first), - sharing != preferredSharing.end() ? sharing->second : 1'000'000, - id_used[it.first].getUniqueHash() }] + sharing != preferredSharing.end() ? sharing->second : 1'000'000 }] .push_back(it.second.details); } } for (auto& it : fitness_workers) { - if (it.first.fitness > used_fitness) { - break; // do not recruit with a greater fitness - } deterministicRandom()->randomShuffle(it.second); for (int i = 0; i < it.second.size(); i++) { results.push_back(it.second[i]); - id_used[it.second[i].interf.locality.processId()].addRole(role); + id_used[it.second[i].interf.locality.processId()]++; if (results.size() == amount) return results; } @@ -1670,7 +1591,7 @@ class ClusterControllerData { recruitment::Fitness worstFit; recruitment::ClusterRole role; int count; - unsigned worstUsed = 1; + int worstUsed = 1; bool degraded = false; RoleFitness(int bestFit, int worstFit, int count, recruitment::ClusterRole role) @@ -1686,7 +1607,7 @@ class ClusterControllerData { RoleFitness(const std::vector& workers, recruitment::ClusterRole role, - const WorkerUsages& id_used) + const std::map>, int>& id_used) : role(role) { // Every recruitment will attempt to recruit the preferred amount through GoodFit, // So a recruitment which only has BestFit is not better than one that has a GoodFit process @@ -1702,7 +1623,7 @@ class ClusterControllerData { TraceEvent(SevError, "UsedNotFound").detail("ProcessId", it.interf.locality.processId().get()); ASSERT(false); } - if (thisUsed->second.getWeight() == 0) { + if (thisUsed->second == 0) { TraceEvent(SevError, "UsedIsZero").detail("ProcessId", it.interf.locality.processId().get()); ASSERT(false); } @@ -1711,9 +1632,9 @@ class ClusterControllerData { if (thisFit > worstFit) { worstFit = thisFit; - worstUsed = thisUsed->second.getWeight(); + worstUsed = thisUsed->second; } else if (thisFit == worstFit) { - worstUsed = std::max(worstUsed, thisUsed->second.getWeight()); + worstUsed = std::max(worstUsed, thisUsed->second); } degraded = degraded || it.degraded; } @@ -1761,7 +1682,7 @@ class ClusterControllerData { degraded == r.degraded; } - std::string toString() const { return format("%d %u %d %d %d", worstFit, worstUsed, count, degraded, bestFit); } + std::string toString() const { return format("%d %d %d %d %d", worstFit, worstUsed, count, degraded, bestFit); } }; std::set>> getDatacenters(DatabaseConfiguration const& conf, @@ -1777,15 +1698,15 @@ class ClusterControllerData { return result; } - void updateKnownIds(WorkerUsages* id_used) { - (*id_used)[masterProcessId].addRole(recruitment::Master); - (*id_used)[clusterControllerProcessId].addRole(recruitment::ClusterController); + void updateKnownIds(std::map>, int>* id_used) { + (*id_used)[masterProcessId]++; + (*id_used)[clusterControllerProcessId]++; } RecruitRemoteFromConfigurationReply findRemoteWorkersForConfiguration( RecruitRemoteFromConfigurationRequest const& req) { RecruitRemoteFromConfigurationReply result; - WorkerUsages id_used; + std::map>, int> id_used; updateKnownIds(&id_used); // Primary and satellite TLogs can share the remote DC. Account for their workers when placing log routers so @@ -1793,7 +1714,7 @@ class ClusterControllerData { for (const auto& [processId, worker] : id_worker) { if (std::find(req.exclusionWorkerIds.begin(), req.exclusionWorkerIds.end(), worker.details.interf.id()) != req.exclusionWorkerIds.end()) { - id_used[processId].addRole(recruitment::TLog); + id_used[processId]++; } } @@ -1862,7 +1783,7 @@ class ClusterControllerData { Optional dcId, bool checkGoodRecruitment) { RecruitFromConfigurationReply result; - WorkerUsages id_used; + std::map>, int> id_used; updateKnownIds(&id_used); ASSERT(dcId.present()); @@ -1912,6 +1833,13 @@ class ClusterControllerData { dcId, recruitment::Resolver, recruitment::ExcludeFit, req.configuration, id_used, preferredSharing); preferredSharing[first_resolver.worker.interf.locality.processId()] = 2; + // If one of the first process recruitments is forced to share a process, allow all of next recruitments + // to also share a process. + auto maxUsed = std::max({ first_commit_proxy.used, first_grv_proxy.used, first_resolver.used }); + first_commit_proxy.used = maxUsed; + first_grv_proxy.used = maxUsed; + first_resolver.used = maxUsed; + auto commit_proxies = getWorkersForRoleInDatacenter(dcId, recruitment::CommitProxy, req.configuration.getDesiredCommitProxies(), @@ -2098,7 +2026,7 @@ class ClusterControllerData { throw no_more_servers(); } else { RecruitFromConfigurationReply result; - WorkerUsages id_used; + std::map>, int> id_used; updateKnownIds(&id_used); auto tlogs = getWorkersForTlogs(req.configuration, req.configuration.tLogReplicationFactor, @@ -2162,6 +2090,13 @@ class ClusterControllerData { preferredSharing); preferredSharing[first_resolver.worker.interf.locality.processId()] = 2; + // If one of the first process recruitments is forced to share a process, allow all of next + // recruitments to also share a process. + auto maxUsed = std::max({ first_commit_proxy.used, first_grv_proxy.used, first_resolver.used }); + first_commit_proxy.used = maxUsed; + first_grv_proxy.used = maxUsed; + first_resolver.used = maxUsed; + auto commit_proxies = getWorkersForRoleInDatacenter(dcId, recruitment::CommitProxy, req.configuration.getDesiredCommitProxies(), @@ -2289,18 +2224,17 @@ class ClusterControllerData { } void updateIdUsed(const std::vector& workers, - recruitment::ClusterRole role, - WorkerUsages& id_used) { + std::map>, int>& id_used) { for (auto& it : workers) { - id_used[it.locality.processId()].addRole(role); + id_used[it.locality.processId()]++; } } void compareWorkers(const DatabaseConfiguration& conf, const std::vector& first, - WorkerUsages& firstUsed, + const std::map>, int>& firstUsed, const std::vector& second, - WorkerUsages& secondUsed, + const std::map>, int>& secondUsed, recruitment::ClusterRole role, std::string description) { std::vector firstDetails; @@ -2354,17 +2288,17 @@ class ClusterControllerData { if (!remoteDCUsedAsSatellite) { RecruitFromConfigurationReply compare = findWorkersForConfigurationDispatch(req, false); - WorkerUsages firstUsed; - WorkerUsages secondUsed; + std::map>, int> firstUsed; + std::map>, int> secondUsed; updateKnownIds(&firstUsed); updateKnownIds(&secondUsed); - updateIdUsed(rep.tLogs, recruitment::TLog, firstUsed); - updateIdUsed(compare.tLogs, recruitment::TLog, secondUsed); + updateIdUsed(rep.tLogs, firstUsed); + updateIdUsed(compare.tLogs, secondUsed); compareWorkers( req.configuration, rep.tLogs, firstUsed, compare.tLogs, secondUsed, recruitment::TLog, "TLog"); - updateIdUsed(rep.satelliteTLogs, recruitment::TLog, firstUsed); - updateIdUsed(compare.satelliteTLogs, recruitment::TLog, secondUsed); + updateIdUsed(rep.satelliteTLogs, firstUsed); + updateIdUsed(compare.satelliteTLogs, secondUsed); compareWorkers(req.configuration, rep.satelliteTLogs, firstUsed, @@ -2372,12 +2306,12 @@ class ClusterControllerData { secondUsed, recruitment::TLog, "Satellite"); - updateIdUsed(rep.commitProxies, recruitment::CommitProxy, firstUsed); - updateIdUsed(compare.commitProxies, recruitment::CommitProxy, secondUsed); - updateIdUsed(rep.grvProxies, recruitment::GrvProxy, firstUsed); - updateIdUsed(compare.grvProxies, recruitment::GrvProxy, secondUsed); - updateIdUsed(rep.resolvers, recruitment::Resolver, firstUsed); - updateIdUsed(compare.resolvers, recruitment::Resolver, secondUsed); + updateIdUsed(rep.commitProxies, firstUsed); + updateIdUsed(compare.commitProxies, secondUsed); + updateIdUsed(rep.grvProxies, firstUsed); + updateIdUsed(compare.grvProxies, secondUsed); + updateIdUsed(rep.resolvers, firstUsed); + updateIdUsed(compare.resolvers, secondUsed); compareWorkers(req.configuration, rep.commitProxies, firstUsed, @@ -2399,8 +2333,8 @@ class ClusterControllerData { secondUsed, recruitment::Resolver, "Resolver"); - updateIdUsed(rep.backupWorkers, recruitment::Backup, firstUsed); - updateIdUsed(compare.backupWorkers, recruitment::Backup, secondUsed); + updateIdUsed(rep.backupWorkers, firstUsed); + updateIdUsed(compare.backupWorkers, secondUsed); compareWorkers(req.configuration, rep.backupWorkers, firstUsed, @@ -2425,7 +2359,7 @@ class ClusterControllerData { } try { - WorkerUsages id_used; + std::map>, int> id_used; getWorkerForRoleInDatacenter( regions[0].dcId, recruitment::ClusterController, recruitment::ExcludeFit, db.config, id_used, {}, true); getWorkerForRoleInDatacenter( @@ -2479,9 +2413,10 @@ class ClusterControllerData { } } - void updateIdUsed(const std::vector& workers, recruitment::ClusterRole role, WorkerUsages& id_used) { + void updateIdUsed(const std::vector& workers, + std::map>, int>& id_used) { for (auto& it : workers) { - id_used[it.interf.locality.processId()].addRole(role); + id_used[it.interf.locality.processId()]++; } } @@ -2666,10 +2601,10 @@ class ClusterControllerData { oldMasterFit = std::max(oldMasterFit, recruitment::ExcludeFit); } - WorkerUsages id_used; - WorkerUsages old_id_used; - id_used[clusterControllerProcessId].addRole(recruitment::ClusterController); - old_id_used[clusterControllerProcessId].addRole(recruitment::ClusterController); + std::map>, int> id_used; + std::map>, int> old_id_used; + id_used[clusterControllerProcessId]++; + old_id_used[clusterControllerProcessId]++; WorkerFitnessInfo mworker = getWorkerForRoleInDatacenter( clusterControllerDcId, recruitment::Master, recruitment::NeverAssign, db.config, id_used, {}, true); auto newMasterFit = recruitment::machineClassFitness(mworker.worker.processClass, recruitment::Master); @@ -2677,7 +2612,7 @@ class ClusterControllerData { newMasterFit = std::max(newMasterFit, recruitment::ExcludeFit); } - old_id_used[masterWorker->first].addRole(recruitment::Master); + old_id_used[masterWorker->first]++; if (oldMasterFit < newMasterFit) { TraceEvent("NewRecruitmentIsWorse", id) .detail("OldMasterFit", oldMasterFit) @@ -2717,7 +2652,7 @@ class ClusterControllerData { } // Check tLog fitness - updateIdUsed(tlogs, recruitment::TLog, old_id_used); + updateIdUsed(tlogs, old_id_used); RoleFitness oldTLogFit(tlogs, recruitment::TLog, old_id_used); auto newTLogs = getWorkersForTlogs(db.config, db.config.tLogReplicationFactor, @@ -2742,7 +2677,7 @@ class ClusterControllerData { } } - updateIdUsed(satellite_tlogs, recruitment::TLog, old_id_used); + updateIdUsed(satellite_tlogs, old_id_used); RoleFitness oldSatelliteTLogFit(satellite_tlogs, recruitment::TLog, old_id_used); bool newSatelliteFallback = false; auto newSatelliteTLogs = satellite_tlogs; @@ -2802,7 +2737,7 @@ class ClusterControllerData { return false; } - updateIdUsed(remote_tlogs, recruitment::TLog, old_id_used); + updateIdUsed(remote_tlogs, old_id_used); RoleFitness oldRemoteTLogFit(remote_tlogs, recruitment::TLog, old_id_used); std::vector exclusionWorkerIds; auto fn = [](const WorkerDetails& in) { return in.interf.id(); }; @@ -2826,7 +2761,7 @@ class ClusterControllerData { oldTLogFit.count * std::max(1, db.config.desiredLogRouterCount / std::max(1, oldTLogFit.count)); int newRouterCount = newTLogFit.count * std::max(1, db.config.desiredLogRouterCount / std::max(1, newTLogFit.count)); - updateIdUsed(log_routers, recruitment::LogRouter, old_id_used); + updateIdUsed(log_routers, old_id_used); RoleFitness oldLogRoutersFit(log_routers, recruitment::LogRouter, old_id_used); RoleFitness newLogRoutersFit = oldLogRoutersFit; if (db.config.usableRegions > 1 && dbi.recoveryState == RecoveryState::FULLY_RECOVERED) { @@ -2850,9 +2785,9 @@ class ClusterControllerData { } // Check proxy/grvProxy/resolver fitness - updateIdUsed(commitProxyClasses, recruitment::CommitProxy, old_id_used); - updateIdUsed(grvProxyClasses, recruitment::GrvProxy, old_id_used); - updateIdUsed(resolverClasses, recruitment::Resolver, old_id_used); + updateIdUsed(commitProxyClasses, old_id_used); + updateIdUsed(grvProxyClasses, old_id_used); + updateIdUsed(resolverClasses, old_id_used); RoleFitness oldCommitProxyFit(commitProxyClasses, recruitment::CommitProxy, old_id_used); RoleFitness oldGrvProxyFit(grvProxyClasses, recruitment::GrvProxy, old_id_used); RoleFitness oldResolverFit(resolverClasses, recruitment::Resolver, old_id_used); @@ -2882,6 +2817,10 @@ class ClusterControllerData { preferredSharing, true); preferredSharing[first_resolver.worker.interf.locality.processId()] = 2; + auto maxUsed = std::max({ first_commit_proxy.used, first_grv_proxy.used, first_resolver.used }); + first_commit_proxy.used = maxUsed; + first_grv_proxy.used = maxUsed; + first_resolver.used = maxUsed; auto commit_proxies = getWorkersForRoleInDatacenter(clusterControllerDcId, recruitment::CommitProxy, db.config.getDesiredCommitProxies(), @@ -2912,7 +2851,7 @@ class ClusterControllerData { RoleFitness newResolverFit(resolvers, recruitment::Resolver, id_used); // Check backup worker fitness - updateIdUsed(backup_workers, recruitment::Backup, old_id_used); + updateIdUsed(backup_workers, old_id_used); RoleFitness oldBackupWorkersFit(backup_workers, recruitment::Backup, old_id_used); const int nBackup = backup_addresses.size(); RoleFitness newBackupWorkersFit(getWorkersForRoleInDatacenter(clusterControllerDcId, @@ -3045,30 +2984,30 @@ class ClusterControllerData { } // Returns a map of for all non-singleton roles - WorkerUsages getUsedIds() { - WorkerUsages idUsed; + std::map>, int> getUsedIds() { + std::map>, int> idUsed; updateKnownIds(&idUsed); auto& dbInfo = db.serverInfo->get(); for (const auto& tlogset : dbInfo.logSystemConfig.tLogs) { for (const auto& tlog : tlogset.tLogs) { if (tlog.present()) { - idUsed[tlog.interf().filteredLocality.processId()].addRole(recruitment::TLog); + idUsed[tlog.interf().filteredLocality.processId()]++; } } } for (const CommitProxyInterface& interf : dbInfo.client.commitProxies) { ASSERT(interf.processId.present()); - idUsed[interf.processId].addRole(recruitment::CommitProxy); + idUsed[interf.processId]++; } for (const GrvProxyInterface& interf : dbInfo.client.grvProxies) { ASSERT(interf.processId.present()); - idUsed[interf.processId].addRole(recruitment::GrvProxy); + idUsed[interf.processId]++; } for (const ResolverInterface& interf : dbInfo.resolvers) { ASSERT(interf.locality.processId().present()); - idUsed[interf.locality.processId()].addRole(recruitment::Resolver); + idUsed[interf.locality.processId()]++; } return idUsed; } diff --git a/fdbserver/clustercontroller/ClusterRecovery.cpp b/fdbserver/clustercontroller/ClusterRecovery.cpp index 7c1b16353f2..6b5034ab198 100644 --- a/fdbserver/clustercontroller/ClusterRecovery.cpp +++ b/fdbserver/clustercontroller/ClusterRecovery.cpp @@ -85,8 +85,8 @@ Future recruitNewMaster(ClusterControllerData* cluster, // We must recruit the master in the same data center as the cluster controller. // This should always be possible, because we can recruit the master on the same process as the cluster // controller. - ClusterControllerData::WorkerUsages id_used; - id_used[cluster->clusterControllerProcessId].addRole(recruitment::ClusterController); + std::map>, int> id_used; + id_used[cluster->clusterControllerProcessId]++; masterWorker = cluster->getWorkerForRoleInDatacenter( cluster->clusterControllerDcId, recruitment::Master, recruitment::NeverAssign, db->config, id_used); if ((recruitment::machineClassFitness(masterWorker.worker.processClass, recruitment::Master) > From bff0e4fcd1db709ad0f4fb9c79c772887ad464d6 Mon Sep 17 00:00:00 2001 From: Mark Shabanov Date: Wed, 9 Sep 2026 18:09:13 +0300 Subject: [PATCH 163/170] Fix NonDeterministicRecruitment from worker usage filter and RNG drift in recruitment --- .../clustercontroller/ClusterController.cpp | 61 ++++++++++++++++ .../clustercontroller/ClusterController.h | 71 ++++++++++++++++--- 2 files changed, 122 insertions(+), 10 deletions(-) diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index 2a25161468a..04463bb17bd 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -5324,6 +5324,67 @@ TEST_CASE("/fdbserver/clustercontroller/proxyRecruitmentSpansFitnessLevels") { return Void(); } +// Regression test for NonDeterministicRecruitment: findWorkersForConfiguration() recruits the +// configuration twice in simulation and requires both recruitments to have equal RoleFitness +// (which includes the worst usage of the recruited workers). When the candidate filter ignored +// usage entirely, the two recruitments could pick equal-fitness processes with different usage +// (e.g. GrvProxy fitness "2 2 2 0 2" vs "2 3 2 0 2"), failing the check. Candidates are now +// compared against the usage snapshot taken when the first worker was selected, so repeated +// recruitments admit the same candidate set even as the live id_used counter keeps advancing. +TEST_CASE("/fdbserver/clustercontroller/proxyRecruitmentDeterministicUsage") { + const Key dcId = "dc1"_sr; + ClusterControllerData data = makeRecruitmentTestData(dcId); + + constexpr int kCount = 3; + auto stateless = addRecruitmentTestWorkers(data, dcId, ProcessClass::StatelessClass, "sl"_sr, kCount); + + // Two of the three stateless processes already host the master and cluster controller. + data.masterProcessId = stateless[0].locality.processId(); + data.clusterControllerProcessId = stateless[1].locality.processId(); + + DatabaseConfiguration config; + config.initialized = true; + + ClusterControllerData::RoleFitness firstFit; + ClusterControllerData::RoleFitness secondFit; + for (int pass = 0; pass < 2; pass++) { + // Each pass re-recruits from scratch with the same initial id_used, mimicking the + // two-pass determinism check in findWorkersForConfiguration. + std::map>, int> id_used; + data.updateKnownIds(&id_used); + + auto first = + data.getWorkerForRoleInDatacenter(dcId, recruitment::GrvProxy, recruitment::ExcludeFit, config, id_used); + auto proxies = + data.getWorkersForRoleInDatacenter(dcId, recruitment::GrvProxy, kCount, config, id_used, {}, first); + + // The pool is not artificially cut: every proxy lands on a distinct stateless process. + ASSERT_EQ(proxies.size(), kCount); + std::set>> pids; + for (const auto& w : proxies) { + pids.insert(w.interf.locality.processId()); + } + ASSERT_EQ(pids.size(), kCount); + + // Mimic findWorkersForConfiguration's comparison accounting: usage of the recruited + // workers is added on top of the initial id_used before computing the fitness. + std::map>, int> compareUsed; + data.updateKnownIds(&compareUsed); + for (const auto& w : proxies) { + compareUsed[w.interf.locality.processId()]++; + } + ClusterControllerData::RoleFitness fit(proxies, recruitment::GrvProxy, compareUsed); + if (pass == 0) { + firstFit = fit; + } else { + secondFit = fit; + } + } + + ASSERT(firstFit == secondFit); + return Void(); +} + // Without a minWorker there is no fitness ceiling, so recruitment fills across fitness levels // from the best available. This locks in that the fix (which gates on the minWorker's fitness) // does not change the no-minWorker path used by log-router recruitment. diff --git a/fdbserver/clustercontroller/ClusterController.h b/fdbserver/clustercontroller/ClusterController.h index 489245887ae..1b627dd2254 100644 --- a/fdbserver/clustercontroller/ClusterController.h +++ b/fdbserver/clustercontroller/ClusterController.h @@ -106,6 +106,11 @@ struct WorkerFitnessInfo { WorkerDetails worker; recruitment::Fitness fitness; int used; + // Snapshot of the id_used map taken when this worker was selected. Callers that accept a + // minWorker compare candidates against this snapshot rather than the live id_used map, so + // that repeated recruitments of the same configuration compare against the same reference + // point instead of a counter that keeps advancing between selections. + std::map>, int> idUsedSnapshot; WorkerFitnessInfo() : fitness(recruitment::NeverAssign), used(0) {} WorkerFitnessInfo(WorkerDetails worker, recruitment::Fitness fitness, int used) @@ -1508,9 +1513,11 @@ class ClusterControllerData { if (!fitness_workers.empty()) { auto worker = deterministicRandom()->randomChoice(fitness_workers.begin()->second); id_used[worker.interf.locality.processId()]++; - return WorkerFitnessInfo(worker, + WorkerFitnessInfo result(worker, std::max(recruitment::GoodFit, std::get<0>(fitness_workers.begin()->first)), std::get<1>(fitness_workers.begin()->first)); + result.idUsedSnapshot = id_used; + return result; } throw no_more_servers(); @@ -1544,20 +1551,38 @@ class ClusterControllerData { return results; } + // minWorker's usage is taken at its selection time (before its own increment), so it is + // the usage level of the least-loaded equal-fitness processes. Candidates may be one role + // more loaded than that, which keeps processes that already host the master or cluster + // controller eligible without letting usage grow unboundedly. + Optional minWorkerSnapshotUsed; + if (minWorker.present() && !minWorker.get().idUsedSnapshot.empty()) { + minWorkerSnapshotUsed = minWorker.get().used + 1; + } + for (auto& it : id_worker) { auto fitness = recruitment::machineClassFitness(it.second.details.processClass, role); - // Candidates must not be worse than the already-accepted minWorker. Usage is - // deliberately NOT part of this check: comparing a candidate's usage against - // minWorker's emptied the pool whenever the desired count exceeded the - // least-used equal-fitness processes (e.g. every stateless process when some - // already host master/cluster controller). Spreading across processes is still - // provided by the bucket ordering on `used` in the fill loop below. + // Candidates must not be worse than the already-accepted minWorker. "Worse" means a + // worse fitness class or a higher usage than the minWorker had when it was selected. + // Usage is compared against the snapshot taken at minWorker selection time, not the + // live id_used counter, because the live counter keeps advancing as earlier roles and + // datacenters are recruited. Comparing against it would admit different candidate + // sets in repeated recruitments of the same configuration (breaking recruitment + // determinism), while comparing against minWorker.used alone empties the pool + // whenever the desired count exceeds the least-used equal-fitness processes (e.g. + // every stateless process when some already host master/cluster controller). + // Spreading across processes is still provided by the bucket ordering on `used` in + // the fill loop below. if (workerAvailable(it.second, checkStable) && !conf.isExcludedServer(it.second.details.interf.addresses(), it.second.details.interf.locality) && !isExcludedDegradedServer(it.second.details.interf.addresses()) && it.second.details.interf.locality.dcId() == dcId && - (!minWorker.present() || (it.second.details.interf.id() != minWorker.get().worker.interf.id() && - fitness <= minWorker.get().fitness))) { + (!minWorker.present() || + (it.second.details.interf.id() != minWorker.get().worker.interf.id() && + (fitness < minWorker.get().fitness || + (fitness == minWorker.get().fitness && + (!minWorkerSnapshotUsed.present() || + minWorker.get().idUsedSnapshot[it.first] <= minWorkerSnapshotUsed.get())))))) { auto sharing = preferredSharing.find(it.first); fitness_workers[{ fitness, id_used[it.first], @@ -2260,14 +2285,36 @@ class ClusterControllerData { RoleFitness secondFitness(secondDetails, role, secondUsed); if (!(firstFitness == secondFitness)) { + auto describe = [&](const std::vector& details, + const std::map>, int>& used) { + std::string s; + // Cap the dump so the trace event stays well under the size limit. + const int n = std::min(details.size(), 8); + for (int i = 0; i < n; i++) { + auto pid = details[i].interf.locality.processId(); + auto u = used.find(pid); + s += "(" + pid.get().toString() + ",fit=" + + std::to_string((int)recruitment::machineClassFitness(details[i].processClass, role)) + + ",used=" + (u != used.end() ? std::to_string(u->second) : std::string("?")) + ") "; + } + if ((int)details.size() > n) { + s += "...(" + std::to_string(details.size()) + " total)"; + } + return s; + }; TraceEvent(SevError, "NonDeterministicRecruitment") .detail("FirstFitness", firstFitness.toString()) .detail("SecondFitness", secondFitness.toString()) - .detail("ClusterRole", role); + .detail("ClusterRole", role) + .detail("FirstWorkers", describe(firstDetails, firstUsed)) + .detail("SecondWorkers", describe(secondDetails, secondUsed)); } } RecruitFromConfigurationReply findWorkersForConfiguration(RecruitFromConfigurationRequest const& req) { + // Capture the RNG state so the second determinism-check pass below can replay the + // recruitment from the same starting point. + uint64_t randomState = deterministicRandom()->peek(); RecruitFromConfigurationReply rep = findWorkersForConfigurationDispatch(req, true); if (g_network->isSimulated()) { try { @@ -2286,6 +2333,10 @@ class ClusterControllerData { } } if (!remoteDCUsedAsSatellite) { + // Replay the recruitment from the same RNG state so both passes of the + // determinism check consume an identical random sequence and any divergence + // is attributable to non-deterministic logic rather than RNG drift. + deterministicRandom()->resetSeed(randomState); RecruitFromConfigurationReply compare = findWorkersForConfigurationDispatch(req, false); std::map>, int> firstUsed; From c1ae0e5d9fc9c118693926178f43ee99b8f45a67 Mon Sep 17 00:00:00 2001 From: Mark Shabanov Date: Thu, 10 Sep 2026 13:22:13 +0300 Subject: [PATCH 164/170] Add RNG state save restore and use it for the recruitment determinism check --- .../clustercontroller/ClusterController.cpp | 12 ++-- .../clustercontroller/ClusterController.h | 63 +++++++------------ flow/DeterministicRandom.cpp | 53 ++++++++++++++++ flow/include/flow/DeterministicRandom.h | 2 + flow/include/flow/IRandom.h | 9 +++ 5 files changed, 92 insertions(+), 47 deletions(-) diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index 04463bb17bd..a21b6694919 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -5325,12 +5325,12 @@ TEST_CASE("/fdbserver/clustercontroller/proxyRecruitmentSpansFitnessLevels") { } // Regression test for NonDeterministicRecruitment: findWorkersForConfiguration() recruits the -// configuration twice in simulation and requires both recruitments to have equal RoleFitness -// (which includes the worst usage of the recruited workers). When the candidate filter ignored -// usage entirely, the two recruitments could pick equal-fitness processes with different usage -// (e.g. GrvProxy fitness "2 2 2 0 2" vs "2 3 2 0 2"), failing the check. Candidates are now -// compared against the usage snapshot taken when the first worker was selected, so repeated -// recruitments admit the same candidate set even as the live id_used counter keeps advancing. +// configuration twice in simulation and requires both recruitments to produce equal RoleFitness +// (which includes the worst usage of the recruited workers). The determinism check replays the +// random sequence of the first pass, so equal fitness here additionally means the candidate +// filter did not starve the pool: with the master and cluster controller occupying two of the +// three stateless processes, every proxy must still land on a distinct process, and both passes +// must agree on the worst usage that results. TEST_CASE("/fdbserver/clustercontroller/proxyRecruitmentDeterministicUsage") { const Key dcId = "dc1"_sr; ClusterControllerData data = makeRecruitmentTestData(dcId); diff --git a/fdbserver/clustercontroller/ClusterController.h b/fdbserver/clustercontroller/ClusterController.h index 1b627dd2254..05a1749703e 100644 --- a/fdbserver/clustercontroller/ClusterController.h +++ b/fdbserver/clustercontroller/ClusterController.h @@ -23,6 +23,7 @@ #include #include #include +#include #include "fdbclient/DatabaseContext.h" #include "fdbclient/ProcessClass.h" @@ -106,11 +107,6 @@ struct WorkerFitnessInfo { WorkerDetails worker; recruitment::Fitness fitness; int used; - // Snapshot of the id_used map taken when this worker was selected. Callers that accept a - // minWorker compare candidates against this snapshot rather than the live id_used map, so - // that repeated recruitments of the same configuration compare against the same reference - // point instead of a counter that keeps advancing between selections. - std::map>, int> idUsedSnapshot; WorkerFitnessInfo() : fitness(recruitment::NeverAssign), used(0) {} WorkerFitnessInfo(WorkerDetails worker, recruitment::Fitness fitness, int used) @@ -1513,11 +1509,9 @@ class ClusterControllerData { if (!fitness_workers.empty()) { auto worker = deterministicRandom()->randomChoice(fitness_workers.begin()->second); id_used[worker.interf.locality.processId()]++; - WorkerFitnessInfo result(worker, + return WorkerFitnessInfo(worker, std::max(recruitment::GoodFit, std::get<0>(fitness_workers.begin()->first)), std::get<1>(fitness_workers.begin()->first)); - result.idUsedSnapshot = id_used; - return result; } throw no_more_servers(); @@ -1551,38 +1545,20 @@ class ClusterControllerData { return results; } - // minWorker's usage is taken at its selection time (before its own increment), so it is - // the usage level of the least-loaded equal-fitness processes. Candidates may be one role - // more loaded than that, which keeps processes that already host the master or cluster - // controller eligible without letting usage grow unboundedly. - Optional minWorkerSnapshotUsed; - if (minWorker.present() && !minWorker.get().idUsedSnapshot.empty()) { - minWorkerSnapshotUsed = minWorker.get().used + 1; - } - for (auto& it : id_worker) { auto fitness = recruitment::machineClassFitness(it.second.details.processClass, role); - // Candidates must not be worse than the already-accepted minWorker. "Worse" means a - // worse fitness class or a higher usage than the minWorker had when it was selected. - // Usage is compared against the snapshot taken at minWorker selection time, not the - // live id_used counter, because the live counter keeps advancing as earlier roles and - // datacenters are recruited. Comparing against it would admit different candidate - // sets in repeated recruitments of the same configuration (breaking recruitment - // determinism), while comparing against minWorker.used alone empties the pool - // whenever the desired count exceeds the least-used equal-fitness processes (e.g. - // every stateless process when some already host master/cluster controller). - // Spreading across processes is still provided by the bucket ordering on `used` in - // the fill loop below. + // Candidates must not be worse than the already-accepted minWorker. Usage is + // deliberately not part of this check: gating on it empties the pool whenever the + // desired count exceeds the number of least-used equal-fitness processes (e.g. + // every stateless process, when some of them already host the master or cluster + // controller). Spreading across processes is instead provided by the bucket + // ordering on `used` in the fill loop below. if (workerAvailable(it.second, checkStable) && !conf.isExcludedServer(it.second.details.interf.addresses(), it.second.details.interf.locality) && !isExcludedDegradedServer(it.second.details.interf.addresses()) && it.second.details.interf.locality.dcId() == dcId && - (!minWorker.present() || - (it.second.details.interf.id() != minWorker.get().worker.interf.id() && - (fitness < minWorker.get().fitness || - (fitness == minWorker.get().fitness && - (!minWorkerSnapshotUsed.present() || - minWorker.get().idUsedSnapshot[it.first] <= minWorkerSnapshotUsed.get())))))) { + (!minWorker.present() || (it.second.details.interf.id() != minWorker.get().worker.interf.id() && + fitness <= minWorker.get().fitness))) { auto sharing = preferredSharing.find(it.first); fitness_workers[{ fitness, id_used[it.first], @@ -2312,9 +2288,12 @@ class ClusterControllerData { } RecruitFromConfigurationReply findWorkersForConfiguration(RecruitFromConfigurationRequest const& req) { - // Capture the RNG state so the second determinism-check pass below can replay the - // recruitment from the same starting point. - uint64_t randomState = deterministicRandom()->peek(); + // Snapshot the RNG state before the first pass so the determinism-check pass below can + // replay the recruitment from the same starting point. Only the simulation check needs + // it, so production runs avoid the snapshot cost entirely. + Optional> savedRandomState; + if (g_network->isSimulated()) + savedRandomState = deterministicRandom()->saveState(); RecruitFromConfigurationReply rep = findWorkersForConfigurationDispatch(req, true); if (g_network->isSimulated()) { try { @@ -2333,10 +2312,12 @@ class ClusterControllerData { } } if (!remoteDCUsedAsSatellite) { - // Replay the recruitment from the same RNG state so both passes of the - // determinism check consume an identical random sequence and any divergence - // is attributable to non-deterministic logic rather than RNG drift. - deterministicRandom()->resetSeed(randomState); + // Replay the recruitment from the identical RNG state captured before the + // first pass, so both passes consume the same random sequence. Recruitment + // randomizes deliberately (e.g. among equal-fitness candidates), so without + // this the two passes would diverge on RNG drift alone; with it, any + // divergence is attributable to non-deterministic logic. + deterministicRandom()->restoreState(savedRandomState.get()); RecruitFromConfigurationReply compare = findWorkersForConfigurationDispatch(req, false); std::map>, int> firstUsed; diff --git a/flow/DeterministicRandom.cpp b/flow/DeterministicRandom.cpp index 8a3079dfce7..2a1cf14e997 100644 --- a/flow/DeterministicRandom.cpp +++ b/flow/DeterministicRandom.cpp @@ -24,6 +24,7 @@ #include "flow/UnitTest.h" #include +#include uint64_t DeterministicRandom::gen64() { uint64_t curr = next; @@ -158,6 +159,22 @@ void DeterministicRandom::resetSeed(uint64_t seed) { next = rng(); } +std::vector DeterministicRandom::saveState() const { + // `next` must be captured along with the engine: gen64() returns the previously drawn + // value and prefetches the next one, so the engine leads the observable stream by one step. + std::ostringstream ss; + ss << rng << " " << next; + const std::string s = ss.str(); + return std::vector(s.begin(), s.end()); +} + +void DeterministicRandom::restoreState(std::vector const& state) { + if (state.empty()) + return; + std::istringstream ss(std::string(state.begin(), state.end())); + ss >> rng >> next; +} + void DeterministicRandom::addref() { ReferenceCounted::addref(); } @@ -265,3 +282,39 @@ TEST_CASE("/flow/DeterministicRandom/truePercent") { return Void(); } + +TEST_CASE("/flow/DeterministicRandom/saveRestoreState") { + DeterministicRandom rng(1234567); + // Move off the freshly seeded state so the snapshot covers a mid-stream engine state. + for (int i = 0; i < 100; ++i) + rng.randomUInt64(); + + auto const saved = rng.saveState(); + ASSERT(!saved.empty()); + + std::vector expected; + for (int i = 0; i < 64; ++i) + expected.push_back(rng.randomUInt64()); + + // Restoring in place must replay the same stream. + rng.restoreState(saved); + for (int i = 0; i < 64; ++i) + ASSERT_EQ(expected[i], rng.randomUInt64()); + + // The snapshot must fully determine the stream, so restoring it into an unrelated + // generator must yield the same values. + DeterministicRandom other(987654321); + other.restoreState(saved); + for (int i = 0; i < 64; ++i) + ASSERT_EQ(expected[i], other.randomUInt64()); + + // An empty snapshot is a no-op and must leave the generator's stream untouched. + DeterministicRandom untouched(42); + untouched.randomUInt64(); + untouched.restoreState(std::vector()); + DeterministicRandom reference(42); + reference.randomUInt64(); + ASSERT_EQ(reference.randomUInt64(), untouched.randomUInt64()); + + return Void(); +} diff --git a/flow/include/flow/DeterministicRandom.h b/flow/include/flow/DeterministicRandom.h index 298b2365e63..86c38a5f1ac 100644 --- a/flow/include/flow/DeterministicRandom.h +++ b/flow/include/flow/DeterministicRandom.h @@ -65,6 +65,8 @@ class SWIFT_CXX_REF_DETERMINISTICRANDOM DeterministicRandom final : public IRand bool truePercent(const int percent) override; uint64_t peek() const override; void resetSeed(uint64_t seed) override; // Reset the random number generator with a new seed + std::vector saveState() const override; + void restoreState(std::vector const& state) override; void addref() override; void delref() override; }; diff --git a/flow/include/flow/IRandom.h b/flow/include/flow/IRandom.h index 9d7eaf637ff..08072367ee2 100644 --- a/flow/include/flow/IRandom.h +++ b/flow/include/flow/IRandom.h @@ -34,6 +34,7 @@ #endif #include #include +#include // Until we move to C++20, we'll need something to take the place of operator<=>. // This is as good a place as any, I guess. @@ -168,6 +169,14 @@ class IRandom { // Reset the random number generator with a new seed (only supported by deterministic generators) virtual void resetSeed(uint64_t seed) {} + // Opaque snapshot of the generator's full internal state. Callers use it to replay a code + // path from a known point (for example, a simulation determinism check that runs the same + // recruitment twice) without perturbing the global random stream by consuming it twice. + // Only deterministic generators implement this; the default reports an empty state, and + // restoreState() on such a generator is a no-op. + virtual std::vector saveState() const { return {}; } + virtual void restoreState(std::vector const& state) {} + virtual void addref() = 0; virtual void delref() = 0; From 32ec33b2f72de463cacfdba6e3e574de25e308bd Mon Sep 17 00:00:00 2001 From: Mark Shabanov Date: Fri, 11 Sep 2026 12:31:27 +0300 Subject: [PATCH 165/170] fix clang-tidy --- flow/DeterministicRandom.cpp | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/flow/DeterministicRandom.cpp b/flow/DeterministicRandom.cpp index 2a1cf14e997..cc2e7b60ecd 100644 --- a/flow/DeterministicRandom.cpp +++ b/flow/DeterministicRandom.cpp @@ -292,9 +292,9 @@ TEST_CASE("/flow/DeterministicRandom/saveRestoreState") { auto const saved = rng.saveState(); ASSERT(!saved.empty()); - std::vector expected; + std::vector expected(64); for (int i = 0; i < 64; ++i) - expected.push_back(rng.randomUInt64()); + expected[i] = rng.randomUInt64(); // Restoring in place must replay the same stream. rng.restoreState(saved); From 3e870c15efe8fe663d6b1a615c94f4f714089772 Mon Sep 17 00:00:00 2001 From: Mark Shabanov Date: Wed, 16 Sep 2026 15:30:42 +0300 Subject: [PATCH 166/170] Replace RNG saveState/restoreState with relaxed deterministic recruitment check from equality to no-regression --- .../clustercontroller/ClusterController.h | 15 +----- flow/DeterministicRandom.cpp | 53 ------------------- flow/include/flow/DeterministicRandom.h | 2 - flow/include/flow/IRandom.h | 9 ---- 4 files changed, 1 insertion(+), 78 deletions(-) diff --git a/fdbserver/clustercontroller/ClusterController.h b/fdbserver/clustercontroller/ClusterController.h index 05a1749703e..a33013236fb 100644 --- a/fdbserver/clustercontroller/ClusterController.h +++ b/fdbserver/clustercontroller/ClusterController.h @@ -23,7 +23,6 @@ #include #include #include -#include #include "fdbclient/DatabaseContext.h" #include "fdbclient/ProcessClass.h" @@ -2260,7 +2259,7 @@ class ClusterControllerData { } RoleFitness secondFitness(secondDetails, role, secondUsed); - if (!(firstFitness == secondFitness)) { + if (firstFitness < secondFitness) { // second pass produced a worse result → regression auto describe = [&](const std::vector& details, const std::map>, int>& used) { std::string s; @@ -2288,12 +2287,6 @@ class ClusterControllerData { } RecruitFromConfigurationReply findWorkersForConfiguration(RecruitFromConfigurationRequest const& req) { - // Snapshot the RNG state before the first pass so the determinism-check pass below can - // replay the recruitment from the same starting point. Only the simulation check needs - // it, so production runs avoid the snapshot cost entirely. - Optional> savedRandomState; - if (g_network->isSimulated()) - savedRandomState = deterministicRandom()->saveState(); RecruitFromConfigurationReply rep = findWorkersForConfigurationDispatch(req, true); if (g_network->isSimulated()) { try { @@ -2312,12 +2305,6 @@ class ClusterControllerData { } } if (!remoteDCUsedAsSatellite) { - // Replay the recruitment from the identical RNG state captured before the - // first pass, so both passes consume the same random sequence. Recruitment - // randomizes deliberately (e.g. among equal-fitness candidates), so without - // this the two passes would diverge on RNG drift alone; with it, any - // divergence is attributable to non-deterministic logic. - deterministicRandom()->restoreState(savedRandomState.get()); RecruitFromConfigurationReply compare = findWorkersForConfigurationDispatch(req, false); std::map>, int> firstUsed; diff --git a/flow/DeterministicRandom.cpp b/flow/DeterministicRandom.cpp index cc2e7b60ecd..8a3079dfce7 100644 --- a/flow/DeterministicRandom.cpp +++ b/flow/DeterministicRandom.cpp @@ -24,7 +24,6 @@ #include "flow/UnitTest.h" #include -#include uint64_t DeterministicRandom::gen64() { uint64_t curr = next; @@ -159,22 +158,6 @@ void DeterministicRandom::resetSeed(uint64_t seed) { next = rng(); } -std::vector DeterministicRandom::saveState() const { - // `next` must be captured along with the engine: gen64() returns the previously drawn - // value and prefetches the next one, so the engine leads the observable stream by one step. - std::ostringstream ss; - ss << rng << " " << next; - const std::string s = ss.str(); - return std::vector(s.begin(), s.end()); -} - -void DeterministicRandom::restoreState(std::vector const& state) { - if (state.empty()) - return; - std::istringstream ss(std::string(state.begin(), state.end())); - ss >> rng >> next; -} - void DeterministicRandom::addref() { ReferenceCounted::addref(); } @@ -282,39 +265,3 @@ TEST_CASE("/flow/DeterministicRandom/truePercent") { return Void(); } - -TEST_CASE("/flow/DeterministicRandom/saveRestoreState") { - DeterministicRandom rng(1234567); - // Move off the freshly seeded state so the snapshot covers a mid-stream engine state. - for (int i = 0; i < 100; ++i) - rng.randomUInt64(); - - auto const saved = rng.saveState(); - ASSERT(!saved.empty()); - - std::vector expected(64); - for (int i = 0; i < 64; ++i) - expected[i] = rng.randomUInt64(); - - // Restoring in place must replay the same stream. - rng.restoreState(saved); - for (int i = 0; i < 64; ++i) - ASSERT_EQ(expected[i], rng.randomUInt64()); - - // The snapshot must fully determine the stream, so restoring it into an unrelated - // generator must yield the same values. - DeterministicRandom other(987654321); - other.restoreState(saved); - for (int i = 0; i < 64; ++i) - ASSERT_EQ(expected[i], other.randomUInt64()); - - // An empty snapshot is a no-op and must leave the generator's stream untouched. - DeterministicRandom untouched(42); - untouched.randomUInt64(); - untouched.restoreState(std::vector()); - DeterministicRandom reference(42); - reference.randomUInt64(); - ASSERT_EQ(reference.randomUInt64(), untouched.randomUInt64()); - - return Void(); -} diff --git a/flow/include/flow/DeterministicRandom.h b/flow/include/flow/DeterministicRandom.h index 86c38a5f1ac..298b2365e63 100644 --- a/flow/include/flow/DeterministicRandom.h +++ b/flow/include/flow/DeterministicRandom.h @@ -65,8 +65,6 @@ class SWIFT_CXX_REF_DETERMINISTICRANDOM DeterministicRandom final : public IRand bool truePercent(const int percent) override; uint64_t peek() const override; void resetSeed(uint64_t seed) override; // Reset the random number generator with a new seed - std::vector saveState() const override; - void restoreState(std::vector const& state) override; void addref() override; void delref() override; }; diff --git a/flow/include/flow/IRandom.h b/flow/include/flow/IRandom.h index 08072367ee2..9d7eaf637ff 100644 --- a/flow/include/flow/IRandom.h +++ b/flow/include/flow/IRandom.h @@ -34,7 +34,6 @@ #endif #include #include -#include // Until we move to C++20, we'll need something to take the place of operator<=>. // This is as good a place as any, I guess. @@ -169,14 +168,6 @@ class IRandom { // Reset the random number generator with a new seed (only supported by deterministic generators) virtual void resetSeed(uint64_t seed) {} - // Opaque snapshot of the generator's full internal state. Callers use it to replay a code - // path from a known point (for example, a simulation determinism check that runs the same - // recruitment twice) without perturbing the global random stream by consuming it twice. - // Only deterministic generators implement this; the default reports an empty state, and - // restoreState() on such a generator is a no-op. - virtual std::vector saveState() const { return {}; } - virtual void restoreState(std::vector const& state) {} - virtual void addref() = 0; virtual void delref() = 0; From 2e8fd3e9c1ced4a799cc6b8a4cb9b282cb93961d Mon Sep 17 00:00:00 2001 From: Mark Shabanov Date: Wed, 16 Sep 2026 16:30:21 +0300 Subject: [PATCH 167/170] Relax NonDeterministicRecruitment check: exclude worstUsed from comparison --- fdbserver/clustercontroller/ClusterController.h | 13 ++++++++++++- 1 file changed, 12 insertions(+), 1 deletion(-) diff --git a/fdbserver/clustercontroller/ClusterController.h b/fdbserver/clustercontroller/ClusterController.h index a33013236fb..321dd31e3ee 100644 --- a/fdbserver/clustercontroller/ClusterController.h +++ b/fdbserver/clustercontroller/ClusterController.h @@ -2259,7 +2259,18 @@ class ClusterControllerData { } RoleFitness secondFitness(secondDetails, role, secondUsed); - if (firstFitness < secondFitness) { // second pass produced a worse result → regression + auto worseIgnoringWorstUsed = [](const RoleFitness& a, const RoleFitness& b) { + if (a.worstFit != b.worstFit) + return a.worstFit < b.worstFit; + if (a.count != b.count) + return a.count > b.count; + if (a.degraded != b.degraded) + return b.degraded; + if (a.role != recruitment::TLog && a.role != recruitment::LogRouter && a.bestFit != b.bestFit) + return a.bestFit < b.bestFit; + return false; + }; + if (worseIgnoringWorstUsed(firstFitness, secondFitness)) { // second pass produced a worse result -> regression auto describe = [&](const std::vector& details, const std::map>, int>& used) { std::string s; From 25602d26380a645f6e918b6ac831015b00d97490 Mon Sep 17 00:00:00 2001 From: Mark Shabanov Date: Thu, 17 Sep 2026 12:00:11 +0300 Subject: [PATCH 168/170] Reseed deterministic RNG so both recruitment passes start from the same state --- .../clustercontroller/ClusterController.h | 25 ++++++++++--------- 1 file changed, 13 insertions(+), 12 deletions(-) diff --git a/fdbserver/clustercontroller/ClusterController.h b/fdbserver/clustercontroller/ClusterController.h index 321dd31e3ee..a6dc44d4423 100644 --- a/fdbserver/clustercontroller/ClusterController.h +++ b/fdbserver/clustercontroller/ClusterController.h @@ -2259,18 +2259,7 @@ class ClusterControllerData { } RoleFitness secondFitness(secondDetails, role, secondUsed); - auto worseIgnoringWorstUsed = [](const RoleFitness& a, const RoleFitness& b) { - if (a.worstFit != b.worstFit) - return a.worstFit < b.worstFit; - if (a.count != b.count) - return a.count > b.count; - if (a.degraded != b.degraded) - return b.degraded; - if (a.role != recruitment::TLog && a.role != recruitment::LogRouter && a.bestFit != b.bestFit) - return a.bestFit < b.bestFit; - return false; - }; - if (worseIgnoringWorstUsed(firstFitness, secondFitness)) { // second pass produced a worse result -> regression + if (!(firstFitness == secondFitness)) { auto describe = [&](const std::vector& details, const std::map>, int>& used) { std::string s; @@ -2298,8 +2287,20 @@ class ClusterControllerData { } RecruitFromConfigurationReply findWorkersForConfiguration(RecruitFromConfigurationRequest const& req) { + // The determinism check below re-runs recruitment and compares the result against the first + // pass. Recruitment deliberately randomizes (randomShuffle/randomChoice among equal candidates), + // so both passes must start from the same RNG state or they would trivially disagree. Seed the + // generator before the first pass, then reseed it from the same value before the replay, so both + // passes draw an identical random sequence. This is simulation-only; production runs are unaffected + // because the generator there is not seeded deterministically. + uint64_t seed = 0; + if (g_network->isSimulated()) { + seed = deterministicRandom()->randomUInt64(); + deterministicRandom()->resetSeed(seed); + } RecruitFromConfigurationReply rep = findWorkersForConfigurationDispatch(req, true); if (g_network->isSimulated()) { + deterministicRandom()->resetSeed(seed); try { // FIXME: The logic to pick a satellite in a remote region is not // deterministic and can therefore break this nondeterminism check. From 350a8a087728f07a159f4850a9b816c5cddf84c8 Mon Sep 17 00:00:00 2001 From: Mark Shabanov Date: Sun, 20 Sep 2026 17:20:30 +0300 Subject: [PATCH 169/170] Fix NonDeterministicRecruitment: isolate the RNG sample when comparing TLog. --- .../clustercontroller/ClusterController.h | 40 ++++++++++++++++--- 1 file changed, 34 insertions(+), 6 deletions(-) diff --git a/fdbserver/clustercontroller/ClusterController.h b/fdbserver/clustercontroller/ClusterController.h index a6dc44d4423..db3d95f4524 100644 --- a/fdbserver/clustercontroller/ClusterController.h +++ b/fdbserver/clustercontroller/ClusterController.h @@ -22,6 +22,7 @@ #include #include +#include #include #include "fdbclient/DatabaseContext.h" @@ -1283,9 +1284,15 @@ class ClusterControllerData { exclusionWorkerIds); if (g_network->isSimulated()) { + // The comparison below validates the TLog method, not recruitment, so its draws must + // not leak into the shared stream: the replay pass of the determinism check would + // otherwise continue from a different stream and report a divergence the + // recruitment did not cause. + uint64_t entryFingerprint = deterministicRandom()->peek(); try { auto testWorkers = getWorkersForTlogsBackup( conf, required, desired, policy, testUsed, checkStable, dcIds, exclusionWorkerIds); + deterministicRandom()->resetSeed(entryFingerprint); RoleFitness testFitness(testWorkers, recruitment::TLog, testUsed); RoleFitness fitness(workers, recruitment::TLog, id_used); @@ -1320,6 +1327,7 @@ class ClusterControllerData { ASSERT(false); } } catch (Error& e) { + deterministicRandom()->resetSeed(entryFingerprint); ASSERT(false); // Simulation only validation should not throw errors } } @@ -1340,9 +1348,14 @@ class ClusterControllerData { getWorkersForTlogsSimple(conf, required, desired, id_used, checkStable, dcIds, exclusionWorkerIds); if (g_network->isSimulated()) { + // The comparison below validates the TLog method, not recruitment, so its draws must not + // leak into the shared stream: the replay pass of the determinism check would otherwise + // continue from a different stream and report a divergence the recruitment did not cause. + uint64_t entryFingerprint = deterministicRandom()->peek(); try { auto testWorkers = getWorkersForTlogsBackup( conf, required, desired, policy, testUsed, checkStable, dcIds, exclusionWorkerIds); + deterministicRandom()->resetSeed(entryFingerprint); RoleFitness testFitness(testWorkers, recruitment::TLog, testUsed); RoleFitness fitness(workers, recruitment::TLog, id_used); // backup recruitment is not required to use degraded processes that have better fitness @@ -1362,6 +1375,7 @@ class ClusterControllerData { ASSERT(false); } } catch (Error& e) { + deterministicRandom()->resetSeed(entryFingerprint); ASSERT(false); // Simulation only validation should not throw errors } } @@ -2238,28 +2252,36 @@ class ClusterControllerData { recruitment::ClusterRole role, std::string description) { std::vector firstDetails; + std::set>> firstPids; for (auto& worker : first) { auto w = id_worker.find(worker.locality.processId()); ASSERT(w != id_worker.end()); auto const& [_, workerInfo] = *w; ASSERT(!conf.isExcludedServer(workerInfo.details.interf.addresses(), workerInfo.details.interf.locality)); firstDetails.push_back(workerInfo.details); + firstPids.insert(worker.locality.processId()); //TraceEvent("CompareAddressesFirst").detail(description.c_str(), w->second.details.interf.address()); } RoleFitness firstFitness(firstDetails, role, firstUsed); std::vector secondDetails; + std::set>> secondPids; for (auto& worker : second) { auto w = id_worker.find(worker.locality.processId()); ASSERT(w != id_worker.end()); auto const& [_, workerInfo] = *w; ASSERT(!conf.isExcludedServer(workerInfo.details.interf.addresses(), workerInfo.details.interf.locality)); secondDetails.push_back(workerInfo.details); + secondPids.insert(worker.locality.processId()); //TraceEvent("CompareAddressesSecond").detail(description.c_str(), w->second.details.interf.address()); } RoleFitness secondFitness(secondDetails, role, secondUsed); - if (!(firstFitness == secondFitness)) { + // Compare the recruited process sets as well as the fitness summary: the summary alone can + // coincide for different sets, which would hide the divergence here and instead surface + // against a later role whose fitness is computed from the usage this role left behind. + bool sameWorkerSet = firstPids == secondPids; + if (!sameWorkerSet || !(firstFitness == secondFitness)) { auto describe = [&](const std::vector& details, const std::map>, int>& used) { std::string s; @@ -2268,7 +2290,7 @@ class ClusterControllerData { for (int i = 0; i < n; i++) { auto pid = details[i].interf.locality.processId(); auto u = used.find(pid); - s += "(" + pid.get().toString() + ",fit=" + + s += "(" + (pid.present() ? pid.get().toString() : std::string("[not set]")) + ",fit=" + std::to_string((int)recruitment::machineClassFitness(details[i].processClass, role)) + ",used=" + (u != used.end() ? std::to_string(u->second) : std::string("?")) + ") "; } @@ -2278,6 +2300,8 @@ class ClusterControllerData { return s; }; TraceEvent(SevError, "NonDeterministicRecruitment") + .detail("Kind", sameWorkerSet ? "Fitness" : "WorkerSet") + .detail("Description", description) .detail("FirstFitness", firstFitness.toString()) .detail("SecondFitness", secondFitness.toString()) .detail("ClusterRole", role) @@ -2337,12 +2361,12 @@ class ClusterControllerData { secondUsed, recruitment::TLog, "Satellite"); + // Each role is compared against the usage map as it stood when that role was + // recruited: roles recruited later (grv proxies, resolvers) must not leak their + // usage into the comparison of an earlier role, which would compare it against a + // placement it never produced and blame it for a later divergence. updateIdUsed(rep.commitProxies, firstUsed); updateIdUsed(compare.commitProxies, secondUsed); - updateIdUsed(rep.grvProxies, firstUsed); - updateIdUsed(compare.grvProxies, secondUsed); - updateIdUsed(rep.resolvers, firstUsed); - updateIdUsed(compare.resolvers, secondUsed); compareWorkers(req.configuration, rep.commitProxies, firstUsed, @@ -2350,6 +2374,8 @@ class ClusterControllerData { secondUsed, recruitment::CommitProxy, "CommitProxy"); + updateIdUsed(rep.grvProxies, firstUsed); + updateIdUsed(compare.grvProxies, secondUsed); compareWorkers(req.configuration, rep.grvProxies, firstUsed, @@ -2357,6 +2383,8 @@ class ClusterControllerData { secondUsed, recruitment::GrvProxy, "GrvProxy"); + updateIdUsed(rep.resolvers, firstUsed); + updateIdUsed(compare.resolvers, secondUsed); compareWorkers(req.configuration, rep.resolvers, firstUsed, From 910d4785c3408532431cf1a24d12b233fc4882fa Mon Sep 17 00:00:00 2001 From: Mark Shabanov Date: Mon, 21 Sep 2026 14:24:51 +0300 Subject: [PATCH 170/170] Defer the DC priority update so it cannot run inside a recruitment pass --- fdbserver/clustercontroller/ClusterController.cpp | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index a21b6694919..3e848c0a720 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -2593,6 +2593,13 @@ Future updatedChangingDatacenters(ClusterControllerData* self) { } co_await onChange; + // React on the next event loop turn instead of in the caller's stack. The body above updates + // worker priorities and completes pending worker registrations, whose continuations can draw + // from the deterministic generator; doing that inside whatever called desiredDcIds.set() puts + // those draws inside unrelated work. The recruitment determinism check is one such caller: a + // draw that only happens on its first pass makes the replay pick different (equally fit) + // workers and fail the check. + co_await delay(0); } }