diff --git a/.clang-tidy b/.clang-tidy index 5dc41eadab3..a54a674c412 100644 --- a/.clang-tidy +++ b/.clang-tidy @@ -1,6 +1,7 @@ --- Checks: > -*, + bugprone-assignment-in-if-condition, bugprone-dangling-handle, bugprone-implicit-widening-of-multiplication-result, bugprone-inaccurate-erase, @@ -13,6 +14,7 @@ Checks: > bugprone-sizeof-expression, bugprone-string-constructor, bugprone-string-integer-assignment, + bugprone-string-literal-with-embedded-nul, bugprone-stringview-nullptr, bugprone-suspicious-memory-comparison, bugprone-suspicious-memset-usage, @@ -32,6 +34,7 @@ Checks: > modernize-use-using, performance-for-range-copy, performance-implicit-conversion-in-loop, + performance-inefficient-vector-operation, performance-move-const-arg, performance-move-constructor-init, readability-braces-around-statements, diff --git a/.github/workflows/tidy.yml b/.github/workflows/tidy.yml index c190bd73f1f..ec8a3e6264d 100644 --- a/.github/workflows/tidy.yml +++ b/.github/workflows/tidy.yml @@ -48,12 +48,6 @@ jobs: fdb_c_generated \ fdb-java - # all flow actors, for generated headers - ACTORS=$( - ninja -t targets all | grep -v /build_output/ | grep '_actors:' | cut -d: -f1 - ) - ninja -v $ACTORS - # all protobuf headers PB_HEADERS=$( ninja -t targets all | grep -v /build_output/ | grep '\.pb\.h:' | cut -d: -f1 @@ -73,9 +67,6 @@ jobs: if [[ $FILE == contrib/* ]]; then continue # skip contrib sources fi - if [[ $FILE == *.actor.* ]]; then - continue # actor syntax is not plain c++ - fi if [[ $FILE == flow/include/flow/CoroutinesImpl.h ]]; then continue # internal implementation header, included only from Coroutines.h fi diff --git a/.gitignore b/.gitignore index fae11ee2e97..341e95f1933 100644 --- a/.gitignore +++ b/.gitignore @@ -87,7 +87,6 @@ foundationdb.VC.db foundationdb.VC.VC.opendb ipch/ compile_commands.json -flow/actorcompiler/obj flow/coveragetool/obj *.code-workspace diff --git a/AGENTS.md b/AGENTS.md index a14fb093860..08b1dde69be 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -62,20 +62,17 @@ Enable simulation tests in cmake: `-DENABLE_SIMULATION_TESTS=ON` FoundationDB is a distributed ordered key-value store with strict serializability. The codebase is organized into ~12 subsystems. For background on subsystems before diving into code, the `design/` directory holds human-authored design docs and `design/AI-generated/` holds subsystem maps and per-subsystem diagrams (start with `design/AI-generated/foundationdb_subsystem_map.md`). -### Concurrency Model: Flow Actors and C++ Coroutines +### Concurrency Model: C++ Coroutines -FDB uses cooperative single-threaded concurrency. Code is written using either: - -- **Flow actors** (`.actor.cpp` / `.actor.h` files): A custom preprocessor (`actorcompiler`) translates `ACTOR`, `state`, `wait()`, `choose/when` syntax into generated C++ state machines. The `#include "flow/actorcompiler.h"` must be the **last** include in actor files. -- **C++ coroutines**: Newer code uses `co_await` and `co_return` instead of `wait()` and actor-style `return`. Coroutines can appear in regular `.cpp` files and in `.actor.cpp` files that still contain actorcompiler input; actors and coroutines can be mixed. New code should use coroutines; see `design/coroutines.md`. +FDB uses cooperative single-threaded concurrency. Asynchronous functions use +standard C++ coroutines with `co_await` and `co_return` in ordinary `.cpp` and +`.h` files. See `design/coroutines.md` for coroutine and cancellation semantics. Key types: `Future`, `Promise`, `PromiseStream`, `Reference` (ref-counted pointer), `Optional`, `ErrorOr`, `Arena` (region-based allocation). #### Common pitfalls -- `wait()` / `waitNext()` cannot appear inside ternary expressions, function arguments, or other sub-expressions. Assign to a `state` variable first, or use a small gating actor. - C++ coroutines: `co_await` is not allowed inside a `catch` handler. Capture the error, exit the catch, then `co_await` outside. -- `ACTOR` functions declared in headers must not be defined inside an anonymous namespace, or call sites get ambiguous-overload errors. - Errors are integer codes (`flow/include/flow/error_definitions.h`), not exceptions with messages. When you `catch (Error& e)`, re-throw `actor_cancelled` (and never silently swallow `broken_promise`) — eating cancellation causes hangs and leaks. Transaction retry goes through `tr.onError(e)`, not a bare loop. - `StringRef`/`KeyRef`/`ValueRef` are non-owning views into an `Arena`. Returning or storing one past its arena's lifetime is a dangling-reference bug; use `Standalone<>` (or `Key`/`Value`) when you need to own the bytes. @@ -83,7 +80,7 @@ Key types: `Future`, `Promise`, `PromiseStream`, `Reference` (ref-co - **`flow/`** — Async runtime, event loop, tracing, deterministic random, arenas - **`fdbrpc/`** — Endpoint-addressed RPC, peer management, failure monitor, Sim2 (deterministic simulation network) -- **`fdbclient/`** — Transaction API (`NativeAPI.actor.cpp`), read-your-writes (`ReadYourWrites.actor.cpp`), location cache, multi-version client +- **`fdbclient/`** — Transaction API (`NativeAPI.cpp`), read-your-writes (`ReadYourWrites.cpp`), location cache, multi-version client - **`fdbserver/`** — All server roles, organized by subdirectory: - `clustercontroller/` — Leader election, role recruitment, ServerDBInfo broadcasting - `coordinator/` — Paxos-based coordination state (generation registers) @@ -134,11 +131,11 @@ Before changing a serialized type that persists on disk, inspect its `serializer `clang-format` is used. Python code uses `black` and `flake8` (pre-commit hooks: `pip install pre-commit && pre-commit install`). -Edit `.actor.cpp` and `.actor.h` sources, not actorcompiler-generated output under the build directory. +Edit tracked `.cpp` and `.h` sources, not generated output under the build directory. ## Source File Headers -Every new `.cpp` / `.h` / `.actor.cpp` / `.actor.h` file starts with the standard Apache 2.0 license block, with the filename on line 2 and the current year on the copyright line. Copy from any existing file in the tree (e.g. `flow/Knobs.cpp`). Add file-purpose comments *after* the license block, not in place of it. +Every new `.cpp` / `.h` file starts with the standard Apache 2.0 license block, with the filename on line 2 and the current year on the copyright line. Copy from any existing file in the tree (e.g. `flow/Knobs.cpp`). Add file-purpose comments *after* the license block, not in place of it. ## Code Comments diff --git a/CMakeLists.txt b/CMakeLists.txt index e4a50c580b7..ea9e7e70b2e 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -179,10 +179,9 @@ endif() include(utils) -# First thing we need is the actor compiler option( WITH_CSHARP - "Prefer C# build tools (actor compiler, coverage tool, vexillographer) when a toolchain is available" + "Prefer C# build tools (coverage tool, vexillographer) when a toolchain is available" ON) set(FDB_USE_CSHARP_TOOLS_EXPLICIT FALSE) @@ -227,8 +226,6 @@ if(FDB_USE_CSHARP_TOOLS_EXPLICIT AND FDB_USE_CSHARP_TOOLS "FDB_USE_CSHARP_TOOLS is enabled, but CSHARP_TOOLCHAIN_FOUND is FALSE. Install .NET (dotnet) or Mono, or set WITH_CSHARP=OFF.") endif() -include(CompileActorCompiler) - if(FDB_USE_CSHARP_TOOLS AND CSHARP_TOOLCHAIN_FOUND) include(CompileCoverageTool) set(COVERAGETOOL_AVAILABLE TRUE) @@ -238,7 +235,6 @@ endif() # Vexilographer generation is configured inside fdbclient include(CompileVexillographer) -# with the actor compiler, we can now make the flow commands available include(FlowCommands) ############################################################################### @@ -346,7 +342,7 @@ if(CMAKE_EXPORT_COMPILE_COMMANDS AND WITH_PYTHON) COMMAND $ ${CMAKE_CURRENT_SOURCE_DIR}/contrib/gen_compile_db.py ARGS -b - ${CMAKE_CURRENT_BINARY_DIR} -s ${CMAKE_CURRENT_SOURCE_DIR} -o + ${CMAKE_CURRENT_BINARY_DIR} -o ${CMAKE_CURRENT_SOURCE_DIR}/compile_commands.json -ninjatool ${CMAKE_MAKE_PROGRAM} ${CMAKE_CURRENT_BINARY_DIR}/compile_commands.json DEPENDS ${CMAKE_CURRENT_SOURCE_DIR}/contrib/gen_compile_db.py @@ -357,8 +353,7 @@ if(CMAKE_EXPORT_COMPILE_COMMANDS AND WITH_PYTHON) DEPENDS ${CMAKE_CURRENT_SOURCE_DIR}/compile_commands.json ${CMAKE_CURRENT_BINARY_DIR}/compile_commands.json) - # A prebuild target ensures that all actors, Swift-generated headers, and - # Swift modules are built. + # A prebuild target ensures that Swift-generated headers and modules are built. if(WITH_SWIFT) add_custom_target(prebuild_for_ide ALL DEPENDS fdbserver_swift processed_compile_commands) diff --git a/README.md b/README.md index 7a0b5c6a608..67b248af25f 100755 --- a/README.md +++ b/README.md @@ -2,7 +2,7 @@ ![Build Status](https://codebuild.us-west-2.amazonaws.com/badges?uuid=eyJlbmNyeXB0ZWREYXRhIjoiVjVzb1RQNUZTaGxGNm9iUnk4OUZ1d09GdTMzZnVOT1YzaUU1RU1xR2o2TENRWFZjb3ZrTHJEcngrZVdnNE40bXJJVDErOGVwendIL3lFWFY3Y3oxQmdjPSIsIml2UGFyYW1ldGVyU3BlYyI6IlJUbWhnaUlJVXRORUNJTjQiLCJtYXRlcmlhbFNldFNlcmlhbCI6MX0%3D&branch=main) -FoundationDB is a distributed database designed to handle large volumes of structured data across clusters of commodity servers. It organizes data as an ordered key-value store and employs ACID transactions for all operations. It is especially well-suited for read/write workloads, but also has excellent performance for write-intensive workloads. Users interact with the database using API language binding. +FoundationDB is a distributed database designed to handle large volumes of structured data across clusters of commodity servers. It organizes data as an ordered key-value store and employs ACID transactions for all operations. It is especially well-suited for read/write workloads, but also has excellent performance for write-intensive workloads. Users interact with the database using API language bindings. To learn more about FoundationDB, visit [foundationdb.org](https://www.foundationdb.org/) @@ -12,17 +12,17 @@ Documentation can be found online at . Th ## Forums -[The FoundationDB Forums](https://forums.foundationdb.org/) are the home for most of the discussion and communication about the FoundationDB project. We welcome your participation! We want FoundationDB to be a great project to be a part of, and as part of that, we have established a [Code of Conduct](CODE_OF_CONDUCT.md) to define what constitutes permissible modes of interaction. +[The FoundationDB Forums](https://forums.foundationdb.org/) are the home for most of the discussion and communication about the FoundationDB project. We welcome your participation! We want FoundationDB to be a great project to be a part of, and as part of that, we have established a [Code of Conduct](CODE_OF_CONDUCT.md) to define what constitutes permissible modes of interaction. ## Contributing -Contributing to FoundationDB can be in contributions to the codebase, sharing your experience and insights in the community on the Forums, or contributing to projects that make use of FoundationDB. Please see the [contributing guide](CONTRIBUTING.md) for more specifics. +Contributions to FoundationDB can include contributions to the codebase, sharing your experience and insights with the community on the Forums, or contributing to projects that make use of FoundationDB. Please see the [contributing guide](CONTRIBUTING.md) for more specifics. ## Getting Started ### Latest Stable Releases -The latest stable releases are (were) versions that are recommended for production use, which have been extensively validated via simulation and real cluster tests and used in our production environment. +The latest stable releases are versions that are recommended for production use, which have been extensively validated via simulation and real cluster tests and used in our production environment. | Branch | Latest Production Release | Notes | |:--------:|:-------------:|------:| @@ -84,7 +84,7 @@ defined in `/root/.bashrc` in the container image. To build outside of the official Docker image, you'll need at least these dependencies: -1. [CMake](https://cmake.org/) version 3.24.2 or higher +1. [CMake](https://cmake.org/) version 3.24.2 or higher 1. [Mono](https://www.mono-project.com/download/stable/) 1. [ninja](https://ninja-build.org/) @@ -140,7 +140,7 @@ Building FoundationDB requires at least 8GB of memory. More memory is needed whe ### macOS -The build under macOS will work the same way as on Linux. [Homebrew](https://brew.sh/) can be used to install the `boost` library and the `ninja` build tool. Be careful, current main branch uses boost 1.86, do install this version or just let cmake download one. Also, if the Swift binding is not of interest, use -DBUILD_SWIFT_BINDING=OFF. +The build under macOS will work the same way as on Linux. [Homebrew](https://brew.sh/) can be used to install the `boost` library and the `ninja` build tool. Be careful: the current main branch uses Boost 1.86; install this version or let CMake download it. Also, if the Swift binding is not of interest, use `-DBUILD_SWIFT_BINDING=OFF`. ```sh cmake -G Ninja -B @@ -156,10 +156,10 @@ To generate an installable package, ### Windows -Under Windows, only Visual Studio with ClangCl is supported +Under Windows, only Visual Studio with ClangCl is supported. 1. Install Visual Studio 2019 (IDE or Build Tools), and enable LLVM support -1. Install [CMake 3.24.2](https://cmake.org/download/) or higher +1. Install [CMake 3.24.2](https://cmake.org/download/) or higher 1. Download [Boost 1.86.0](https://archives.boost.io/release/1.86.0/source/boost_1_86_0.tar.bz2) 1. Unpack boost to C:\boost, or use `-DBOOST_ROOT=` with `cmake` if unpacked elsewhere 1. Install [Python](https://www.python.org/downloads/) if it is not already installed by Visual Studio @@ -169,7 +169,7 @@ Under Windows, only Visual Studio with ClangCl is supported 1. `mkdir build && cd build` 1. `cmake -G "Visual Studio 16 2019" -A x64 -T ClangCl ` 1. `msbuild /p:Configuration=Release foundationdb.sln` -1. To increase build performance, use `/p:UseMultiToolTask=true` and `/p:CL_MPCount=` +1. To increase build performance, use `/p:UseMultiToolTask=true` and `/p:CL_MPCount=` ### Language Bindings @@ -180,11 +180,11 @@ Generally, CMake will build all language bindings for which it can find all nece ### Generating `compile_commands.json` -CMake can build a compilation database for you. However, the default generated one is not too useful as it operates on the generated files. When running `ninja`, the build system creates another `compile_commands.json` file in the source directory. This can then be used for tools such as [CCLS](https://github.com/MaskRay/ccls) and [CQuery](https://github.com/cquery-project/cquery), among others. This way, you can get code completion and code navigation in flow. It is not yet perfect (it will show a few errors), but we are continually working to improve the development experience. +CMake can generate a compilation database for code completion, navigation, and static analysis of the C++20 coroutine sources. Pass `-DCMAKE_EXPORT_COMPILE_COMMANDS=ON` when configuring a Ninja or Makefile build, then point your tooling at `compile_commands.json` in the build directory. -CMake will not produce a `compile_commands.json` by default; you must pass `-DCMAKE_EXPORT_COMPILE_COMMANDS=ON`. This also enables the target `processed_compile_commands`, which rewrites `compile_commands.json` to describe the actor compiler source file, not the post-processed output files, and places the output file in the source directory. This file should then be picked up automatically by any tooling. +When Python support is enabled, this option also enables the `processed_compile_commands` target, which writes a database to the source directory. With Ninja, it includes Swift compilation commands as well. -Note that if the building is done inside the `foundationdb/build` Docker image, the resulting paths will still be incorrect and require manual fixing. One will wish to re-run `cmake` with `-DCMAKE_EXPORT_COMPILE_COMMANDS=OFF` to prevent it from reverting the manual changes. +If the build runs inside a container, the database contains container paths. Run the tooling in the same environment or map those paths to the host checkout. ### Code Formatting and Static Analysis @@ -192,9 +192,9 @@ Note that if the building is done inside the `foundationdb/build` Docker image, ### Using IDEs -CMake provides built-in support for several popular IDEs. However, most FoundationDB files are written in the `flow` language, which is an extension of the C++ programming language, for coroutine support (Note that when FoundationDB was being developed, C++20 was not available). The `flow` language will be transpiled into C++ code using `actorcompiler`, while preventing most IDEs from recognizing `flow`-specific syntax. +CMake provides built-in support for several popular IDEs. FoundationDB's asynchronous code uses standard C++20 coroutines with the Flow runtime, so use an IDE or language server with C++20 support. See the [coroutine guide](design/coroutines.md) for the programming model. -It is possible to generate project files for editing `flow` with a supported IDE. There is a CMake option called `OPEN_FOR_IDE`, which creates a project that can be opened in an IDE for editing. This project cannot be built, but you will be able to edit the files and utilize most of the editing and navigation features that your IDE supports. +The CMake option `OPEN_FOR_IDE` creates an editing-only project for a supported IDE. This project cannot be built, but supports editing and navigation. For example, if you want to use Xcode to make changes to FoundationDB, you can create an Xcode project with the following command: diff --git a/SWIFT_GUIDE.md b/SWIFT_GUIDE.md index 031a0e4ce80..b63a9e485ca 100644 --- a/SWIFT_GUIDE.md +++ b/SWIFT_GUIDE.md @@ -52,8 +52,6 @@ Then, you can then include the generated module in C++: ```cpp // ... #include "SwiftModules/FDBServer" - -#include "flow/actorcompiler.h" // This must be the last #include. ``` ## Swift Basics diff --git a/SWIFT_IDE_SETUP.md b/SWIFT_IDE_SETUP.md index 835ea3a6037..d45e4e4d2bc 100644 --- a/SWIFT_IDE_SETUP.md +++ b/SWIFT_IDE_SETUP.md @@ -45,7 +45,7 @@ export FOUNDATIONDB_LLVM_TOOLCHAIN_ROOT=~/Downloads/clang+llvm-15.0.7-arm64-appl ``` -For actor compiler: Download and install mono: [https://www.mono-project.com](https://www.mono-project.com/), e.g. +For the C# build tools, including option generation, install [Mono](https://www.mono-project.com/), e.g. ```bash brew install mono @@ -147,7 +147,5 @@ Setup: ## Known issues -* jump-to-definition fails to open actor header files. * jump-to-definition from C++ to Swift does not work. * Code completion for semantic responses in Swift can be slow sometimes especially in files that import both FDBServer and FDBClient - diff --git a/bindings/c/test/apitester/TesterTransactionExecutor.cpp b/bindings/c/test/apitester/TesterTransactionExecutor.cpp index 2c0f9d0afa1..54f170621bf 100644 --- a/bindings/c/test/apitester/TesterTransactionExecutor.cpp +++ b/bindings/c/test/apitester/TesterTransactionExecutor.cpp @@ -295,6 +295,7 @@ class TransactionContextBase : public ITransactionContext { std::vector retriedErrorCodes() { std::vector retriedErrorCodes; + retriedErrorCodes.reserve(retriedErrors.size()); for (auto e : retriedErrors) { retriedErrorCodes.push_back(e.code()); } diff --git a/bindings/c/test/apitester/TesterWorkload.cpp b/bindings/c/test/apitester/TesterWorkload.cpp index a40d4faf117..67bdfc41473 100644 --- a/bindings/c/test/apitester/TesterWorkload.cpp +++ b/bindings/c/test/apitester/TesterWorkload.cpp @@ -305,6 +305,7 @@ void WorkloadManager::schedulePrintStatistics(int timeIntervalMs) { std::vector> WorkloadManager::getActiveWorkloads() { std::unique_lock lock(mutex); std::vector> res; + res.reserve(workloads.size()); for (const auto& iter : workloads) { res.push_back(iter.second.ref); } diff --git a/bindings/c/test/client_memory_test.cpp b/bindings/c/test/client_memory_test.cpp index 981667ca25c..e852fcf23e3 100644 --- a/bindings/c/test/client_memory_test.cpp +++ b/bindings/c/test/client_memory_test.cpp @@ -64,6 +64,7 @@ int main(int argc, char** argv) { }; std::vector threads; constexpr auto kThreadCount = 64; + threads.reserve(kThreadCount); for (int i = 0; i < kThreadCount; ++i) { threads.emplace_back(thread_func); } diff --git a/bindings/flow/fdb_flow.cpp b/bindings/flow/fdb_flow.cpp index 9599cbece25..a031ecb9939 100644 --- a/bindings/flow/fdb_flow.cpp +++ b/bindings/flow/fdb_flow.cpp @@ -53,6 +53,7 @@ Future _test() { // for (int i = 0; i < 100000; i++) { // Version v = wait( tr->getReadVersion() ); // } + versions.reserve(100000); for (int i = 0; i < 100000; i++) { versions.push_back(tr->getReadVersion()); } diff --git a/bindings/go/README.md b/bindings/go/README.md index 13e35307c04..5e7e9bc7930 100644 --- a/bindings/go/README.md +++ b/bindings/go/README.md @@ -50,7 +50,7 @@ Generating options file ---------- The [generated.go](./src/fdb/generated.go) is generated based on the [fdb.options](../../fdbclient/vexillographer/fdb.options) file. -If you change the `fdb.options` file you can update the `generated.go` by running the following command from the root of this reporsitory: +If you change the `fdb.options` file you can update the `generated.go` by running the following command from the root of this repository: ```bash go run bindings/go/src/_util/translate_fdb_options.go < fdbclient/vexillographer/fdb.options > bindings/go/src/fdb/generated.go diff --git a/cmake/CompileActorCompiler.cmake b/cmake/CompileActorCompiler.cmake deleted file mode 100644 index 534208d223b..00000000000 --- a/cmake/CompileActorCompiler.cmake +++ /dev/null @@ -1,88 +0,0 @@ -find_package(Python3 REQUIRED COMPONENTS Interpreter) - -if(NOT DEFINED FDB_USE_CSHARP_TOOLS) - set(FDB_USE_CSHARP_TOOLS TRUE) -endif() - -set(ACTORCOMPILER_PY_SRCS - ${CMAKE_CURRENT_SOURCE_DIR}/flow/actorcompiler_py/__main__.py - ${CMAKE_CURRENT_SOURCE_DIR}/flow/actorcompiler_py/errors.py - ${CMAKE_CURRENT_SOURCE_DIR}/flow/actorcompiler_py/actor_parser.py - ${CMAKE_CURRENT_SOURCE_DIR}/flow/actorcompiler_py/actor_compiler.py) - -set(ACTORCOMPILER_CSPROJ - ${CMAKE_CURRENT_SOURCE_DIR}/flow/actorcompiler/actorcompiler.csproj) - -set(ACTORCOMPILER_LEGACY_SRCS - ${CMAKE_CURRENT_SOURCE_DIR}/flow/actorcompiler/ActorCompiler.cs - ${CMAKE_CURRENT_SOURCE_DIR}/flow/actorcompiler/ActorParser.cs - ${CMAKE_CURRENT_SOURCE_DIR}/flow/actorcompiler/ParseTree.cs - ${CMAKE_CURRENT_SOURCE_DIR}/flow/actorcompiler/Program.cs - ${CMAKE_CURRENT_SOURCE_DIR}/flow/actorcompiler/Properties/AssemblyInfo.cs) - -set(ACTOR_COMPILER_REFERENCES - "-r:System,System.Core,System.Xml.Linq,System.Data.DataSetExtensions,Microsoft.CSharp,System.Data,System.Xml" -) - -add_custom_target(actorcompiler_py DEPENDS ${ACTORCOMPILER_PY_SRCS}) - -set(ACTORCOMPILER_PY_COMMAND - ${Python3_EXECUTABLE} -m flow.actorcompiler_py - CACHE INTERNAL "Command to run the Python actor compiler") -set(ACTORCOMPILER_CSHARP_COMMAND "" - CACHE INTERNAL "Command to run the C# actor compiler") - -set(ACTORCOMPILER_COMMAND ${ACTORCOMPILER_PY_COMMAND} - CACHE INTERNAL "Command to run the actor compiler") - -set(actorcompiler_dependencies actorcompiler_py) - -if(FDB_USE_CSHARP_TOOLS AND CSHARP_TOOLCHAIN_FOUND) - if(WIN32) - add_executable(actorcompiler_csharp ${ACTORCOMPILER_LEGACY_SRCS}) - target_compile_options(actorcompiler_csharp PRIVATE "/langversion:6") - set_property( - TARGET actorcompiler_csharp - PROPERTY VS_DOTNET_REFERENCES - "System" - "System.Core" - "System.Xml.Linq" - "System.Data.DataSetExtensions" - "Microsoft.CSharp" - "System.Data" - "System.Xml") - set(ACTORCOMPILER_CSHARP_COMMAND $ - CACHE INTERNAL "Command to run the C# actor compiler") - list(APPEND actorcompiler_dependencies actorcompiler_csharp) - elseif(CSHARP_USE_MONO) - add_custom_command( - OUTPUT actorcompiler.exe - COMMAND ${CSHARP_COMPILER_EXECUTABLE} ARGS ${ACTOR_COMPILER_REFERENCES} - ${ACTORCOMPILER_LEGACY_SRCS} "-target:exe" "-out:actorcompiler.exe" - DEPENDS ${ACTORCOMPILER_LEGACY_SRCS} - COMMENT "Compile actor compiler" - VERBATIM) - add_custom_target(actorcompiler_csharp - DEPENDS ${CMAKE_CURRENT_BINARY_DIR}/actorcompiler.exe) - set(actor_exe "${CMAKE_CURRENT_BINARY_DIR}/actorcompiler.exe") - set(ACTORCOMPILER_CSHARP_COMMAND ${MONO_EXECUTABLE} ${actor_exe} - CACHE INTERNAL "Command to run the C# actor compiler") - list(APPEND actorcompiler_dependencies actorcompiler_csharp) - else() - dotnet_build(${ACTORCOMPILER_CSPROJ} SOURCE ${ACTORCOMPILER_LEGACY_SRCS}) - set(actor_exe "${actorcompiler_EXECUTABLE_PATH}") - message(STATUS "Actor compiler path: ${actor_exe}") - set(ACTORCOMPILER_CSHARP_COMMAND ${dotnet_EXECUTABLE} ${actor_exe} - CACHE INTERNAL "Command to run the C# actor compiler") - endif() -endif() - -if(NOT TARGET actorcompiler) - add_custom_target(actorcompiler) -endif() -add_dependencies(actorcompiler ${actorcompiler_dependencies}) - -if(ACTORCOMPILER_CSHARP_COMMAND) - set(ACTORCOMPILER_COMMAND ${ACTORCOMPILER_CSHARP_COMMAND} - CACHE INTERNAL "Command to run the actor compiler" FORCE) -endif() diff --git a/cmake/ConfigureCompiler.cmake b/cmake/ConfigureCompiler.cmake index a43e222c778..7e0b26b0404 100644 --- a/cmake/ConfigureCompiler.cmake +++ b/cmake/ConfigureCompiler.cmake @@ -132,10 +132,6 @@ if(USE_CLANG_TIDY) message(STATUS "clang-tidy enabled for C/C++ compilation: ${_clang_tidy_command_display}") endif() -if(NOT OPEN_FOR_IDE) - add_compile_definitions(NO_INTELLISENSE) -endif() - if(NOT WIN32) include(CheckIncludeFile) CHECK_INCLUDE_FILE("stdatomic.h" HAS_C11_ATOMICS) diff --git a/cmake/FlowCommands.cmake b/cmake/FlowCommands.cmake index d2921630260..2f1269d969b 100644 --- a/cmake/FlowCommands.cmake +++ b/cmake/FlowCommands.cmake @@ -204,8 +204,7 @@ endfunction() function(add_flow_target) set(options EXECUTABLE STATIC_LIBRARY DYNAMIC_LIBRARY LINK_TEST) set(oneValueArgs NAME) - set(multiValueArgs SRCS COVERAGE_FILTER_OUT DISABLE_ACTOR_DIAGNOSTICS - ADDL_SRCS) + set(multiValueArgs SRCS COVERAGE_FILTER_OUT ADDL_SRCS) cmake_parse_arguments(AFT "${options}" "${oneValueArgs}" "${multiValueArgs}" "${ARGN}") if(NOT AFT_NAME) @@ -214,79 +213,18 @@ function(add_flow_target) if(NOT AFT_SRCS) message(FATAL_ERROR "No sources provided") endif() - # foreach(src IN LISTS AFT_SRCS) is_header(h "${src}") if(NOT h) list(SRCS - # "${CMAKE_CURRENT_SOURCE_DIR}/${src}") endif() endforeach() if(OPEN_FOR_IDE) - # Intentionally omit ${AFT_DISABLE_ACTOR_DIAGNOSTICS} since we don't want - # diagnostics set(sources ${AFT_SRCS} ${AFT_ADDL_SRCS}) add_library(${AFT_NAME} OBJECT ${sources}) else() - create_build_dirs(${AFT_SRCS} ${AFT_DISABLE_ACTOR_DIAGNOSTICS}) - foreach(src IN LISTS AFT_SRCS AFT_DISABLE_ACTOR_DIAGNOSTICS) + foreach(src IN LISTS AFT_SRCS) is_header(hdr ${src}) - set(in_filename "${src}") - if(${src} MATCHES ".*\\.actor\\.(h|cpp)") - set(is_actor_file YES) - if(${src} MATCHES ".*\\.h") - string(REPLACE ".actor.h" ".actor.g.h" out_filename ${in_filename}) - else() - string(REPLACE ".actor.cpp" ".actor.g.cpp" out_filename - ${in_filename}) - endif() - else() - set(is_actor_file NO) - set(out_filename "${src}") - endif() - - set(in_file "${CMAKE_CURRENT_SOURCE_DIR}/${in_filename}") - if(is_actor_file) - if(hdr) - list(APPEND HEADER_LIST ${in_file}) - endif() - set(out_file "${CMAKE_CURRENT_BINARY_DIR}/${out_filename}") - else() - set(out_file "${in_file}") - endif() - + get_filename_component(source_file "${src}" ABSOLUTE + BASE_DIR "${CMAKE_CURRENT_SOURCE_DIR}") if(hdr) - list(APPEND HEADER_LIST ${out_file}) - endif() - - list(APPEND sources ${out_file}) - set(actor_compiler_flags "") - if(is_actor_file) - list(APPEND actors ${in_file}) - list(APPEND actor_compiler_flags "--generate-probes") - foreach(s IN LISTS AFT_DISABLE_ACTOR_DIAGNOSTICS) - if("${s}" STREQUAL "${src}") - list(APPEND actor_compiler_flags "--disable-diagnostics") - break() - endif() - endforeach() - - list(APPEND generated_files ${out_file}) - if(ACTORCOMPILER_CSHARP_COMMAND) - set(py_out_file "${out_file}.py_gen") - set(cs_out_file "${out_file}.cs_gen") - add_custom_command(OUTPUT "${out_file}" - COMMAND ${CMAKE_COMMAND} -E env "PYTHONPATH=${CMAKE_SOURCE_DIR}" - ${ACTORCOMPILER_PY_COMMAND} "${in_file}" "${py_out_file}" ${actor_compiler_flags} - COMMAND ${ACTORCOMPILER_CSHARP_COMMAND} "${in_file}" "${cs_out_file}" ${actor_compiler_flags} - COMMAND ${Python3_EXECUTABLE} - ${CMAKE_SOURCE_DIR}/flow/actorcompiler_py/compare_actor_output.py - "${cs_out_file}" "${py_out_file}" - COMMAND ${CMAKE_COMMAND} -E copy "${py_out_file}" "${out_file}" - DEPENDS "${in_file}" actorcompiler - COMMENT "Compile and compare actor: ${src}") - else() - add_custom_command(OUTPUT "${out_file}" - COMMAND ${CMAKE_COMMAND} -E env "PYTHONPATH=${CMAKE_SOURCE_DIR}" - ${ACTORCOMPILER_COMMAND} "${in_file}" "${out_file}" ${actor_compiler_flags} - DEPENDS "${in_file}" actorcompiler - COMMENT "Compile actor: ${src}") - endif() + list(APPEND HEADER_LIST ${source_file}) endif() + list(APPEND sources ${source_file}) endforeach() if(PASS_COMPILATION_UNIT) foreach(s IN LISTS sources) @@ -319,34 +257,19 @@ function(add_flow_target) add_executable(${AFT_NAME} ${sources} ${AFT_ADDL_SRCS}) endif() - foreach(src IN LISTS sources AFT_ADDL_SRCS) - get_filename_component(dname ${CMAKE_CURRENT_SOURCE_DIR} NAME) - string(REGEX REPLACE "\\..*" "" fname ${src}) - string(REPLACE / _ fname ${fname}) - # set_source_files_properties(${src} PROPERTIES COMPILE_DEFINITIONS - # FNAME=${dname}_${fname}) - endforeach() - set_property(TARGET ${AFT_NAME} PROPERTY SOURCE_FILES ${AFT_SRCS}) set_property(TARGET ${AFT_NAME} PROPERTY HEADER_FILES ${HEADER_LIST}) set_property(TARGET ${AFT_NAME} PROPERTY COVERAGE_FILTERS ${AFT_SRCS}) - if(generated_files) - set_source_files_properties(${generated_files} PROPERTIES SKIP_LINTING ON) - endif() - - add_custom_target(${AFT_NAME}_actors DEPENDS ${generated_files}) if(TARGET fdboptions AND NOT "${AFT_NAME}" STREQUAL "fdboptions") if(DEFINED FDB_OPTIONS_H) set_source_files_properties(${sources} ${AFT_ADDL_SRCS} APPEND PROPERTY OBJECT_DEPENDS ${FDB_OPTIONS_H}) endif() - add_dependencies(${AFT_NAME}_actors fdboptions) add_dependencies(${AFT_NAME} fdboptions) if(TARGET fdboptions_vex) - add_dependencies(${AFT_NAME}_actors fdboptions_vex) + add_dependencies(${AFT_NAME} fdboptions_vex) endif() endif() - add_dependencies(${AFT_NAME} ${AFT_NAME}_actors) generate_coverage_xml(${AFT_NAME}) if(strip_target) strip_debug_symbols(${AFT_NAME}) diff --git a/cmake/utils.cmake b/cmake/utils.cmake index e90811ec16b..2d034f0b1fc 100644 --- a/cmake/utils.cmake +++ b/cmake/utils.cmake @@ -30,20 +30,6 @@ function(is_prefix out prefix str) set(${out} ${res} PARENT_SCOPE) endfunction() -function(create_build_dirs) - foreach(src IN LISTS ARGV) - get_filename_component(d "${src}" DIRECTORY) - if(IS_ABSOLUTE "${d}") - file(RELATIVE_PATH d "${CMAKE_CURRENT_SOURCE_DIR}" "${src}") - endif() - list(APPEND dirs "${d}") - endforeach() - list(REMOVE_DUPLICATES dirs) - foreach(dir IN LISTS dirs) - make_directory("${CMAKE_CURRENT_BINARY_DIR}/${dir}") - endforeach() -endfunction() - function(fdb_find_sources out) file(GLOB res LIST_DIRECTORIES false diff --git a/contrib/gen_compile_db.py b/contrib/gen_compile_db.py index 8234e703e54..1c2d289d423 100644 --- a/contrib/gen_compile_db.py +++ b/contrib/gen_compile_db.py @@ -2,36 +2,14 @@ from argparse import ArgumentParser import os import json -import re import subprocess import shlex -def actorFile(actor: str, build: str, src: str): - res = actor.replace(build, src, 1) - res = res.replace("actor.g.cpp", "actor.cpp") - return res.replace("actor.g.h", "actor.h") - - -def rreplace(s, old, new, occurrence=1): - li = s.rsplit(old, occurrence) - return new.join(li) - - -def actorCommand(cmd: str, build: str, src: str): - r1 = re.compile(r"-c (.+)(actor\.g\.cpp)") - m1 = r1.search(cmd) - if m1 is None: - return cmd - cmd1 = r1.sub("\\1actor.cpp", cmd) - return rreplace(cmd1, build, src) - - parser = ArgumentParser( - description="Generates a new compile_commands.json for rtags+flow" + description="Generates compile_commands.json for C++ and Swift IDE support" ) parser.add_argument("-b", help="Build directory", dest="builddir", default=os.getcwd()) -parser.add_argument("-s", help="Build directory", dest="srcdir", default=os.getcwd()) parser.add_argument("-o", help="Output file", dest="out", default="processed_compile_commands.json") parser.add_argument("-ninjatool", help="Ninja tool", dest="ninjatool", default="") parser.add_argument("input", help="compile_commands.json", default="compile_commands.json", nargs="?") @@ -72,16 +50,7 @@ def actorCommand(cmd: str, build: str, src: str): result = [] for cmd in cmds: - additional_flags = ["-Wno-unknown-attributes"] - cmd["command"] = cmd["command"].replace( - " -DNO_INTELLISENSE ", " {} ".format(" ".join(additional_flags)) - ) - if cmd["file"].endswith("actor.g.cpp"): - # here we need to rewrite the rule - cmd["command"] = actorCommand(cmd["command"], args.builddir, args.srcdir) - cmd["file"] = actorFile(cmd["file"], args.builddir, args.srcdir) - result.append(cmd) - elif cmd['file'].endswith('.swift'): + if cmd['file'].endswith('.swift'): if cmd['file'] in swiftCompilationCommands: result.append(swiftCompilationCommands[cmd['file']]) else: diff --git a/contrib/serialize-check/README.md b/contrib/serialize-check/README.md index c7e31b7ef3a..e45b39ba562 100644 --- a/contrib/serialize-check/README.md +++ b/contrib/serialize-check/README.md @@ -20,9 +20,8 @@ To run `renormalize.py`, a compilation database generated by CMake or other buil Typical usage is to copy both `source_scanner` and `renormalize.py` to the root directory of FoundationDB, and run ```bash -# Generate the compilation database, note that compile_commands.json may not be generated if the build directory is not empty. -# NOTE: OPEN_FOR_IDE is required to prevent the ACTOR compiler transpiles ACTORs. -CC=clang CXX=clang++ LD=lld CMAKE_EXPORT_COMPILE_COMMANDS=1 cmake . -B preconfigure/ -DOPEN_FOR_IDE=ON +# Generate the compilation database for the C++ sources. +CC=clang CXX=clang++ LD=lld cmake . -B preconfigure/ -DCMAKE_EXPORT_COMPILE_COMMANDS=ON ./renormalize.py --compilation-database preconfigure/compile_commands.json ``` diff --git a/design/AI-generated/FDB_NETWORK_PROTOCOL.md b/design/AI-generated/FDB_NETWORK_PROTOCOL.md index 626ef8843bb..3b67aa7defb 100644 --- a/design/AI-generated/FDB_NETWORK_PROTOCOL.md +++ b/design/AI-generated/FDB_NETWORK_PROTOCOL.md @@ -1785,7 +1785,7 @@ difference exceeds `bytesLimit`. On cancellation, the client sends ## Appendix D: Client Library Protocol Flows -**Source:** `fdbclient/NativeAPI.actor.cpp`, `fdbclient/MonitorLeader.cpp`, +**Source:** `fdbclient/NativeAPI.cpp`, `fdbclient/MonitorLeader.cpp`, `fdbclient/GlobalConfig.cpp` This appendix describes the exact sequence of protocol messages the FDB client @@ -2143,7 +2143,7 @@ Backoff: starts at `CLIENT_KNOBS->BACKOFF_DELAY`, multiplied by ## Appendix E: fdbcli Protocol Usage -**Source:** `fdbcli/*.actor.cpp`, `fdbclient/include/fdbclient/ManagementAPI.h` +**Source:** `fdbcli/*.cpp`, `fdbclient/include/fdbclient/ManagementAPI.h` The `fdbcli` command-line tool uses three protocol layers: 1. **Normal transactions** via the NativeAPI (same as any client). @@ -2309,7 +2309,7 @@ Taskbucket workflows that run as normal transactions within the cluster. ## Appendix F: Server-Side Protocol Flows -**Source:** `fdbserver/commitproxy/CommitProxyServer.actor.cpp`, +**Source:** `fdbserver/commitproxy/CommitProxyServer.cpp`, `fdbserver/clustercontroller/ClusterRecovery.cpp`, `fdbserver/storageserver/storageserver.cpp`, `fdbserver/sequencer/masterserver.cpp` diff --git a/design/AI-generated/diagram_01_flow_runtime.md b/design/AI-generated/diagram_01_flow_runtime.md index 3fc47498cc6..3dbb674919d 100644 --- a/design/AI-generated/diagram_01_flow_runtime.md +++ b/design/AI-generated/diagram_01_flow_runtime.md @@ -3,10 +3,10 @@ ```mermaid graph TB subgraph ActorSystem["Actor System"] - Actor["ACTOR function\n(.actor.cpp source)"] - Compiler["Actor Compiler\n(Python preprocessor)"] - StateMachine["Generated State Machine\n(.actor.g.h)"] - Coro["C++20 Coroutine\n(co_await migration)"] + Coro["C++20 Coroutine\n(.cpp / .h source)"] + Compiler["C++ Compiler\n(co_await / co_return)"] + Frame["Coroutine Frame\n(parameters and locals)"] + CoroPromise["CoroPromise / CoroActor\n(Flow integration)"] end subgraph AsyncPrimitives["Async Primitives"] @@ -38,9 +38,8 @@ graph TB Buggify["BUGGIFY\n(fault injection)"] end - Actor --> Compiler --> StateMachine - Actor -.->|migrating to| Coro - StateMachine --> Future + Coro --> Compiler --> Frame + Frame --> CoroPromise --> SAV Promise --> SAV --> Future PS --> FS @@ -64,13 +63,11 @@ sequenceDiagram participant Producer as Producer Actor participant SAV as SAV (SharedState) participant Consumer as Consumer Actor - participant Scheduler as Net2 Scheduler Producer->>SAV: Promise created (SAV allocated) Consumer->>SAV: Future obtained (refcount++) - Consumer->>SAV: wait(future) — register callback + Consumer->>SAV: co_await pendingFuture — register callback Note over Consumer: Suspended Producer->>SAV: promise.send(value) - SAV->>Scheduler: enqueue callback - Scheduler->>Consumer: Resume with value + SAV->>Consumer: Fire callback and resume coroutine with value ``` diff --git a/design/AI-generated/foundationdb_subsystem_map.md b/design/AI-generated/foundationdb_subsystem_map.md index 94729380775..4a49035018c 100644 --- a/design/AI-generated/foundationdb_subsystem_map.md +++ b/design/AI-generated/foundationdb_subsystem_map.md @@ -34,13 +34,13 @@ Plus supporting code: [`fdbserver/worker/`](https://github.com/apple/foundationd **Key abstractions:** - **Future / Promise** — single-assignment async values. A Promise delivers once; a Future receives. Connected via reference-counted `SingleAssignmentVar` (SAV). - **FutureStream / PromiseStream** — multi-value async channels. The primary request/reply pattern: send a struct containing a `ReplyPromise` into a `RequestStream`, and the receiver sends the reply back through it (self-addressed envelope). -- **ACTOR functions** — compiled by a custom C# actor compiler (`actorcompiler.exe`) into state machines. `wait(expr)` suspends; `state` variables persist across suspensions. Being migrated to C++20 coroutines (`co_await`). +- **C++20 coroutines** — asynchronous functions compiled directly from `.cpp` and `.h` files. `co_await` awaits Futures or streams; `co_return` completes the result Future. Parameters and local variables needed across suspension live in the coroutine frame. - **Arena / StringRef / Standalone** — zero-copy memory management. `StringRef` is a non-owning pointer+length; `Arena` owns the backing memory; `Standalone` bundles a T with its Arena. - **Net2** ([`Net2.cpp`](https://github.com/apple/foundationdb/blob/main/flow/Net2.cpp)) — the real event loop. Single-threaded, priority-based task scheduling over Boost.ASIO. Handles timers, yields, I/O multiplexing. - **Deterministic random** ([`IRandom.h`](https://github.com/apple/foundationdb/blob/main/flow/include/flow/IRandom.h)) — `deterministicRandom()` provides a seedable PRNG used everywhere, enabling reproducible simulation. - **Tracing** ([`Trace.cpp`](https://github.com/apple/foundationdb/blob/main/flow/Trace.cpp)) — structured event logging with `.detail()` chains, used for diagnostics and simulation analysis. -**Key dynamic behavior:** Every server-side component is a collection of ACTOR coroutines scheduled on Net2's run loop. There is no threading within a process (except for disk I/O thread pools). All concurrency is cooperative, driven by Future resolution. +**Key dynamic behavior:** Server-side components run as collections of C++ coroutines, still called actors, on Net2's event loop. Their concurrency is cooperative, driven by Future resolution; blocking work such as disk I/O uses separate thread pools. **Principal files:** [`flow.h`](https://github.com/apple/foundationdb/blob/main/flow/include/flow/flow.h), [`Arena.h`](https://github.com/apple/foundationdb/blob/main/flow/include/flow/Arena.h), [`Error.h`](https://github.com/apple/foundationdb/blob/main/flow/include/flow/Error.h), [`genericactors.h`](https://github.com/apple/foundationdb/blob/main/flow/include/flow/genericactors.h), [`Net2.cpp`](https://github.com/apple/foundationdb/blob/main/flow/Net2.cpp), [`Trace.cpp`](https://github.com/apple/foundationdb/blob/main/flow/Trace.cpp), [`serialize.h`](https://github.com/apple/foundationdb/blob/main/flow/include/flow/serialize.h), [`CoroutinesImpl.h`](https://github.com/apple/foundationdb/blob/main/flow/include/flow/CoroutinesImpl.h) @@ -78,10 +78,10 @@ Plus supporting code: [`fdbserver/worker/`](https://github.com/apple/foundationd **Key dynamic behavior:** - **Location cache**: `DatabaseContext` caches key-range → storage-server mappings. On cache miss or stale entry, re-resolves via a commit proxy's `getKeyServerLocations`. - **Proxy load balancing**: `commitProxyLoadBalance()` / `grvProxyLoadBalance()` — round-robin with failover and automatic retry on proxy set changes. -- **Retry loop**: `Transaction::onError()` handles `not_committed`, `transaction_too_old`, etc. by resetting and retrying with backoff. The canonical usage pattern is `loop { try { ... tr.commit(); break; } catch(Error& e) { wait(tr.onError(e)); } }`. +- **Retry loop**: `Transaction::onError()` handles `not_committed`, `transaction_too_old`, etc. by resetting and retrying with backoff. A coroutine awaits `tr.commit()` in a `while (true)` loop, catches and saves any `Error`, then awaits `tr.onError(error)` outside the catch handler before retrying. See the [transaction retry example](../coroutines.md). - **Watches**: `tr.watch(key)` registers interest; the storage server notifies when the key changes. -**Principal files:** [`NativeAPI.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbclient/NativeAPI.actor.cpp), [`ReadYourWrites.cpp`](https://github.com/apple/foundationdb/blob/main/fdbclient/ReadYourWrites.cpp), [`DatabaseContext.cpp`](https://github.com/apple/foundationdb/blob/main/fdbclient/DatabaseContext.cpp), [`MultiVersionTransaction.cpp`](https://github.com/apple/foundationdb/blob/main/fdbclient/MultiVersionTransaction.cpp), [`CommitTransaction.h`](https://github.com/apple/foundationdb/blob/main/fdbclient/include/fdbclient/CommitTransaction.h), [`SystemData.h`](https://github.com/apple/foundationdb/blob/main/fdbclient/include/fdbclient/SystemData.h), [`MonitorLeader.h`](https://github.com/apple/foundationdb/blob/main/fdbclient/include/fdbclient/MonitorLeader.h) +**Principal files:** [`NativeAPI.cpp`](https://github.com/apple/foundationdb/blob/main/fdbclient/NativeAPI.cpp), [`ReadYourWrites.cpp`](https://github.com/apple/foundationdb/blob/main/fdbclient/ReadYourWrites.cpp), [`DatabaseContext.cpp`](https://github.com/apple/foundationdb/blob/main/fdbclient/DatabaseContext.cpp), [`MultiVersionTransaction.cpp`](https://github.com/apple/foundationdb/blob/main/fdbclient/MultiVersionTransaction.cpp), [`CommitTransaction.h`](https://github.com/apple/foundationdb/blob/main/fdbclient/include/fdbclient/CommitTransaction.h), [`SystemData.h`](https://github.com/apple/foundationdb/blob/main/fdbclient/include/fdbclient/SystemData.h), [`MonitorLeader.h`](https://github.com/apple/foundationdb/blob/main/fdbclient/include/fdbclient/MonitorLeader.h) --- @@ -107,7 +107,7 @@ Plus supporting code: [`fdbserver/worker/`](https://github.com/apple/foundationd - `ServerDBInfo` is the cluster-wide configuration broadcast. Contains: master interface, proxy lists, log system config, recovery state, latency band config. - Updated by CC and distributed to all workers. Workers react to changes (e.g., new proxy set). -**Principal files:** [`ClusterController.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/clustercontroller/ClusterController.actor.cpp), `ClusterController.h`, [`Coordination.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/coordinator/Coordination.cpp), [`LeaderElection.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/core/LeaderElection.cpp), [`CoordinatedState.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/core/CoordinatedState.cpp), [`WorkerInterface.h`](https://github.com/apple/foundationdb/blob/main/fdbserver/core/include/fdbserver/core/WorkerInterface.h) +**Principal files:** [`ClusterController.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/clustercontroller/ClusterController.cpp), `ClusterController.h`, [`Coordination.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/coordinator/Coordination.cpp), [`LeaderElection.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/core/LeaderElection.cpp), [`CoordinatedState.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/core/CoordinatedState.cpp), [`WorkerInterface.h`](https://github.com/apple/foundationdb/blob/main/fdbserver/core/include/fdbserver/core/WorkerInterface.h) --- @@ -149,11 +149,11 @@ Phase 4: REPLY CommitProxy **GRV Proxy** ([`GrvProxyServer.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/grvproxy/GrvProxyServer.cpp)): Handles `GetReadVersionRequest` from clients. Assigns read versions that are guaranteed to see all previously committed transactions. Enforces cluster-wide flow control: receives TPS limits from Ratekeeper and gates transaction starts using a smoothed token bucket (`GrvTransactionRateInfo`). Batch-priority transactions are throttled more aggressively and dropped first under pressure. -**Commit Proxy** ([`CommitProxyServer.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/commitproxy/CommitProxyServer.actor.cpp)): The workhorse. Batches commits (`commitBatcher`), drives the 4-phase pipeline, handles metadata mutations (system key writes), and updates the key-server location map. +**Commit Proxy** ([`CommitProxyServer.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/commitproxy/CommitProxyServer.cpp)): The workhorse. Batches commits (`commitBatcher`), drives the 4-phase pipeline, handles metadata mutations (system key writes), and updates the key-server location map. **Resolver** ([`Resolver.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/resolver/Resolver.cpp), [`ConflictSet.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/resolver/ConflictSet.cpp)): Maintains a sliding window of committed write ranges. For each incoming batch, checks if any transaction's read ranges overlap with committed writes at versions newer than the transaction's read version. Pure conflict detection — no side effects. -**Principal files:** [`CommitProxyServer.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/commitproxy/CommitProxyServer.actor.cpp), [`GrvProxyServer.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/grvproxy/GrvProxyServer.cpp), [`masterserver.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/sequencer/masterserver.cpp), [`Resolver.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/resolver/Resolver.cpp), [`ConflictSet.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/resolver/ConflictSet.cpp), [`LogSystem.h`](https://github.com/apple/foundationdb/blob/main/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h) +**Principal files:** [`CommitProxyServer.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/commitproxy/CommitProxyServer.cpp), [`GrvProxyServer.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/grvproxy/GrvProxyServer.cpp), [`masterserver.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/sequencer/masterserver.cpp), [`Resolver.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/resolver/Resolver.cpp), [`ConflictSet.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/resolver/ConflictSet.cpp), [`LogSystem.h`](https://github.com/apple/foundationdb/blob/main/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h) --- @@ -161,7 +161,7 @@ Phase 4: REPLY CommitProxy **What it is:** The durable write-ahead log. All committed mutations flow through TLogs before being applied to storage servers. This is FDB's durability guarantee. -**TLog** ([`TLogServer.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/tlog/TLogServer.actor.cpp)): +**TLog** ([`TLogServer.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/tlog/TLogServer.cpp)): - Receives mutation batches from commit proxies via `push()`. - Writes to a `DiskQueue` (append-only on-disk log) and an in-memory index. - Mutations are **tagged** — each mutation carries one or more `Tag`s indicating which storage server(s) it's relevant to. Tags are assigned by the commit proxy based on the key-to-shard mapping. @@ -180,7 +180,7 @@ Phase 4: REPLY CommitProxy **Key dynamic behavior:** Storage servers are *consumers* of the log. They continuously peek their assigned tag, pulling committed mutations and applying them to local storage. This decoupling means commits don't wait for storage servers — only for TLog quorum. -**Principal files:** [`TLogServer.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/tlog/TLogServer.actor.cpp), [`LogSystem.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/logsystem/LogSystem.cpp), [`LogRouter.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/logrouter/LogRouter.cpp), [`LogSystem.h`](https://github.com/apple/foundationdb/blob/main/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h), [`LogSystemConfig.h`](https://github.com/apple/foundationdb/blob/main/fdbserver/core/include/fdbserver/core/LogSystemConfig.h), [`IDiskQueue.h`](https://github.com/apple/foundationdb/blob/main/fdbserver/core/include/fdbserver/core/IDiskQueue.h) +**Principal files:** [`TLogServer.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/tlog/TLogServer.cpp), [`LogSystem.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/logsystem/LogSystem.cpp), [`LogRouter.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/logrouter/LogRouter.cpp), [`LogSystem.h`](https://github.com/apple/foundationdb/blob/main/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h), [`LogSystemConfig.h`](https://github.com/apple/foundationdb/blob/main/fdbserver/core/include/fdbserver/core/LogSystemConfig.h), [`IDiskQueue.h`](https://github.com/apple/foundationdb/blob/main/fdbserver/core/include/fdbserver/core/IDiskQueue.h) --- @@ -196,13 +196,13 @@ Phase 4: REPLY CommitProxy **Storage Engines** ([`fdbserver/kvstore/`](https://github.com/apple/foundationdb/tree/main/fdbserver/kvstore)): - **IKeyValueStore** interface: `readValue()`, `readRange()`, `set()`, `clear()`, `commit()`. All operations are on a versioned key-value store. -- **RocksDB** ([`KeyValueStoreRocksDB.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/KeyValueStoreRocksDB.actor.cpp)) — primary production engine. Also a sharded variant ([`KeyValueStoreShardedRocksDB.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/KeyValueStoreShardedRocksDB.actor.cpp)) that uses per-shard column families. +- **RocksDB** ([`KeyValueStoreRocksDB.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/KeyValueStoreRocksDB.cpp)) — primary production engine. Also a sharded variant ([`KeyValueStoreShardedRocksDB.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/KeyValueStoreShardedRocksDB.cpp)) that uses per-shard column families. - **SQLite** ([`KeyValueStoreSQLite.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/KeyValueStoreSQLite.cpp)) — legacy engine, still supported. - **Memory** ([`KeyValueStoreMemory.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/KeyValueStoreMemory.cpp)) — for testing and special uses (e.g., txnStateStore during recovery). **Key dynamic behavior:** The storage server is *pull-based*. It pulls from the log system at its own pace. If a storage server falls behind, it catches up by reading more from TLogs. If it falls too far behind, it may be removed from its team and re-replicated. Reads at a given version are served from a snapshot — the SS maintains enough history to serve reads at any version between its oldest and current. -**Principal files:** [`storageserver.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/storageserver/storageserver.cpp), [`KeyValueStoreRocksDB.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/KeyValueStoreRocksDB.actor.cpp), [`KeyValueStoreShardedRocksDB.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/KeyValueStoreShardedRocksDB.actor.cpp), [`KeyValueStoreSQLite.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/KeyValueStoreSQLite.cpp), [`KeyValueStoreMemory.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/KeyValueStoreMemory.cpp), [`IKeyValueStore.h`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/include/fdbserver/kvstore/IKeyValueStore.h), [`StorageMetrics.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/core/StorageMetrics.cpp) +**Principal files:** [`storageserver.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/storageserver/storageserver.cpp), [`KeyValueStoreRocksDB.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/KeyValueStoreRocksDB.cpp), [`KeyValueStoreShardedRocksDB.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/KeyValueStoreShardedRocksDB.cpp), [`KeyValueStoreSQLite.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/KeyValueStoreSQLite.cpp), [`KeyValueStoreMemory.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/KeyValueStoreMemory.cpp), [`IKeyValueStore.h`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/include/fdbserver/kvstore/IKeyValueStore.h), [`StorageMetrics.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/core/StorageMetrics.cpp) --- @@ -219,14 +219,14 @@ Phase 4: REPLY CommitProxy **Components:** - **DDShardTracker** ([`DDShardTracker.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/datadistributor/DDShardTracker.cpp)) — monitors shard sizes and read/write rates. Triggers splits, merges, and moves. -- **DDTeamCollection** ([`DDTeamCollection.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/datadistributor/DDTeamCollection.actor.cpp)) — builds "teams" of storage servers that satisfy the replication policy (e.g., 3 servers in 3 different zones). Maintains healthy/unhealthy team lists. +- **DDTeamCollection** ([`DDTeamCollection.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/datadistributor/DDTeamCollection.cpp)) — builds "teams" of storage servers that satisfy the replication policy (e.g., 3 servers in 3 different zones). Maintains healthy/unhealthy team lists. - **DDRelocationQueue** ([`DDRelocationQueue.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/datadistributor/DDRelocationQueue.cpp)) — prioritized queue of shard movements. Executes moves concurrently up to a parallelism limit. - **MoveKeys** ([`fdbserver/core/MoveKeys.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/core/MoveKeys.cpp)) — the protocol for atomically transferring shard ownership. Uses a "move keys lock" to serialize concurrent moves. Updates the `keyServers` and `serverKeys` system key spaces. - **DataMovement** ([`fdbserver/core/DataMovement.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/core/DataMovement.cpp)) — data move metadata and physical shard ID tracking. **Key dynamic behavior:** Data distribution is a continuous background process. The DD singleton (recruited by CC) watches for changes in shard metrics, server health, and configuration. It produces `RelocateShard` requests that flow through the relocation queue, which executes the MoveKeys protocol. During a move, the destination SS fetches data from the source SS and begins consuming from the log system. When caught up, ownership transfers atomically via a system key transaction. -**Principal files:** [`DataDistribution.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/datadistributor/DataDistribution.cpp), [`DDTeamCollection.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/datadistributor/DDTeamCollection.actor.cpp), [`DDRelocationQueue.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/datadistributor/DDRelocationQueue.cpp), [`DDShardTracker.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/datadistributor/DDShardTracker.cpp), [`MoveKeys.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/core/MoveKeys.cpp), [`DataMovement.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/core/DataMovement.cpp), [`ShardsAffectedByTeamFailure.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/datadistributor/ShardsAffectedByTeamFailure.cpp) +**Principal files:** [`DataDistribution.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/datadistributor/DataDistribution.cpp), [`DDTeamCollection.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/datadistributor/DDTeamCollection.cpp), [`DDRelocationQueue.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/datadistributor/DDRelocationQueue.cpp), [`DDShardTracker.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/datadistributor/DDShardTracker.cpp), [`MoveKeys.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/core/MoveKeys.cpp), [`DataMovement.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/core/DataMovement.cpp), [`ShardsAffectedByTeamFailure.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/datadistributor/ShardsAffectedByTeamFailure.cpp) --- diff --git a/design/AI-generated/subsystem_01_flow_runtime.md b/design/AI-generated/subsystem_01_flow_runtime.md index fa7a41e332e..56ebec53782 100644 --- a/design/AI-generated/subsystem_01_flow_runtime.md +++ b/design/AI-generated/subsystem_01_flow_runtime.md @@ -105,54 +105,50 @@ No explicit backpressure -- values accumulate in the queue. Implicit pressure: i --- -## The Actor System +## Actors and C++20 Coroutines -### ACTOR Compiler +### Coroutine Functions -FDB uses a custom **C# actor compiler** (`actorcompiler.exe`) that transforms `.actor.cpp` files into standard C++. The `ACTOR` keyword marks a function as a coroutine: +FDB's asynchronous functions use standard C++20 coroutines in `.cpp` and `.h` files. A coroutine returns a `Future` and uses `co_await` to await asynchronous results: ```cpp -ACTOR Future myActor(Future input) { - state int x = 42; // persists across waits - int val = wait(input); // suspension point - return val + x; +Future myActor(Future input) { + int x = 42; + int val = co_await input; + co_return val + x; } ``` -The compiler: -1. Scans for `wait()` and `waitNext()` calls -2. Breaks the function into callback states at each suspension point -3. Generates a C++ class inheriting from `Actor` with a SAV -4. Each `wait()` becomes a callback registration + return -5. `state` variables are promoted to class members (persist across suspensions) - -**Key macros** (`actorcompiler.h`): -- `ACTOR` -- marks coroutine function -- `state` -- marks variables that survive across `wait()` -- `wait(Future)` -- suspend until Future ready, return T -- `waitNext(FutureStream)` -- suspend until next stream value -- `choose { when(wait(a)) {...} when(wait(b)) {...} }` -- race two futures -- `loop` -- `while(true)` alias +The C++ compiler preserves parameters and local variables needed across suspension in the coroutine frame. Flow connects that frame to its existing Future/Promise and callback machinery; these asynchronous tasks are still called actors. + +**Common operations:** +- `co_await future` -- obtain a Future's value, suspending only if it is not ready +- `co_await stream` -- consume the next value from a `FutureStream` +- `co_return value` -- complete a `Future` coroutine; use `co_return;` for `Future` +- `co_await race(a, b)` -- await the first of several futures through `flow/CoroUtils.h` +- `while (true)` -- repeat an asynchronous service loop + +See [Writing Coroutines in FoundationDB](../coroutines.md) for usage and lifetime rules. ### C++20 Coroutine Integration -- [`CoroutinesImpl.h`](https://github.com/apple/foundationdb/blob/main/flow/include/flow/CoroutinesImpl.h) -The codebase is being migrated from the custom actor compiler to C++20 native coroutines: +Flow supplies the promise and awaiter types used by the C++ coroutine machinery: -**CoroPromise** ([`CoroutinesImpl.h`](https://github.com/apple/foundationdb/blob/main/flow/include/flow/CoroutinesImpl.h)`:737-844`): +**CoroPromise** ([`CoroutinesImpl.h`](https://github.com/apple/foundationdb/blob/main/flow/include/flow/CoroutinesImpl.h)): - The `promise_type` for `Future` coroutines - Embeds a `CoroActor` which inherits from `Actor` (which inherits from SAV) - `get_return_object()` returns `Future` backed by the coroutine's SAV - `initial_suspend()` returns `suspend_never` (eager start) - `await_transform()` overloads convert `Future` into `AwaitableFuture` -**AwaitableFuture** ([`CoroutinesImpl.h`](https://github.com/apple/foundationdb/blob/main/flow/include/flow/CoroutinesImpl.h)`:453-534`): +**AwaitableFuture** ([`CoroutinesImpl.h`](https://github.com/apple/foundationdb/blob/main/flow/include/flow/CoroutinesImpl.h)): - The awaiter type for `co_await Future` - `await_ready()` -- checks if future is ready or coroutine cancelled - `await_suspend()` -- registers callback with the future's SAV - `await_resume()` -- extracts value or throws error - Inherits from `Callback` so it can be inserted into SAV's callback chain -**Cancellation:** `CoroActor::cancel()` sets `ACTOR_WAIT_STATE_CANCELLED` and resumes the coroutine, which checks the flag in `await_ready()` and throws `actor_cancelled()`. +**Cancellation:** For an ordinary cancellable coroutine, `CoroActor::cancel()` sets `ACTOR_WAIT_STATE_CANCELLED` and resumes the coroutine if it is suspended. The awaiter's resume path throws `actor_cancelled()` into the coroutine. If cancellation is already set when a new await begins, the readiness check routes directly to that error path. --- @@ -342,12 +338,12 @@ Key utility actors: ### Actor Execution Cycle ``` -1. Actor created → SAV allocated, initial code runs until first wait() -2. wait(future) → registers callback on future's SAV, returns to event loop +1. Coroutine called → frame and SAV created, code runs eagerly until it suspends or completes +2. co_await pendingFuture → registers callback on the future's SAV and suspends 3. Event fires (network data, timer, another actor sends) → SAV::send(value) 4. Callback chain walked → actor's callback called -5. Actor resumes from wait point, runs until next wait() or return -6. return value → SAV::send(value) on actor's own SAV → caller's callback fires +5. Coroutine resumes from suspension, runs until another pending await or completion +6. co_return value → completes the coroutine's own SAV → caller's callback fires ``` ### Memory Flow @@ -375,4 +371,4 @@ Key utility actors: | `flow/include/flow/FastAlloc.h` | Thread-local pool allocator | | [`flow/Net2.cpp`](https://github.com/apple/foundationdb/blob/main/flow/Net2.cpp) | Real event loop (Boost.ASIO based) | | [`flow/Trace.cpp`](https://github.com/apple/foundationdb/blob/main/flow/Trace.cpp) | Structured event tracing | -| `flow/include/flow/actorcompiler.h` | ACTOR/wait/state macro definitions | +| [`flow/include/flow/CoroUtils.h`](https://github.com/apple/foundationdb/blob/main/flow/include/flow/CoroUtils.h) | Coroutine selection helpers (Choose, race) | diff --git a/design/AI-generated/subsystem_02_rpc_transport.md b/design/AI-generated/subsystem_02_rpc_transport.md index dd1239f90fb..3df4c0bdd41 100644 --- a/design/AI-generated/subsystem_02_rpc_transport.md +++ b/design/AI-generated/subsystem_02_rpc_transport.md @@ -110,7 +110,7 @@ GetValueRequest req; req.key = myKey; req.reply = ReplyPromise(); // self-addressed envelope stream.send(req); -GetValueReply reply = wait(req.reply.getFuture()); +GetValueReply reply = co_await req.reply.getFuture(); ``` ### ReplyPromise ([`fdbrpc.h`](https://github.com/apple/foundationdb/blob/main/fdbrpc/include/fdbrpc/fdbrpc.h)`:131-204`) @@ -142,7 +142,7 @@ SERVER: CLIENT: 12. connectionReader() receives reply packet 13. Token lookup → NetSAV::receive() → deserialize → SAV::send(value) -14. Client's wait() resolves +14. Client's `co_await` resumes with the reply ``` ### ReplyPromiseStream ([`fdbrpc.h`](https://github.com/apple/foundationdb/blob/main/fdbrpc/include/fdbrpc/fdbrpc.h)`:460-640`) @@ -361,7 +361,7 @@ Simulates network impairment: --- -## Load Balancing -- `LoadBalance.actor.h` +## Load Balancing -- `LoadBalance.h` Client-side request distribution: @@ -398,7 +398,7 @@ Client-side request distribution: | [`fdbrpc/include/fdbrpc/FailureMonitor.h`](https://github.com/apple/foundationdb/blob/main/fdbrpc/include/fdbrpc/FailureMonitor.h) | Failure detection interface and implementation | | [`fdbrpc/include/fdbrpc/Locality.h`](https://github.com/apple/foundationdb/blob/main/fdbrpc/include/fdbrpc/Locality.h) | LocalityData, ProcessClass, LBDistance | | [`fdbrpc/include/fdbrpc/ReplicationPolicy.h`](https://github.com/apple/foundationdb/blob/main/fdbrpc/include/fdbrpc/ReplicationPolicy.h) | PolicyOne, PolicyAcross, PolicyAnd | -| `fdbrpc/include/fdbrpc/LoadBalance.actor.h` | Client-side load balancing | +| `fdbrpc/include/fdbrpc/LoadBalance.h` | Client-side load balancing | | [`fdbrpc/sim2.cpp`](https://github.com/apple/foundationdb/blob/main/fdbrpc/sim2.cpp) | Sim2 deterministic simulator | | [`fdbrpc/include/fdbrpc/simulator.h`](https://github.com/apple/foundationdb/blob/main/fdbrpc/include/fdbrpc/simulator.h) | ISimulator interface, ProcessInfo | | `flow/include/flow/IConnection.h` | IConnection, IListener interfaces | diff --git a/design/AI-generated/subsystem_03_client_library.md b/design/AI-generated/subsystem_03_client_library.md index 0723149a6ca..01cb4eb22a0 100644 --- a/design/AI-generated/subsystem_03_client_library.md +++ b/design/AI-generated/subsystem_03_client_library.md @@ -119,7 +119,7 @@ Caches recent read versions to reduce GRV proxy load. Background updater via `ba --- -## Commit Flow -- [`NativeAPI.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbclient/NativeAPI.actor.cpp)`:4369+` +## Commit Flow -- [`NativeAPI.cpp`](https://github.com/apple/foundationdb/blob/main/fdbclient/NativeAPI.cpp)`:4369+` ### tryCommit() Actor @@ -236,8 +236,8 @@ while (true) { ```cpp Future watch = tr.watch(key); // register interest -wait(tr.commit()); // commit first -wait(watch); // then wait for change +co_await tr.commit(); // commit first +co_await watch; // then wait for change ``` - Watch metadata tracked in `DatabaseContext::watchMap` @@ -277,7 +277,7 @@ Management API exposed as key-value operations: | File | Purpose | |------|---------| -| [`fdbclient/NativeAPI.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbclient/NativeAPI.actor.cpp) | Core transaction implementation, tryCommit, getValue, getRange | +| [`fdbclient/NativeAPI.cpp`](https://github.com/apple/foundationdb/blob/main/fdbclient/NativeAPI.cpp) | Core transaction implementation, tryCommit, getValue, getRange | | `fdbclient/include/fdbclient/NativeAPI.h` | Transaction class, TransactionState | | [`fdbclient/ReadYourWrites.cpp`](https://github.com/apple/foundationdb/blob/main/fdbclient/ReadYourWrites.cpp) | RYW layer, write map, snapshot cache merging | | `fdbclient/include/fdbclient/DatabaseContext.h` | DatabaseContext: location cache, proxy tracking, watches | diff --git a/design/AI-generated/subsystem_04_cluster_controller.md b/design/AI-generated/subsystem_04_cluster_controller.md index f466e142f31..5a905289c32 100644 --- a/design/AI-generated/subsystem_04_cluster_controller.md +++ b/design/AI-generated/subsystem_04_cluster_controller.md @@ -174,7 +174,7 @@ struct WorkerHealth { --- -## Worker Registration -- [`ClusterController.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/clustercontroller/ClusterController.actor.cpp)`:1252-1434` +## Worker Registration -- [`ClusterController.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/clustercontroller/ClusterController.cpp)`:1252-1434` ### RegisterWorkerRequest @@ -310,7 +310,7 @@ RPCs exposed by every worker process: | File | Purpose | |------|---------| -| [`fdbserver/clustercontroller/ClusterController.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/clustercontroller/ClusterController.actor.cpp) | CC main loop, worker registration, event handling | +| [`fdbserver/clustercontroller/ClusterController.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/clustercontroller/ClusterController.cpp) | CC main loop, worker registration, event handling | | `fdbserver/clustercontroller/ClusterController.h` | ClusterControllerData, fitness calculation, recruitment | | [`fdbserver/coordinator/Coordination.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/coordinator/Coordination.cpp) | leaderRegister, generation register, coordination | | [`fdbserver/core/LeaderElection.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/core/LeaderElection.cpp) | tryBecomeLeaderInternal, candidacy, heartbeat | diff --git a/design/AI-generated/subsystem_05_commit_pipeline.md b/design/AI-generated/subsystem_05_commit_pipeline.md index 93d62016b7c..880992dfee1 100644 --- a/design/AI-generated/subsystem_05_commit_pipeline.md +++ b/design/AI-generated/subsystem_05_commit_pipeline.md @@ -27,7 +27,7 @@ Phase 4: REPLY ▼ --- -## Commit Proxy -- [`CommitProxyServer.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/commitproxy/CommitProxyServer.actor.cpp) +## Commit Proxy -- [`CommitProxyServer.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/commitproxy/CommitProxyServer.cpp) The workhorse of the commit pipeline. Batches client commits and drives them through all phases. @@ -51,7 +51,7 @@ struct ProxyCommitData { ### commitBatch() -- 5-Phase Pipeline -**CommitBatchContext** (`CommitProxyServer.actor.cpp:491`): +**CommitBatchContext** (`CommitProxyServer.cpp:491`): ``` struct CommitBatchContext { std::vector trs; // batch of transactions @@ -295,7 +295,7 @@ Tags connect mutations to storage servers: | File | Purpose | |------|---------| -| [`fdbserver/commitproxy/CommitProxyServer.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/commitproxy/CommitProxyServer.actor.cpp) | 5-phase commit batch pipeline | +| [`fdbserver/commitproxy/CommitProxyServer.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/commitproxy/CommitProxyServer.cpp) | 5-phase commit batch pipeline | | `fdbserver/commitproxy/ProxyCommitData.h` | ProxyCommitData, tag lookup | | [`fdbserver/grvproxy/GrvProxyServer.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/grvproxy/GrvProxyServer.cpp) | Read version assignment, rate limiting enforcement | | `fdbserver/grvproxy/GrvTransactionRateInfo.h` | Token-bucket rate limiter driven by ratekeeper | diff --git a/design/AI-generated/subsystem_06_tlog_logsystem.md b/design/AI-generated/subsystem_06_tlog_logsystem.md index 429c613b510..8613b906797 100644 --- a/design/AI-generated/subsystem_06_tlog_logsystem.md +++ b/design/AI-generated/subsystem_06_tlog_logsystem.md @@ -14,7 +14,7 @@ The TLog subsystem is FDB's durability guarantee. All committed mutations flow t --- -## TLog Server -- [`TLogServer.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/tlog/TLogServer.actor.cpp) +## TLog Server -- [`TLogServer.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/tlog/TLogServer.cpp) ### TLogData (lines 295-420) @@ -235,7 +235,7 @@ Optional per-storage-server version tracking: | File | Purpose | |------|---------| -| [`fdbserver/tlog/TLogServer.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/tlog/TLogServer.actor.cpp) | TLog server: push, peek, spill, DiskQueue | +| [`fdbserver/tlog/TLogServer.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/tlog/TLogServer.cpp) | TLog server: push, peek, spill, DiskQueue | | [`fdbserver/logsystem/LogSystem.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/logsystem/LogSystem.cpp) | LogSystem: push/peek/pop across TLog replicas | | [`fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h`](https://github.com/apple/foundationdb/blob/main/fdbserver/logsystem/include/fdbserver/logsystem/LogSystem.h) | LogSystem, IPeekCursor interfaces | | [`fdbserver/core/include/fdbserver/core/LogSystemConfig.h`](https://github.com/apple/foundationdb/blob/main/fdbserver/core/include/fdbserver/core/LogSystemConfig.h) | Log system configuration and epoch tracking | diff --git a/design/AI-generated/subsystem_07_storage_server.md b/design/AI-generated/subsystem_07_storage_server.md index c6c1f63c0cc..c751f526592 100644 --- a/design/AI-generated/subsystem_07_storage_server.md +++ b/design/AI-generated/subsystem_07_storage_server.md @@ -277,7 +277,7 @@ IKeyValueStore* openKVStore(KeyValueStoreType storeType, std::string filename, U --- -### RocksDB Engine -- [`KeyValueStoreRocksDB.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/KeyValueStoreRocksDB.actor.cpp) (~3K lines) +### RocksDB Engine -- [`KeyValueStoreRocksDB.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/KeyValueStoreRocksDB.cpp) (~3K lines) The primary production storage engine. Wraps Facebook's RocksDB LSM-tree engine, adapting it to the `IKeyValueStore` interface via thread pools for non-blocking I/O. @@ -395,7 +395,7 @@ Comprehensive metrics are emitted via `rocksDBMetricLogger` (line 1008): --- -### Redwood (VersionedBTree) Engine -- [`VersionedBTree.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/VersionedBTree.actor.cpp) (~11K lines) +### Redwood (VersionedBTree) Engine -- [`VersionedBTree.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/VersionedBTree.cpp) (~11K lines) A custom versioned copy-on-write B-tree storage engine built from scratch for FoundationDB. Named "Redwood," it is designed for high space efficiency through delta compression and for tight integration with FDB's versioned storage model. @@ -757,8 +757,8 @@ Client ──GetValueRequest──▶ StorageServer getValueQ() | [`fdbserver/storageserver/storageserver.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/storageserver/storageserver.cpp) | SS main loop, update, read serving, shard management | | [`fdbserver/kvstore/include/fdbserver/kvstore/IKeyValueStore.h`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/include/fdbserver/kvstore/IKeyValueStore.h) | IKeyValueStore interface definition | | [`fdbserver/kvstore/IKeyValueStore.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/IKeyValueStore.cpp) | Factory function `openKVStore()` | -| [`fdbserver/kvstore/KeyValueStoreRocksDB.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/KeyValueStoreRocksDB.actor.cpp) | RocksDB engine implementation | -| [`fdbserver/kvstore/VersionedBTree.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/VersionedBTree.actor.cpp) | Redwood engine (VersionedBTree + DWALPager) | +| [`fdbserver/kvstore/KeyValueStoreRocksDB.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/KeyValueStoreRocksDB.cpp) | RocksDB engine implementation | +| [`fdbserver/kvstore/VersionedBTree.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/VersionedBTree.cpp) | Redwood engine (VersionedBTree + DWALPager) | | [`fdbserver/kvstore/IPager.h`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/IPager.h) | Pager interface (IPager2) for Redwood | | [`fdbserver/kvstore/DeltaTree.h`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/DeltaTree.h) | DeltaTree2 delta-compressed sorted structure | | [`fdbserver/kvstore/KeyValueStoreSQLite.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/kvstore/KeyValueStoreSQLite.cpp) | SQLite engine (V1/V2) | diff --git a/design/AI-generated/subsystem_08_data_distribution.md b/design/AI-generated/subsystem_08_data_distribution.md index 2d1a149db7c..ec95f3f8db0 100644 --- a/design/AI-generated/subsystem_08_data_distribution.md +++ b/design/AI-generated/subsystem_08_data_distribution.md @@ -35,7 +35,7 @@ Plus optional bulk load/dump cores. --- -## DDTeamCollection -- [`DDTeamCollection.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/datadistributor/DDTeamCollection.actor.cpp) +## DDTeamCollection -- [`DDTeamCollection.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/datadistributor/DDTeamCollection.cpp) ### What is a Team? @@ -262,7 +262,7 @@ MoveKeys protocol: | File | Purpose | |------|---------| | [`fdbserver/datadistributor/DataDistribution.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/datadistributor/DataDistribution.cpp) | Main DD actor, initialization | -| [`fdbserver/datadistributor/DDTeamCollection.actor.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/datadistributor/DDTeamCollection.actor.cpp) | Team building, health monitoring, SS recruitment | +| [`fdbserver/datadistributor/DDTeamCollection.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/datadistributor/DDTeamCollection.cpp) | Team building, health monitoring, SS recruitment | | [`fdbserver/datadistributor/DDShardTracker.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/datadistributor/DDShardTracker.cpp) | Shard metrics, split/merge decisions | | [`fdbserver/datadistributor/DDRelocationQueue.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/datadistributor/DDRelocationQueue.cpp) | Relocation queue, prioritization, execution | | [`fdbserver/core/MoveKeys.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/core/MoveKeys.cpp) | Atomic shard transfer protocol | diff --git a/design/AI-generated/subsystem_09_cluster_recovery.md b/design/AI-generated/subsystem_09_cluster_recovery.md index 2316caa1703..ce95bd2857f 100644 --- a/design/AI-generated/subsystem_09_cluster_recovery.md +++ b/design/AI-generated/subsystem_09_cluster_recovery.md @@ -147,7 +147,7 @@ All old generation TLogs have been fully recruited into the system. Monitors TLog set completeness and drives state transitions: ``` -loop { +while (true) { // Convert current logSystem to DBCoreState self->logSystem->toCoreState(newState); diff --git a/design/AcAC.md b/design/AcAC.md index 00183100d46..0a78f8ca63e 100644 --- a/design/AcAC.md +++ b/design/AcAC.md @@ -1,5 +1,7 @@ # AcAC -- Actor Active Context +This document records the actor-context tooling and traces from the implementation before the C++ coroutine migration. The captured output and `.actor.cpp` paths below are historical; they do not describe current coroutine source files. + ## Introduction FoundationDB developers often face multiple difficulties debugging the code, and one of the significant challenges is that the call stack is usually hard to interpret. Because of the underlying "chain reaction" mechanism, when a single-assigned value (SAV) is fired, actors waiting for the `Promise`s will be called recursively, creating a huge stack and overwhelming the reader. Providing an actor-wise call stack for the current running actor would be beneficial. For this purpose, AcAC is developed. diff --git a/design/Commit/How a commit is done in FDB.md b/design/Commit/How a commit is done in FDB.md index 7ccccc5e626..2dd7d218847 100644 --- a/design/Commit/How a commit is done in FDB.md +++ b/design/Commit/How a commit is done in FDB.md @@ -84,7 +84,7 @@ The format of `location` is, in general, `.>> decodeRangeFileBlock(Reference file, int64_t offset, int len, Database cx)`. +The current range-block decoder is `fileBackup::decodeRangeFileBlock()` in [`BackupFileFormat.cpp`](../fdbclient/BackupFileFormat.cpp). ### Data format in a log file @@ -82,7 +82,7 @@ where type is the mutation type, such as Set or Clear, `kLen` and `vLen` respect The code related to how a log file is written is in the `struct LogFileWriter` in `namespace fileBackup`. -The code that decodes a mutation block is in `ACTOR Future>> decodeLogFileBlock(Reference file, int64_t offset, int len)`. +The current mutation-block decoder is `fileBackup::decodeMutationLogFileBlock()` in [`BackupFileFormat.cpp`](../fdbclient/BackupFileFormat.cpp). ### Endianness diff --git a/design/cdc.md b/design/cdc.md index 178110c5009..254f889db8f 100644 --- a/design/cdc.md +++ b/design/cdc.md @@ -176,9 +176,9 @@ A typical consumer loop is: ```cpp co_await registerNativeCdcStreamClient(db, "orders"_sr, KeyRangeRef("order/"_sr, "order0"_sr)); -state Reference consumer = co_await createNativeCdcConsumer(db, "orders"_sr); +Reference consumer = co_await createNativeCdcConsumer(db, "orders"_sr); -loop { +while (true) { CDCConsumeReply reply = co_await consumer->consume(); for (auto const& versionedMutations : reply.mutations) { // Apply all mutations for versionedMutations.version. @@ -359,6 +359,7 @@ than transaction state: | --- | --- | --- | | `\xff\x02/cdc/minVersion/` | `Version` | Earliest version that an active stream may still require. | | `\xff\x02/cdc/retiredTagPopVersion/` | `Version` | Final pop watermark required after a stream using a tag is removed. | +| `\xff\x02/cdc/tagOwner/` | `CDCStreamId` | Derived representative stream used to look up a current tag's proxy owner. | The initial `minVersion` is written with a versionstamp at stream registration. When a consumer acknowledges processing through version `V`, the @@ -395,6 +396,22 @@ properties rather than exceptional cases. The current proxy assignment at registration uses an available CDC proxy; it does not yet balance by stream traffic, memory use, or consumer lag. +Registration reuses a tag's owner through a persisted representative stream. +The lookup validates, in the registration transaction, that the representative +is active and still uses that current tag, then reads its authoritative +per-stream proxy assignment. Proxy replacement therefore does not require a +second ownership update. A missing or stale entry is reconstructed from active +stream metadata; registration on an unused tag establishes its first +representative. Removing the representative clears the entry without disturbing +other streams or their retained history. + +This index avoids repeated global ownership discovery for stable shared tags. +It is derived storage-backed system data, not routing or retention authority, +and can be discarded and rebuilt. Existing streams need no eager migration; +validation also rejects stale representatives left by older metadata writers. +The allocator's stream-count scan remains necessary, and ownership discovery +still scans global metadata when the representative is absent or invalid. + ### Metadata lifecycle example Assume a client registers stream name `orders` for range diff --git a/design/coroutines.md b/design/coroutines.md index 9a0d6074ebd..c61dbbc21ec 100644 --- a/design/coroutines.md +++ b/design/coroutines.md @@ -1,34 +1,31 @@ # Coroutines in Flow -* [Introduction](#Introduction) -* [Coroutines vs ACTORs](#coroutines-vs-actors) +* [Introduction](#introduction) * [Basic Types](#basic-types) -* [Choose-When](#choose-when) - * [Execution in when-expressions](#execution-in-when-expressions) - * [Waiting in When-Blocks](#waiting-in-when-blocks) - * [Migrating `choose`-`when`](#migrating-choose-when) +* [Waiting for Multiple Futures](#waiting-for-multiple-futures) + * [Ordered Evaluation with Choose](#ordered-evaluation-with-choose) + * [Concurrent Request Handling](#concurrent-request-handling) * [Generators](#generators) - * [Generators and ranges](#generators-and-ranges) - * [Eager vs Lazy Execution](#eager-vs-lazy-execution) + * [Generators and Ranges](#generators-and-ranges) + * [Execution and Value Ownership](#execution-and-value-ownership) * [Generators vs Promise Streams](#generators-vs-promise-streams) -* [Uncancellable](#uncancellable) -* [Porting ACTOR's to C++ Coroutines](#porting-actors-to-c-coroutines) - * [Lifetime of Locals](#lifetime-of-locals) - * [Unnecessary Helper Actors](#unnecessary-helper-actors) - * [Replace Locals with Temporaries](#replace-locals-with-temporaries) - * [Don't Wait in Error-Handlers](#dont-wait-in-error-handlers) - * [Make Static Functions Class Members](#make-static-functions-class-members) - * [Initialization of Locals](#initialization-of-locals) -* [Internals Reference](fdb-coroutines-internals.md) — deep dive into the coroutine runtime (SAV, Actor, awaiters, cancellation) +* [Cancellation](#cancellation) + * [Uncancellable](#uncancellable) + * [NoThrowOnCancel](#nothrowoncancel) +* [Lifetime and Ownership](#lifetime-and-ownership) + * [Locals and Scope](#locals-and-scope) + * [Parameters and Objects](#parameters-and-objects) +* [Error Handlers](#error-handlers) +* [Direct Await Expressions](#direct-await-expressions) +* [Internals Reference](fdb-coroutines-internals.md) — coroutine runtime, awaiters, and cancellation ## Introduction -In the past Flow implemented an actor mode by shipping its own compiler which would -extend the C++ language with a few additional keywords. This, while still supported, -is deprecated in favor of the standard C++20 coroutines. +Flow uses standard C++20 coroutines for asynchronous code. Coroutines work with Flow's network loop, futures, +RPC layer, and deterministic simulator. They use ordinary C++ control flow around `co_await`, `co_return`, and, +for generators, `co_yield`. -Coroutines are meant to be simple, look like serial code, and be easy to reason about. -As simple example for a coroutine function can look like this: +A simple coroutine looks like this: ```c++ Future simpleCoroutine() { @@ -38,996 +35,333 @@ Future simpleCoroutine() { } ``` -This document assumes some familiarity with Flow. As of today, actors and coroutines -can be freely mixed, but new code should be written using coroutines. +The function starts executing when called. It returns a `Future` that becomes ready when the coroutine +returns its result or fails. Awaiting a ready future does not suspend; awaiting a pending future registers a +continuation with Flow and suspends until the value or error is available. -## Coroutines vs ACTORs - -### Performance Characteristics - -**For detailed performance analysis, benchmarking results, and optimization techniques, see [`COROUTINE_PERF_ANALYSIS.md`](../COROUTINE_PERF_ANALYSIS.md).** - -**Key Summary**: C++20 coroutines show pattern-dependent performance: -- **Excellent** for suspension-heavy patterns (YIELD: +83% faster than actors) -- **Competitive** for allocation-heavy patterns (NET2: 3-8% slower than actors) -- **Production ready** with performance characteristics suitable for most FDB workloads +This guide assumes familiarity with Flow's basic types. Their coroutine support is defined in +[`Coroutines.h`](../flow/include/flow/Coroutines.h) and +[`CoroutinesImpl.h`](../flow/include/flow/CoroutinesImpl.h). ## Basic Types -It is important to understand that C++ coroutine support doesn't change anything in Flow: they are not a replacement -of Flow but they replace the actor compiler with a C++ compiler. This means, that the network loop, all Flow types, -the RPC layer, and the simulator all remain unchanged. A coroutine simply returns a special `SAV` which has handle -to a coroutine. - -As defined in the C++20 standard, a function is a coroutine if its body contains at least one `co_await`, `co_yield`, -or `co_return` statement. However, in order for this to work, the return type needs an underlying coroutine -implementation. Flow provides these for the following types: - -* `Future` is the primary type we use for coroutines. A coroutine returning - `Future` is allowed to `co_await` other coroutines and it can `co_return` - a single value. `co_yield` is not implemented by this type. - * A special case is `Future`. Void-Futures are what a user would probably - expect `Future<>` to be (it has this type for historical reasons and to - provide compatibility with old Flow `ACTOR`s). A coroutine with return type - `Future` must not return anything. So either the coroutine can run until - the end, or it can be terminated by calling `co_return`. -* `Generator` can return a stream of values. However, they can't `co_await` - other coroutines. These are useful for streams where the values are lazily - computed but don't involve any IO. -* `AsyncGenerator` is similar to `Generator` in that it can return a stream - of values, but in addition to that it can also `co_await` other coroutines. - Due to that, they're slightly less efficient than `Generator`. - `AsyncGenerator` should be used whenever values should be lazily generated - AND need IO. It is an alternative to `PromiseStream`, which can be more efficient, but is - more intuitive to use correctly. - -A more detailed explanation of `Generator` and `AsyncGenerator` can be -found further down. - -## Choose-When - -In actor compiled code we were able to use the keywords `choose` and `when` to wait on a -statically known number of futures and execute corresponding code. Something like this: +A function is a coroutine if its body contains `co_await`, `co_yield`, or `co_return`. Its return type must provide +a compatible coroutine implementation. The main types covered here are: -```c++ -choose { - when(wait(future1)) { - // do something - } - when(Foo f = wait(foo())) { - // do something else - } -} -``` +* `Future` returns one asynchronous result. A coroutine can await other Flow futures and return its value with + `co_return value;`. An unhandled error becomes the future's error. `co_yield` is not supported. +* `Future` represents asynchronous completion without a result value. Use `co_return;`, or reach the end of + the function body. By default, awaiting a `Future` also produces no value. +* `Generator` produces values synchronously, using `co_yield`. It provides an input iterator interface. +* `AsyncGenerator` produces values on demand and can both `co_await` asynchronous work and `co_yield` results. + Calling the generator requests its next value and returns a `Future`. + +`AsyncResult` is also available for a single consumer that does not need a copyable `Future`; it has a distinct +ownership contract. See the [internals reference](fdb-coroutines-internals.md) for that API and the runtime types +such as `SAV` and `Actor` that support coroutine execution. -Since this is a compiler functionality, we can't use this with C++ coroutines. For most -coroutine conversions, prefer `race(...)` and then branch on the returned `std::variant`. -This keeps the control flow explicit and usually maps more directly to what the coroutine -will do after the wait. `Choose` is still available for cases where the old `choose`-`when` -shape is the clearest fit, or where you need its ordered evaluation behavior described below. +## Waiting for Multiple Futures -For example, this actor pattern: +Prefer `race(...)` when waiting for one of several operations and then branching on the winner. It returns a +`std::variant` whose index matches the winning argument. For example: ```c++ -choose { - when(R res = wait(f1)) { - return res; - } - when(wait(timeout(...))) { - throw io_timeout(); +Future withDeadline(Future result, double timeoutSeconds) { + auto winner = co_await race(result, delay(timeoutSeconds)); + if (winner.index() == 0) { + co_return std::get<0>(winner); } + throw io_timeout(); } ``` -should usually become: +If multiple inputs are already ready, the first argument that is ready wins. An error from the winning input is +propagated rather than returned in the variant. For a `FutureStream`, the winning branch consumes one value. +Losing inputs are detached from the race, not explicitly cancelled, but dropping their last reference can cancel +them. All argument expressions are evaluated before `race` is called; use separate statements when their evaluation +order matters. -```c++ -auto result = co_await race(f1, timeout(...)); -if (result.index() == 0) { - co_return std::get<0>(result); -} -throw io_timeout(); -``` +Use helpers such as `quorum`, `waitForAll`, `waitForAllReady`, `timeoutError`, or `operator||` when they express the +required behavior more directly. `race` and `Choose` are defined in +[`CoroUtils.h`](../flow/include/flow/CoroUtils.h); the other helpers are in +[`genericactors.h`](../flow/include/flow/genericactors.h). -`Choose` remains useful when you specifically want callback-style handling of the winner: +### Ordered Evaluation with Choose + +`Choose` supports synchronous callbacks for the winning future: ```c++ co_await Choose() - .When(future1, [](Void& const) { - // do something + .When(future1, [](Void const&) { + // Handle the first result. }) - .When(foo(), [](Foo const& f) { - // do something else + .When(foo(), [](Foo const& value) { + // Handle the second result. }).run(); ``` -While `Choose` and `choose` behave very similarly, there are some minor differences between -the two. These are explained below. - -### Execution in when-expressions - -In the above example, there is one, potentially important difference between the old and new -style: in the statement `when(Foo f = wait(foo()))` is only executed if `future1` is not ready. -Depending on what the intent of the statement is, this could be desirable. Since `Choose::When` -is a normal method, `foo()` will be evaluated whether the statement is already done or not. -This can be worked around by passing a lambda that returns a Future instead: +Each `When` checks readiness in call order. A handler for an already-ready future can run immediately while the +chain is being constructed. Passing `foo()` still calls it even if an earlier branch has already won, because +it is an ordinary C++ argument expression. Pass a factory to avoid creating a later future unnecessarily: ```c++ co_await Choose() - .When(future1, [](Void& const) { - // do something + .When(future1, [](Void const&) { + // Handle the first result. }) - .When([](){ return foo() }, [](Foo const& f) { - // do something else + .When([]() { return foo(); }, [](Foo const& value) { + // Handle the second result. }).run(); ``` -The implementation of `When` will guarantee that this lambda will only be executed if all previous -`When` calls didn't receive a ready future. - -## Waiting in When-Blocks - -In FDB we sometimes see this pattern: - -```c++ -loop { - choose { - when(RequestA req = waitNext(requestAStream.getFuture())) { - wait(handleRequestA(req)); - } - when(RequestB req = waitNext(requestBStream.getFuture())) { - wait(handleRequestb(req)); - } - //... - } -} -``` - -This is not possible to do with `Choose`. However, this is done deliberately as the above is -considered an antipattern: This means that we can't serve two requests concurrently since the loop -won't execute until the request has been served. Instead, this should be written like this: +The factory is called only if no earlier branch received a ready future. Use `Choose` when this ordered, lazy +creation matters, or when synchronous callbacks make the code clearer. Handlers must return `void`; they cannot +suspend with `co_await`. If the winner needs asynchronous follow-up work, use `race` and await it in the outer +coroutine. -```c++ -state ActorCollection actors(false); -loop { - choose { - when(RequestA req = waitNext(requestAStream.getFuture())) { - actors.add(handleRequestA(req)); - } - when(RequestB req = waitNext(requestBStream.getFuture())) { - actors.add(handleRequestb(req)); - } - //... - when(wait(actors.getResult())) { - // this only makes sure that errors are thrown correctly - UNREACHABLE(); - } - } -} -``` +### Concurrent Request Handling -And so the above can easily be rewritten using `Choose`: +Awaiting a request handler inside a receive loop serializes request handling. This can be intentional, but when +requests should run concurrently, retain each handler's future in an `ActorCollection` and observe the collection's +result to propagate failures: ```c++ ActorCollection actors(false); -loop { - co_await Choose() - .When(requestAStream.getFuture(), [&actors](RequestA const& req) { - actors.add(handleRequestA(req)); - }) - .When(requestBStream.getFuture(), [&actors](RequestB const& req) { - actors.add(handleRequestB(req)); - }) - .When(actors.getResult(), [](Void const&) { - UNREACHABLE(); - }).run(); -} -``` - -### Migrating `choose`-`when` - -When porting actor code, use this rule of thumb: - -* Prefer `race(...)` when the `choose` picks one winner and then the code branches on which future completed. -* Use `Choose()` when you need ordered `when` evaluation, lazy creation of later futures, or the callback style is - genuinely clearer than branching on a `std::variant`. -* Use more specific helpers like `quorum`, `waitForAll`, `waitForAllReady`, or `operator||` when they express the - intent better than either `race` or `Choose`. - -`race(...)` is usually the best direct replacement for patterns like: - -```c++ -choose { - when(R res = wait(f1)) { - return res; - } - when(S res = wait(f2)) { - return use(res); +while (true) { + auto request = co_await race(requestAStream.getFuture(), requestBStream.getFuture(), actors.getResult()); + if (request.index() == 0) { + actors.add(handleRequestA(std::get<0>(request))); + } else if (request.index() == 1) { + actors.add(handleRequestB(std::get<1>(request))); + } else { + // With returnWhenEmptied=false, the collection only completes with an error. + UNREACHABLE(); } } ``` -which becomes: - -```c++ -auto result = co_await race(f1, f2); -if (result.index() == 0) { - co_return std::get<0>(result); -} -co_return use(std::get<1>(result)); -``` - -However, often using `choose`-`when` is overkill and other facilities like `quorum` and -`operator||` should be used instead. For example this: - -```c++ -choose { - when(R res = wait(f1)) { - return res; - } - when(wait(timeout(...))) { - throw io_timeout(); - } -} -``` - -Should be written like this: - -```c++ -co_await (f1 || timeout(...)); -if (f1.isReady()) { - co_return f1.get(); -} -throw io_timeout(); -``` - -(The above could also be packed into a helper function in `genericactors.h`). +Include [`ActorCollection.h`](../flow/include/flow/ActorCollection.h) for this pattern. Handlers must own any request +data they use after suspension; references into the local variant do not outlive the loop iteration. Bound +concurrency when needed rather than allowing an unbounded collection of outstanding work. ## Generators -With C++ coroutines we introduce two new basic types in Flow: `Generator` and `AsyncGenerator`. A generator is a -special type of coroutine, which can return multiple values. - -`Generator` and `AsyncGenerator` implement a different interface and serve a very different purpose. -`Generator` conforms to the `input_iterator` trait -- so it can be used like a normal iterator (with the exception -that copying the iterator has a different semantics). This also means that it can be used with the new `ranges` -library in STL which was introduced in C++20. +Generators separate value production from consumption. Use `Generator` for synchronous computation and +`AsyncGenerator` when producing the next value requires asynchronous work. -`AsyncGenerator` implements the `()` operator which returns a new value every time it is called. However, this value -HAS to be waited for (dropping it and attempting to call `()` again will result in undefined behavior!). This semantic -difference allows an author to mix `co_await` and `co_yield` statements in a coroutine returning `AsyncGenerator`. +### Generators and Ranges -Since generators can produce infinitely long streams, they can be useful to use in places where we'd otherwise use a -more complex in-line loop. For example, consider the code in `masterserver.actor.cpp` that is responsible generate -version numbers. The logic for this code is currently in a long function. With a `Generator` it can be isolated to -one simple coroutine (which can be a direct member of `MasterData`). A simplified version of such a generator could -look as follows: +A `Generator` exposes its current value through `*generator` and advances with `++generator`. Copies share the +same coroutine and iteration position; they do not create independent sequences. This makes it an input iterator, +not a multipass iterator. ```c++ -Generator MasterData::versionGenerator() { - auto prevVersion = lastEpochEnd; - auto lastVersionTime = now(); +// Produces base^0, base^1, base^2, ... +Generator powersOf(double base) { + double current = 1; while (true) { - auto t1 = now(); - Version toAdd = - std::max(1, - std::min(SERVER_KNOBS->MAX_READ_TRANSACTION_LIFE_VERSIONS, - SERVER_KNOBS->VERSIONS_PER_SECOND * (t1 - self->lastVersionTime))); - lastVersionTime = t1; - co_yield prevVersion + toAdd; - prevVersion += toAdd; + co_yield current; + current *= base; } } ``` -Now that the logic to compute versions is separated, `MasterData` can simply create an instance of `Generator` -by calling `auto vGenerator = MasterData::versionGenerator();` (and possibly storing that as a class member). It can -then access the current version by calling `*vGenerator` and go to the next generator by incrementing the iterator -(`++vGenerator`). - -`AsyncGenerator` should be used in some places where we used promise streams before (though not all of them, this -topic is discussed a bit later). For example: +Use `std::ranges::subrange` with `Generator::end()` to apply range adaptors: ```c++ -template -AsyncGenerator filter(AsyncGenerator gen, F pred) { - while (gen) { - auto val = co_await gen(); - if (pred(val)) { - co_yield val; - } - } +auto powers = std::ranges::subrange(powersOf(2), Generator::end()); +for (double value : powers + | std::views::filter([](double value) { return value > 10; }) + | std::views::take(10)) { + fmt::print("{}\n", value); } ``` -Note how much simpler this function is compared to the old flow function: - -```c++ -ACTOR template -Future filter(FutureStream input, F pred, PromiseStream output) { - loop { - try { - T nextInput = waitNext(input); - if (pred(nextInput)) - output.send(nextInput); - } catch (Error& e) { - if (e.code() == error_code_end_of_stream) { - break; - } else - throw; - } - } - - output.sendError(end_of_stream()); - - return Void(); -} -``` +This prints ten powers of two, from 16 through 8192. Include `` for the standard range facilities. -A `FutureStream` can be converted into an `AsyncGenerator` by using a simple helper function: +An asynchronous generator requests its next value with `co_await generator()`: ```c++ -template -AsyncGenerator toGenerator(FutureStream stream) { - loop { - try { - co_yield co_await stream; - } catch (Error& e) { - if (e.code() == error_code_end_of_stream) { - co_return; - } - throw; - } +AsyncGenerator delayedValues(int count) { + for (int i = 0; i < count; ++i) { + co_await delay(0.01); + co_yield i; } } ``` -### Generators and Ranges +Keep the generator alive until each request finishes, and await that request before making another one. Do not copy +an `AsyncGenerator` or abandon an outstanding request. When the generator reaches the end of its body or executes +`co_return;`, the pending request reports `end_of_stream`. Checking `if (generator)` only tells you whether it has +already finished; the next request can still discover the end of the stream. -`Generator` can be used like an input iterator. This means, that it can also be used with `std::ranges`. Consider -the following coroutine: +[`toGenerator`](../flow/include/flow/CoroUtils.h) adapts a `FutureStream` to an `AsyncGenerator`, translating the +stream's `end_of_stream` into generator completion while preserving other errors. -```c++ -// returns base^0, base^1, base^2, ... -Generator powersOf(double base) { - double curr = 1; - loop { - co_yield curr; - curr *= base; - } -} -``` +### Execution and Value Ownership -We can use this now to generate views. For example: - -```c++ -for (auto v : generatorRange(powersOf(2)) - | std::ranges::views::filter([](auto v) { return v > 10; }) - | std::ranges::views::take(10)) { - fmt::print("{}\n", v); -} -``` +The execution policy depends on the return type: -The above would print all powers of two between 10 and 2^10. +* `Future` coroutines begin immediately and run until suspension or completion. +* `Generator` begins immediately and runs to its first `co_yield` or completion. Incrementing it resumes production. +* `AsyncGenerator` initially suspends. Calling its `()` operator resumes production until a value, error, or + asynchronous suspension is reached. -### Eager vs Lazy Execution - -One major difference between async generators and tasks (coroutines returning only one value through `Future`) is the -execution policy: An async generator will immediately suspend when it is called while a task will immediately start -execution and needs to be explicitly scheduled. - -This is a conscious design decision. Lazy execution makes it much simpler to reason about memory ownership. For example, -the following is ok: +Both generator types suspend at `co_yield` until the next value is requested. This permits a generator to reuse +storage between values, but the consumer must respect that storage's lifetime. For example: ```c++ Generator randomStrings(int minLen, int maxLen) { Arena arena; auto buffer = new (arena) uint8_t[maxLen + 1]; while (true) { - auto sz = deterministicRandom()->randomInt(minLen, maxLen + 1); - for (int i = 0; i < sz; ++i) { + auto size = deterministicRandom()->randomInt(minLen, maxLen + 1); + for (int i = 0; i < size; ++i) { buffer[i] = deterministicRandom()->randomAlphaNumeric(); } - co_yield StringRef(buffer, sz); + co_yield StringRef(buffer, size); } } ``` -The above coroutine returns a stream of random strings. The memory is owned by the coroutine and so it always returns -a `StringRef` and then reuses the memory in the next iteration. This makes this generator very cheap to use, as it only -does one allocation in its lifetime. With eager execution, this would be much harder to write (and reason about): the -coroutine would immediately generate a string and then eagerly compute the next one when the string is retrieved. -However, in Flow a `co_yield` is guaranteed to suspend the coroutine until the value was consumed (this is not generally -a guarantee with `co_yield` -- C++ coroutines give the implementer a great degree of freedom over decisions like this). +Each `StringRef` points into the generator's arena. Its contents can change when the generator advances, and the +storage is freed when the generator is destroyed. Copy a value into owning storage, such as `Standalone`, +if it must survive either event. The same requirement applies when a consumer passes a view to background work. ### Generators vs Promise Streams -Flow provides another mechanism to send streams of messages between actors: `PromiseStream`. In fact, -`AsyncGenerator` uses `PromiseStream` internally. So when should one be used over the other? +`AsyncGenerator` internally uses `PromiseStream`, but the two interfaces express different production policies. +A generator produces another value only when requested. A promise stream lets a producer enqueue values independently +of the consumer. -As a general rule of thumb: whenever possible, use `Generator`, if not, use `AsyncGenerator` if in doubt. +Prefer a synchronous generator for simple computation. Use an asynchronous generator for demand-driven IO, such as +reading the next block of a file. Use a promise stream when production should run ahead, when multiple sources feed +a stream, or when an existing stream interface fits the operation. For example, prefetching file blocks can hide IO +latency while the consumer processes earlier blocks. -For pure computation it almost never makes sense to use a `PromiseStream` (the only exception is if computation -can be expensive enough that `co_await yield()` becomes necessary). `Generator` is more lightweight and therefore -usually more efficient. It is also easier to use. +A producer that runs ahead needs explicit bounds or backpressure, an owner for its future, and error propagation to +the consumer. Send owning values when the producer will reuse or release the underlying storage. A `PromiseStream` +does not by itself bound its queue or keep a producer coroutine alive. -When it comes to IO it becomes a bit more tricky. Assume we want to scan a file on disk, and we want to read it in -4k blocks. This can be done quite elegantly using a coroutine: +## Cancellation -```c++ -AsyncGenerator> blockScanner(Reference file) { - auto sz = co_await file->size(); - decltype(sz) offset = 0; - constexpr decltype(sz) blockSize = 4*1024; - while (offset < sz) { - Arena arena; - auto block = new (arena) int8_t[blockSize]; - auto toRead = std::min(sz - offset, blockSize); - auto r = co_await file->read(block, toRead, offset); - co_yield Standalone(StringRef(block, r), arena); - offset += r; - } -} -``` +By default, explicitly cancelling a coroutine's future or dropping its last future reference requests cancellation. +A suspended coroutine resumes and its await throws `actor_cancelled`. If cancellation is requested while it is +running, the next Flow await observes it. Subsequent Flow awaits also throw cancellation rather than waiting. -The problem with the above generator though, is that we only start reading when the generator is invoked. If consuming -the block takes sometimes a long time (for example because it has to be written somewhere), each call will take as long -as the disk latency is for a read. +Use RAII for cleanup, and rethrow cancellation from error handlers unless the coroutine's contract explicitly +requires something else. Do not silently consume `broken_promise` or other failures either. Keeping a future in a +local or an `ActorCollection` makes the lifetime of asynchronous work explicit. -What if we want to hide this latency? In other words: what if we want to improve throughput and end-to-end latency by -prefetching? +### Uncancellable -Doing this with a generator, while not trivial, is possible. But here it might be easier to use a `PromiseStream` -(we can even reuse the above generator): +Some operations must finish even if their caller stops waiting. Add an `Uncancellable` marker parameter to make +cancellation of the returned future a no-op: ```c++ -Future blockScannerWithPrefetch(Reference file, - PromiseStream promise, - FlowLock lock) { - auto generator = blockScanner(file); - while (generator) { - { - FlowLock::Releaser _(co_await lock.take()); - try { - promise.send(co_await generator()); - } catch (Error& e) { - promise.sendError(e); - co_return; - } - } - // give caller opportunity to take the lock - co_await yield(); - } +Future finishOperation(Future operation, Uncancellable = {}) { + co_await operation; } ``` -With the above the caller can control the prefetching dynamically by taking the lock if the queue becomes too full. - -## Uncancellable - -By default, a coroutine runs until it is either done (reaches the end of the function body, a `co_return` statement, -or throws an exception) or the last `Future` object referencing that object is being dropped. The second use-case is -implemented as follows: - -1. When the future count of a coroutine goes to `0`, the coroutine is immediately resumed and `actor_cancelled` is - thrown within that coroutine (this allows the coroutine to do some cleanup work). -2. Any attempt to run `co_await expr` will immediately throw `actor_cancelled`. - -However, some coroutines aren't safe to be cancelled. This usually concerns disk IO operations. With `ACTOR` we could -either have a return-type `void` or use the `UNCANCELLABLE` keyword to change this behavior: in this case, calling -`Future::cancel()` would be a no-op and dropping all futures wouldn't cause cancellation. - -However, with C++ coroutines, this won't work: - -* We can't introduce new keywords in pure C++ (so `UNCANCELLABLE` would require some preprocessing). -* Implementing a `promise_type` for `void` isn't a good idea, as this would make any `void`-function potentially a - coroutine. - -However, this can also be seen as an opportunity: uncancellable actors are always a bit tricky to use, since we need to -make sure that the caller keeps all memory alive that the uncancellable coroutine might reference until it is done. -Because of that, whenever someone calls a coroutine, they need to be extra careful. However, someone might not know that -the coroutine they call is uncancellable. - -We address this problem with the following definition: - ---- -*Definition*: - -A coroutine is uncancellable if the first argument (or the second, if the coroutine is a class-member) is of type -`Uncancellable` - ---- - -The definition of `Uncancellable` is trivial: `struct Uncancellable {};` -- it is simply used as a marker. So now, if -a user calls an uncancellable coroutine, it will be obvious on the caller side. For example the following is *never* -uncancellable: - -```c++ -co_await foo(); -``` - -But this one is: +Dropping all references to the returned future does not cancel this coroutine. It continues until completion or +failure, so it must own any resources and data it needs for that duration. Use this marker only when required by the +operation's lifetime contract; it does not prevent an awaited operation from failing. -```c++ -co_await bar(Uncancellable()); -``` +### NoThrowOnCancel -## NoThrowOnCancel - -`NoThrowOnCancel` is a marker for coroutines that should still cancel, but should not run cancellation through an -`actor_cancelled` exception inside the coroutine: +`NoThrowOnCancel` keeps cancellation enabled but destroys the coroutine frame without resuming it to throw +`actor_cancelled` inside the coroutine: ```c++ -Future foo(NoThrowOnCancel = {}) { - Resource r; - co_await something(); +Future waitForSignal(Future signal, NoThrowOnCancel = {}) { + co_await signal; } ``` -When a coroutine with this marker is cancelled, the coroutine frame is destroyed and normal C++ RAII cleanup runs for -locals in scope. `catch` blocks inside the coroutine do not observe `actor_cancelled` for cancellation. This differs from -`Uncancellable`, where cancelling the returned future is a no-op and the coroutine continues until completion. - -## Porting `ACTOR`'s to C++ Coroutines - -If you have an existing `ACTOR`, you can port it to a C++ coroutine by following these steps: - -1. Remove `ACTOR` keyword. -2. If the actor is marked with `UNCANCELLABLE`, remove it and make the first argument `Uncancellable`. If the return - type of the actor is `void` make it `Future` instead and add an `Uncancellable` as the first argument. -3. Remove all `state` modifiers from local variables. -4. Replace all `wait(expr)` with `co_await expr`. -5. Remove all `waitNext(expr)` with `co_await expr`. -6. Rewrite existing `choose-when` statements by preferring `race(...)` for first-ready branching; use `Choose` - only when you need ordered `when` semantics or callback-style handling. - -In addition, the following things should be looked out for: - -### Lifetime of locals - -Consider this code: - -```c++ -Local foo; -wait(bar()); -... -``` - -`foo` will be destroyed right after the `wait`-expression. However, after making this a coroutine: - -```c++ -Local foo; -co_await bar(); -... -``` +Cancellation runs normal RAII cleanup for locals in scope, but it does not enter the coroutine's `catch` handlers. +Observers of its returned future still receive `actor_cancelled`. `NoThrowOnCancel` and `Uncancellable` are mutually +exclusive, and neither marker is supported by `AsyncGenerator`. -`foo` will stay alive until we leave the scope. This is better (as it is more intuitive and follows standard C++), but -in some weird corner-cases code might depend on the semantic that locals get destroyed when we call into `wait`. Look -out for things where destructors do semantically important work (like in `FlowLock::Releaser`). +## Lifetime and Ownership -### Unnecessary Helper Actors - -In `flow/genericactors.h` we have a number of useful helpers. Some of them are also useful with C++ coroutines, -others add unnecessary overhead. Look out for those and remove calls to it. The most important ones are `success` and -`store`. - -```c++ -wait(success(f)); -``` - -becomes - -```c++ -co_await f; -``` - -and - -```c++ -wait(store(v, f)); -``` - -becomes - -```c++ -v = co_await f; -``` +### Locals and Scope -### Replace Locals with Temporaries +Coroutine locals follow normal C++ scoping rules. A local remains alive across suspension until its scope exits. +For objects such as lock releasers, make sure that this is the intended lifetime: retaining a lock across a wait +can block other work, while leaving a scope releases it. -In certain places we use locals just to work around actor compiler limitations. Since locals use up space in the -coroutine object they should be removed wherever it makes sense (only if it doesn't make the code less readable!). +The same rule applies to futures. A future declared inside an `if` or `try` block is destroyed at that block's end. +If it is the last reference to pending cancellable work, that work is cancelled. Declare the future in the scope +that must retain it, or add it to an `ActorCollection`; observe its result as well as retaining it. -For example: +Initialize local values explicitly. For a passive aggregate with primitive fields, `SomeStruct value{};` initializes +those fields, while `SomeStruct value;` may leave them uninitialized. Coroutine frames do not change C++ initialization +rules. -```c++ -Foo f = wait(foo); -bar(f); -``` +### Parameters and Objects -might become +Prefer owning parameters passed by value for data used after suspension. Coroutine reference parameters remain +references; the frame does not copy the referenced object. A temporary argument or caller-local object can therefore +be destroyed while the coroutine is suspended. ```c++ -bar(co_await foo); +Future printLater(Key key) { + co_await delay(1.0); + fmt::print("{}\n", key.toString()); +} ``` -### Don't Wait in Error-Handlers +Here the frame owns a `Key`. Passing a `KeyRef`, `ValueRef`, or `StringRef` by value only copies a view and does not +extend its arena's lifetime. Use an owning type such as `Key`, `Value`, or `Standalone` when bytes must outlive +the caller. -Using `co_await` in an error-handler produces a compilation error in C++. However, this was legal with `ACTOR`. There -is no general best way of addressing this issue, but usually it's quite easy to move the `co_await` expression out of -the `catch`-block. +If an API must take a reference, establish the required ownership before suspension and use only the owned copy +thereafter. Copying inside the body before the first await works for an eager `Future` coroutine, but not for an +initially suspended `AsyncGenerator` whose caller may already have destroyed the argument before the body starts. -One place where we use this pattern a lot if in our transaction retry loop: +A coroutine can be a non-static member function, but its `this` pointer does not keep the object alive. The owner must +outlive the coroutine, or the coroutine must retain an appropriate owning reference. Similarly, a coroutine lambda's +captures belong to its closure object; keep that object alive or pass owned values as coroutine parameters instead. -```c++ -state ReadYourWritesTransaction tr(db); -loop { - try { - Value v = wait(tr.get(key)); - tr.set(key2, val2); - wait(tr.commit()); - return Void(); - } catch (Error& e) { - wait(tr.onError(e)); - } -} -``` +## Error Handlers -Luckily, with coroutines, we can do one better: generalize the retry loop. The above could look like this: +C++ does not allow `co_await` inside a `catch` handler. Save the error and await recovery after leaving the handler. +For a transaction retry loop, for example: ```c++ -co_await db.run([&](ReadYourWritesTransaction* tr) -> Future { - Value v = wait(tr->get(key)); - tr->set(key2, val2); - wait(tr->commit()); -}); -``` - -A possible implementation of `Database::run` would be: - -```c++ -template Fun> -Future Database::run(Fun fun) { - ReadYourWritesTransaction tr(*this); - Future onError; +Future writeKey(Database db, Key key, Value value) { + ReadYourWritesTransaction tr(db); while (true) { - if (onError.isValid()) { - co_await onError; - onError = Future(); - } + Error error; try { - co_await fun(&tr); + tr.set(key, value); + co_await tr.commit(); co_return; } catch (Error& e) { - onError = tr.onError(e); + if (e.code() == error_code_actor_cancelled) { + throw; + } + error = e; } + co_await tr.onError(error); } } ``` -### Make Static Functions Class Members - -With actors, we often see the following pattern: - -```c++ -struct Foo : IFoo { - ACTOR static Future bar(Foo* self) { - // use `self` here to access members of `Foo` - } - - Future bar() override { - return bar(this); - } -}; -``` - -This boilerplate is necessary, because `ACTOR`s can't be class members: the actor compiler will generate another -`struct` and move the code there -- so `this` will point to the actor state and not to the class instance. - -With C++ coroutines, this limitation goes away. So a cleaner (and slightly more efficient) implementation of the above -is: - -```c++ -struct Foo : IFoo { - Future bar() override { - // `this` can be used like in any non-coroutine. `co_await` can be used. - } -}; -``` +The successful path returns before recovery; only the failed path calls `onError`. That call applies the transaction's +retry policy and propagates non-retryable errors. Include +[`ReadYourWrites.h`](../fdbclient/include/fdbclient/ReadYourWrites.h) for this transaction API. -### Initialization of Locals +## Direct Await Expressions -There is one very subtle and hard to spot difference between `ACTOR` and a coroutine: the way some local variables are -initialized. Consider the following code: +Await a future directly when no separate helper is needed: ```c++ -struct SomeStruct { - int a; - bool b; -}; - -ACTOR Future someActor() { - // beginning of body - state SomeStruct someStruct; - // rest of body -} -``` - -For state variables, the actor-compiler generates the following code to initialize `SomeStruct someStruct`: - -```c++ -someStruct = SomeStruct(); +co_await future; // Wait and discard a non-Void result. +value = co_await anotherFuture; // Wait and store the result. +consume(co_await nextValue); // Use the result in an expression. ``` -This, however, is different from what might expect since now the default constructor is explicitly called. This means -if the code is translated to: - -```c++ -Future someActor() { - // beginning of body - SomeStruct someStruct; - // rest of body -} -``` - -initialization will be different. The exact equivalent instead would be something like this: - -```c++ -Future someActor() { - // beginning of body - SomeStruct someStruct{}; // auto someStruct = SomeStruct(); - // rest of body -} -``` - -If the struct `SomeStruct` would initialize its primitive members explicitly (for example by using `int a = 0;` and -`bool b = false`) this would be a non-issue. And explicit initialization is probably the right fix here. Sadly, it -doesn't seem like UBSAN finds these kind of subtle bugs. - -Another difference is, that if a `state` variables might be initialized twice: once at the creation of the actor using -the default constructor and a second time at the point where the variable is initialized in the code. With C++ -coroutines we now get the expected behavior, which is better, but nonetheless a potential behavior change. - -### `state` Variables Inside Blocks - -The actor compiler **hoists** all `state` variables into the actor's state struct, regardless of C++ block scope. This -means a `state` variable declared inside an `if`, `else`, `for`, or `try` block lives for the entire actor lifetime. -In a coroutine, these become regular C++ locals that follow normal scoping rules. - -This is a source of subtle bugs. Consider: - -```c++ -ACTOR Future example() { - if (someCondition) { - state Future background = longRunningTask(); - } - // In ACTOR code, `background` is still alive here — it was hoisted. - wait(delay(100.0)); - return Void(); -} -``` - -A naive conversion: - -```c++ -Future example() { - if (someCondition) { - Future background = longRunningTask(); - } - // BUG: `background` was destroyed at the `}` above, cancelling longRunningTask()! - co_await delay(100.0); -} -``` - -The fix is to move the variable to function scope: - -```c++ -Future example() { - Future background; - if (someCondition) { - background = longRunningTask(); - } - // `background` is still alive — correct. - co_await delay(100.0); -} -``` - -**Rule**: When removing `state` from a variable, check whether it is declared inside a block. If so, move the -declaration to function scope. - -### `const&` Parameters - -C++20 coroutines only store a reference in the coroutine frame for `const&` parameters — they do **not** copy the -argument. If the caller passes a temporary (e.g. a default argument value, or a local that goes out of scope), the -reference dangles after the first suspend point. - -```c++ -// DANGEROUS: if caller passes a temporary, `key` dangles after first co_await -Future doSomething(Key const& key) { - co_await delay(1.0); - fmt::print("{}\n", key.toString()); // potential use-after-free -} -``` - -The fix is to copy `const&` parameters to locals before the first `co_await`: - -```c++ -Future doSomething(Key const& key) { - Key keyCopy = key; // safe copy before any suspend - co_await delay(1.0); - fmt::print("{}\n", keyCopy.toString()); // OK -} -``` - -**Rule**: Copy all `const&` parameters to local variables before the first `co_await`. - -### Forward Declarations in `.actor.h` Files - -When a function is converted from `ACTOR` to a coroutine, any forward declarations in `.actor.h` files must have the -`ACTOR` keyword removed. The actor compiler automatically adds `const&` to all parameters in `ACTOR` declarations. -If you also write `const&` explicitly, the generated code will contain `const& const&`, which is a compile error. - -```c++ -// workloads.h — WRONG: ACTOR + const& = double const& -ACTOR Future foo(Database const& cx); - -// workloads.h — CORRECT: remove ACTOR since foo() is now a coroutine -Future foo(Database const& cx); -``` - -### `DESCR` Deprecation - -The older TDMetric `DESCR` shorthand is deprecated. When a coroutine conversion touches metric event types, do not add -new `DESCR(...)`-style declarations. Instead, define an explicit payload type with a -`...Descriptor` suffix and specialize `Descriptor` with `DescribeType<...>` and `DescribeField<...>` next to it. - -This keeps descriptor types unambiguous in coroutine-converted code and matches the old TDMetric pattern in files -such as `flow/EventTypes.h`, `fdbclient/EventTypes.h`, and the workload metric definitions. - -### File Naming - -Converted files should be renamed from `.actor.cpp` to `.cpp` (or `.actor.h` to `.h`) since they no longer need the -actor compiler. Both `fdbserver` and `flow_bench` use `fdb_find_sources()` in their `CMakeLists.txt`, which -automatically picks up files by glob, so the rename is usually sufficient without any CMake changes. - -### Conversion Checklist - -1. Rename the file from `.actor.cpp` to `.cpp`. -2. Remove `ACTOR` from all function definitions. -3. Remove `UNCANCELLABLE`; add `Uncancellable` as the first parameter instead. -4. Remove `state` from all local variable declarations. - - **Check**: is the variable inside a block (`if`/`else`/`for`/`try`)? If so, move it to function scope. -5. Replace `wait(expr)` with `co_await expr`. Replace `waitNext(expr)` with `co_await expr`. -6. Replace `return expr` with `co_return expr`. Replace `return Void()` with `co_return`. -7. Rewrite `choose`/`when` by preferring `race(...)`; use `Choose` only for patterns that do not map cleanly to - `race`. -8. Simplify: `wait(success(f))` → `co_await f`; `wait(store(v, f))` → `v = co_await f`. -9. For `const&` parameters: copy to a local before the first `co_await`. -10. Remove `ACTOR` from any forward declarations of the converted functions in `.actor.h` files. -11. Build and run simulation tests to verify correctness. - -## Performance Analysis & Optimization - -### Performance Summary (Updated February 2026) - -Through optimization and profiling analysis, C++20 coroutines have made some performance improvements, reducing the gap with ACTOR-generated code from ~10% to 3-8% depending on workload patterns. - -#### Current Linux Performance Results (32-core, 3.1 GHz) - -``` -Benchmark Type ACTOR Performance Coroutine Performance Gap Status --------------- ---------------- -------------------- --- ------ -NET2/4096 2.67M/s 2.41M/s -8.5% Target for optimization -YIELD/4096 7.45M/s 13.6M/s +83% Coroutines much faster ✅ -DELAY/4096 1.44M/s 5.22M/s +260% Coroutines much faster ✅ -CALLBACK/1024/64 50.9M/s 8.7M/s (some patterns) -82% Mixed results -``` - -#### Key Insight: Workload Pattern Dependency - -**Coroutines excel in frame-reuse patterns** (YIELD, DELAY) where a single coroutine is suspended/resumed many times. - -**Coroutines lag in allocation-heavy patterns** (NET2) where many short-lived coroutines are created and destroyed. - -### Performance Analysis Deep Dive - -#### Root Cause Identification (February 2026) - -**Original Analysis**: Coroutines had 39.13% CPU overhead in `final_suspend()` that actors completely avoid. - -``` -ACTORS (2.67M/s): 43.31% CPU in direct ActorCallback::fire() -COROUTINES (2.41M/s): 35.61% CPU in QuorumCallback + other overhead = ~75% total -``` - -#### Fix - -**Implementation**: Moved SAV cleanup from `final_suspend()` to `return_value()` to match actor completion timing. - -**Result**: Eliminated final_suspend() overhead from performance profiles (39.13% → 0.21% CPU usage). - -### Current Bottlenecks (February 2026) - -Based on comprehensive Linux profiling of optimized coroutines: - -#### 1. QuorumCallback Overhead (35.61% CPU) -- **Impact**: Shared bottleneck between actors and coroutines -- **Cause**: Callback chain traversal in SAV system -- **Optimization**: Compiler hints provide minimal improvement - -#### 2. FastAllocator<128> Waste (7.19% CPU) -- **Impact**: Frame allocation overhead in NET2 pattern -- **Cause**: Some coroutine frames exceed 64-byte optimal bucket size -- **Evidence**: 3.69% allocate + 3.50% release CPU usage -- **Attempts**: Custom allocator forcing provided <1% improvement - -#### 3. AwaitableFuture Operations (3.66% CPU) -- **Impact**: Coroutine-specific suspend/resume overhead -- **Components**: 2.79% fire() + 0.87% resumeImpl() -- **Nature**: Inherent to C++20 coroutine mechanics - -### Optimization Techniques - What Works and What Doesn't - -#### ✅ Successful Optimizations - -1. **Compiler optimization hints**: `__attribute__((hot))`, `__attribute__((always_inline))`, `__attribute__((flatten))` - - **Impact**: 2-5% performance improvements in hot paths - -2. **Branch prediction hints**: `[[likely]]`, `[[unlikely]]` - - **Impact**: Optimizes common vs error paths - -3. **Architectural changes**: Moving SAV cleanup from final_suspend() to return_value() - - **Impact**: Eliminated 39.13% CPU bottleneck (99.5% reduction) - -#### ❌ Ineffective Optimizations - -1. **Custom FastAllocator forcing**: Attempted to force frames into smaller buckets - - **Result**: Only 0.82% reduction in FastAllocator<128> overhead - - **Risk**: Unsafe for frames that don't fit smaller buckets - -2. **Frame packing**: `__attribute__((packed))`, pointer bit-packing - - **Result**: Added overhead from indirection outweighed space savings - - **Issue**: Increased function call overhead - -3. **Aggressive final_suspend() bypass**: Attempted to skip SAV operations entirely - - **Result**: Broke Flow's reference counting semantics (double-free crashes) - -### Performance Comparison by Platform - -#### Linux (Release, -O3) -- **Coroutines**: 2.41M/s NET2, 13.6M/s YIELD -- **Actors**: 2.67M/s NET2, 7.45M/s YIELD - -#### macOS (Debug, -g) -- **Coroutines**: 930k/s NET2 (significantly slower) -- **Platform difference**: 2.56x performance gap between Linux and macOS - -### Benchmark Comparison Tool - -#### Generating Comprehensive Performance Reports - -To generate complete actor vs coroutine performance comparison reports (matching historical format): - -```bash -cd build_output # or your build directory -python3 ../contrib/benchmark_comparison.py -``` - -**Output**: Complete comparison across all benchmark types: -- DELAY benchmarks (DELAY + YIELD variants, all scales) -- NET2 benchmarks (allocation-heavy patterns, all scales) -- CALLBACK benchmarks (various template sizes and scales) -- OVERALL_GEOMEAN calculations for statistical analysis - -**Requirements**: -- Working flow_bench binary with both actor and coroutine benchmarks -- Benchmark infrastructure must include: bench_net2, coroutine_net2, bench_delay, coroutine_delay_bench, coroutine_yield_bench, bench_callback, coroutine_callback - -**Usage**: Tool automatically runs benchmarks and generates comparison report in the format matching historical coroutine optimization reports. - -### Future Optimization Opportunities - -#### High-Impact Targets -1. **QuorumCallback optimization** (35.61% CPU) - requires deeper architectural changes -2. **Frame allocation strategy** - investigate frame pooling for allocation-heavy patterns -3. **Profile-guided optimization** - compiler-level optimization based on runtime profiles +Use named locals when they clarify ownership or control flow. Helpers remain useful when adapting futures for other +APIs, but wrapping a future in `success` or `store` is unnecessary just to await it in a coroutine. diff --git a/design/data-distributor-internals.md b/design/data-distributor-internals.md index b5b909d1c8f..29b61bd3a09 100644 --- a/design/data-distributor-internals.md +++ b/design/data-distributor-internals.md @@ -666,8 +666,8 @@ Knobs: `AVAILABLE_SPACE_PIVOT_RATIO` (0.5), `MIN_AVAILABLE_SPACE_RATIO` (0.05), `AVAILABLE_SPACE_RATIO_CUTOFF` (0.05), `MIN_AVAILABLE_SPACE` (100 MB). Sources: `TCTeamInfo::getLoadBytes` / `getLoadAverage` / `getMinAvailableSpaceRatio` / `hasHealthyAvailableSpace` in `datadistributor/TCInfo.cpp`; `updateAvailableSpacePivots` / -`updateTeamEligibility` / `getBestTeam` in `datadistributor/DDTeamCollection.actor.cpp`; -the rebalance threshold in `datadistributor/DDRelocationQueue.actor.cpp`. +`updateTeamEligibility` / `getBestTeam` in `datadistributor/DDTeamCollection.cpp`; +the rebalance threshold in `datadistributor/DDRelocationQueue.cpp`. ## 9. Team Building @@ -866,12 +866,14 @@ invariant that every key in the database is assigned to a team. ### 14.3 The moveKeys Protocol in Detail The data transfer protocol is implemented in [`MoveKeys.actor.cpp`](https://github.com/apple/foundationdb/blob/release-7.3/fdbserver/MoveKeys.actor.cpp) and consists of three -coordinated phases. The [`moveKeys()`](https://github.com/apple/foundationdb/blob/release-7.3/fdbserver/MoveKeys.actor.cpp#L3180) function orchestrates them: +coordinated phases. The [`moveKeys()`](https://github.com/apple/foundationdb/blob/release-7.3/fdbserver/MoveKeys.actor.cpp#L3180) function orchestrates them. The sketch below uses the current coroutine syntax from [`MoveKeys.cpp`](https://github.com/apple/foundationdb/blob/main/fdbserver/core/MoveKeys.cpp): ```cpp -wait(rawStartMovement(occ, params, tssMapping)); // Phase 1: startMoveKeys -wait(rawCheckFetchingState(occ, params, ...)); // Phase 2: monitor data copy -wait(rawFinishMovement(occ, params, tssMapping)); // Phase 3: finishMoveKeys +co_await rawStartMovement(occ, params, tssMapping); // Phase 1: startMoveKeys +Future completionSignaller = + rawCheckFetchingState(occ, params, tssMapping); // Phase 2: monitor data copy concurrently +co_await rawFinishMovement(occ, params, tssMapping); // Phase 3: finishMoveKeys +completionSignaller.cancel(); ``` #### Phase 1: `startMoveKeys` ([`MoveKeys.actor.cpp:924`](https://github.com/apple/foundationdb/blob/release-7.3/fdbserver/MoveKeys.actor.cpp#L924)) @@ -1041,18 +1043,22 @@ gap that we address in the recommendations document. ## 15. The High-Level moveKeys Orchestration Wrapping the protocol above, [`moveKeys()`](https://github.com/apple/foundationdb/blob/release-7.3/fdbserver/MoveKeys.actor.cpp#L3180) orchestrates -the three phases: +the three phases. Its [current implementation](https://github.com/apple/foundationdb/blob/main/fdbserver/core/MoveKeys.cpp) uses C++ coroutines: ```cpp -ACTOR Future moveKeys(Database occ, MoveKeysParams params) { - wait(rawStartMovement(occ, params, tssMapping)); // startMoveKeys - state Future completionSignaller = +Future moveKeys(Database occ, MoveKeysParams params) { + ASSERT(!params.destinationTeam.empty()); + std::sort(params.destinationTeam.begin(), params.destinationTeam.end()); + std::map tssMapping; + + co_await rawStartMovement(occ, params, tssMapping); // startMoveKeys + Future completionSignaller = rawCheckFetchingState(occ, params, tssMapping); // monitor fetch progress - wait(rawFinishMovement(occ, params, tssMapping)); // finishMoveKeys + co_await rawFinishMovement(occ, params, tssMapping); // finishMoveKeys completionSignaller.cancel(); if (!params.dataMovementComplete.isSet()) params.dataMovementComplete.send(Void()); - return Void(); + co_return; } ``` diff --git a/design/fdb-coroutines-internals.md b/design/fdb-coroutines-internals.md index 83f1da78287..fef0c3dc564 100644 --- a/design/fdb-coroutines-internals.md +++ b/design/fdb-coroutines-internals.md @@ -259,11 +259,8 @@ int8_t actor_wait_state; | -1 | `ACTOR_WAIT_STATE_CANCELLED` | Cancellation requested | | -2 | `ACTOR_WAIT_STATE_CANCELLED_DURING_READY_CHECK` | Cancellation detected during `await_ready()` before the awaiter registered a callback | -In actor-compiler-generated code, values >1 identify *which* callback group -fired (the generated state machine uses the value to jump to the right -continuation). For C++ coroutines this distinction is unnecessary — the -`coroutine_handle` already encodes the suspend point — so the value is -always 1. +The `coroutine_handle` encodes the suspension point, so a suspended C++ +coroutine uses the single waiting value 1. Constructed with `SAV(1, 1)`: starts with futures=1 and promises=1. The futures=1 is consumed by the `Future` returned from `get_return_object()`. @@ -612,14 +609,13 @@ either resumes the producer to throw `actor_cancelled()` or (with --- -## `choose` / `when` for Coroutines +## Waiting for Multiple Futures ``` flow/include/flow/CoroUtils.h ``` -The coroutine equivalent of the actor compiler's `choose { when(...) { ... } }` -pattern. Two forms: +Flow provides two ways to await the first of several futures: ### `Choose().When(future, handler).When(...).run()` diff --git a/design/mocks3server_chaos_design.md b/design/mocks3server_chaos_design.md index 0e41be8cbab..330cf2f3f83 100644 --- a/design/mocks3server_chaos_design.md +++ b/design/mocks3server_chaos_design.md @@ -8,7 +8,7 @@ Philosophy: mocks3 should be more intolerant/strict than real s3 ## Problem -FoundationDB's S3BlobStore client needs thorough testing against realistic S3 failure scenarios, but the existing [`MockS3Server`](https://github.com/apple/foundationdb/tree/main/fdbserver/MockS3Server.actor.cpp#L43) only provides deterministic "happy path" responses. Real S3 services exhibit various error conditions that clients must handle gracefully. +FoundationDB's S3BlobStore client needs thorough testing against realistic S3 failure scenarios, but the existing [`MockS3Server`](https://github.com/apple/foundationdb/blob/main/fdbserver/mocks3/MockS3Server.cpp) only provides deterministic "happy path" responses. Real S3 services exhibit various error conditions that clients must handle gracefully. ## Design @@ -108,14 +108,14 @@ MockS3ServerChaos Configuration: ## Usage -Replace [`startMockS3Server()`](https://github.com/apple/foundationdb/tree/main/fdbserver/include/fdbserver/MockS3Server.h#L47) calls with `startMockS3ServerChaos()` in simulation tests: +Replace [`startMockS3Server()`](https://github.com/apple/foundationdb/blob/main/fdbserver/mocks3/include/fdbserver/mocks3/MockS3Server.h) calls with `startMockS3ServerChaos()` in simulation tests: ``` // Before: -wait(startMockS3Server(listenAddress)); +co_await startMockS3Server(listenAddress); // After: -wait(startMockS3ServerChaos(listenAddress)); +co_await startMockS3ServerChaos(listenAddress); ``` Chaos behavior is controlled by **S3FaultInjector rates** (0.0-1.0), with **BUGGIFY providing occasional extra chaos** - no master boolean switch. diff --git a/design/recovery-internals.md b/design/recovery-internals.md index 9f77bc60e15..1a620904124 100644 --- a/design/recovery-internals.md +++ b/design/recovery-internals.md @@ -52,7 +52,7 @@ Cluster controller (CC) decides if recovery should be triggered. In case the cur Recovery has 9 phases, which are defined as the 9 states in the source code: `READING_CSTATE = 1, LOCKING_CSTATE = 2, RECRUITING = 3, RECOVERY_TRANSACTION = 4, WRITING_CSTATE = 5, ACCEPTING_COMMITS = 6, ALL_LOGS_RECRUITED = 7, STORAGE_RECOVERED = 8, FULLY_RECOVERED = 9`. -The recovery process is like a state machine, changing from one state to the next state. `ClusterRecovery.actor` implements the cluster recovery. We will describe in the rest of this document what each phase does to drive the recovery to the next state. +The recovery process is like a state machine, changing from one state to the next state. `ClusterRecovery.cpp` implements the cluster recovery. We will describe in the rest of this document what each phase does to drive the recovery to the next state. The system tracks the information of each recovery phase via trace events. In past recovery state machine was driven by the `Master` process, hence, the events used `Master` as event prefix name (for instance: `MasterRecoveryState`). Given the recovery state machine is currently driven by CC, updating name to `ClusterRecoveryState` would have been more appropriate, however, it breaks existing tooling scripts. For now, `ServerKnob::CLUSTER_RECOVERY_EVENT_NAME_PREFIX` determines the event name prefix, default prefix value is `Master`. Recovery tracks the information of each recovery phase in `RecoveryState` trace event. By checking the message, we can find which phase the recovery is stuck at. The status used in the `RecoveryState` trace event is defined as `RecoveryStatus` structure in `RecoveryState.h`. The status, instead of the name of the 9 phases, is typically used in diagnosing production issues. @@ -117,7 +117,7 @@ Consider an old generation with three TLogs: `A, B, C`. Their durable versions a * Situation 1: Too many tLogs in the previous generation permanently died, say due to hardware failure. If force recovery is allowed by system administrator, the CC can choose to force recovery, which can cause data loss; otherwise, to unblock the recovery, system administrator has to bring up those died tLogs, for example by copying their files onto new hardware. -* Situation 2: A tLog may die after it reports alive to the CC in the RECRUITING phase. This may cause the `RecoveryVersion` calculated by the CC in this phase to no longer be valid in the next phases (see `getDurableVersion()` in [LogSystem.cpp](https://github.com/apple/foundationdb/blob/main/fdbserver/logsystem/LogSystem.cpp)). When this happens, the CC will detect it (that the previous computed recovery version is no longer viable, because the tail of versions are no longer retrievable from current set of live TLogs), terminate the current recovery, and start a new recovery. Changes to TLogs can cause changes to old log system (`outLogSystem->set(logSystem);`), which in turn causes the recovery loop in [ClusterRecovery.actor.cpp](https://github.com/apple/foundationdb/blob/main/fdbserver/clustercontroller/ClusterRecovery.actor.cpp) to abandon progress and restart. +* Situation 2: A tLog may die after it reports alive to the CC in the RECRUITING phase. This may cause the `RecoveryVersion` calculated by the CC in this phase to no longer be valid in the next phases (see `getDurableVersion()` in [LogSystem.cpp](https://github.com/apple/foundationdb/blob/main/fdbserver/logsystem/LogSystem.cpp)). When this happens, the CC will detect it (that the previous computed recovery version is no longer viable, because the tail of versions are no longer retrievable from current set of live TLogs), terminate the current recovery, and start a new recovery. Changes to TLogs can cause changes to old log system (`outLogSystem->set(logSystem);`), which in turn causes the recovery loop in [ClusterRecovery.cpp](https://github.com/apple/foundationdb/blob/main/fdbserver/clustercontroller/ClusterRecovery.cpp) to abandon progress and restart. Once we have a `knownCommittedVersion`, the CC will reconstruct the [transaction state store](https://github.com/apple/foundationdb/blob/main/design/transaction-state-store.md) by peeking the txnStateTag in oldLogSystem. Recall that the txnStateStore includes the transaction system’s configuration, such as the assignment of shards to SS and to tLogs and that the txnStateStore was durable on disk in the oldLogSystem. diff --git a/design/special-key-space.md b/design/special-key-space.md index be104915fee..78c8c71610b 100644 --- a/design/special-key-space.md +++ b/design/special-key-space.md @@ -57,21 +57,21 @@ private: std::map CountryToCapitalCity; }; // Instantiate the function object -// In development, you should have a function object pointer in DatabaseContext(DatabaseContext.h) and initialize in DatabaseContext's constructor(NativeAPI.actor.cpp) +// In development, you should have a function object pointer in DatabaseContext(DatabaseContext.h) and initialize in DatabaseContext's constructor(DatabaseContext.cpp) const KeyRangeRef exampleRange("\xff\xff/example/"_sr, "\xff\xff/example/\xff"_sr); SKRExampleImpl exampleImpl(exampleRange); // Assuming the database handler is `cx`, register to special-key-space -// In development, you should register all function objects in the constructor of DatabaseContext(NativeAPI.actor.cpp) +// In development, you should register all function objects in the constructor of DatabaseContext(DatabaseContext.cpp) cx->specialKeySpace->registerKeyRange(exampleRange, &exampleImpl); // Now any ReadYourWritesTransaction associated with `cx` is able to query the info -state ReadYourWritesTransaction tr(cx); +ReadYourWritesTransaction tr(cx); // get -Optional res1 = wait(tr.get("\xff\xff/example/Japan")); -ASSERT(res1.present() && res.getValue() == "Tokyo"_sr); +Optional res1 = co_await tr.get("\xff\xff/example/Japan"); +ASSERT(res1.present() && res1.get() == "Tokyo"_sr); // getRange // Note: for getRange(key1, key2), both key1 and key2 should prefixed with \xff\xff // something like getRange("normal_key", "\xff\xff/...") is not supported yet -RangeResult res2 = wait(tr.getRange("\xff\xff/example/U"_sr, "\xff\xff/example/U\xff"_sr)); +RangeResult res2 = co_await tr.getRange("\xff\xff/example/U"_sr, "\xff\xff/example/U\xff"_sr); // res2 should contain USA and UK ASSERT( res2.size() == 2 && diff --git a/documentation/CMakeLists.txt b/documentation/CMakeLists.txt index 34f5951a77a..d42a95abb03 100644 --- a/documentation/CMakeLists.txt +++ b/documentation/CMakeLists.txt @@ -1,4 +1,3 @@ -add_subdirectory(tutorial) add_subdirectory(coro_tutorial) set(SPHINX_DOCUMENT_DIR "${CMAKE_SOURCE_DIR}/documentation/sphinx") diff --git a/documentation/coro_tutorial/tutorial.cpp b/documentation/coro_tutorial/tutorial.cpp index 4003ff16504..86c0c622f93 100644 --- a/documentation/coro_tutorial/tutorial.cpp +++ b/documentation/coro_tutorial/tutorial.cpp @@ -55,10 +55,8 @@ Future simpleTimer() { } } -// A actor that demonstrates how choose-when -// blocks work. +// A coroutine that demonstrates choosing between futures. Future someFuture(Future ready) { - // loop choose {} works as well here - the braces are optional while (true) { co_await Choose() .When(delay(0.5), [](Void const&) { std::cout << "Still waiting...\n"; }) diff --git a/documentation/sphinx/source/architecture.rst b/documentation/sphinx/source/architecture.rst index f6938654300..44655eeef0b 100644 --- a/documentation/sphinx/source/architecture.rst +++ b/documentation/sphinx/source/architecture.rst @@ -2,7 +2,7 @@ Architecture ############ -FoundationDB makes your architecture flexible and easy to operate. Your applications can send their data directly to the FoundationDB or to a :doc:`layer`, a user-written module that can provide a new data model, compatibility with existing systems, or even serve as an entire framework. In both cases, all data is stored in a single place via an ordered, transactional key-value API. +FoundationDB makes your architecture flexible and easy to operate. Your applications can send their data directly to FoundationDB or to a :doc:`layer`, a user-written module that can provide a new data model, compatibility with existing systems, or even serve as an entire framework. In both cases, all data is stored in a single place via an ordered, transactional key-value API. The following diagram details the logical architecture. @@ -14,9 +14,9 @@ Detailed FoundationDB Architecture The FoundationDB architecture chooses a decoupled design, where processes are assigned different heterogeneous roles (e.g., -Coordinators, Storage Servers, Master). Cluster attempts to recruit +Coordinators, Storage Servers, Master). The cluster attempts to recruit different roles as separate processes, however, it is possible that -multiple Stateless roles gets colocated (recruited) on a single +multiple stateless roles get colocated (recruited) on a single process to meet the cluster recruitment goals. Scaling the database is achieved by horizontally expanding the number of processes for separate roles: @@ -52,18 +52,18 @@ them fail, we will recruit a replacement for all three roles. The master provides the commit versions for batches of the mutations to the commit proxies. -Historically, Ratekeeper and Data Distributor are coupled with Master on -the same process. Since 6.2, both have become a singleton in the -cluster. The life time is no longer tied with Master. +Historically, Ratekeeper and Data Distributor were coupled with the Master on +the same process. Since 6.2, both have become singletons in the +cluster. Their lifetimes are no longer tied to the Master. |image1| GRV Proxies ~~~~~~~~~~~ -The GRV proxies are responsible for providing read versions, communicating -with ratekeeper to control the rate providing read versions. To provide a -read version, a GRV proxy will ask all master to see the largest committed +The GRV proxies are responsible for providing read versions and communicating +with Ratekeeper to control the rate of providing read versions. To provide a +read version, a GRV proxy will ask the master for the largest committed version at this point in time, while simultaneously checking that the transaction logs have not been stopped. Ratekeeper will artificially slow down the rate at which the GRV proxy provides read versions. @@ -71,9 +71,9 @@ down the rate at which the GRV proxy provides read versions. Commit Proxies ~~~~~~~~~~~~~~ -The proxies are responsible for committing transactions, report committed -versions to master and tracking the storage servers responsible for each -range of keys. +The proxies are responsible for committing transactions, reporting committed +versions to the master, and tracking the storage servers responsible for each +range of keys. Commits are accomplished by: @@ -86,7 +86,7 @@ The key space starting with the ``\xff`` byte is reserved for system metadata. All mutations committed into this key space are distributed to all of the commit proxies through the resolvers. This metadata includes a mapping between key ranges and the storage servers which have the data -for that range of keys. The commit proxies provides this information to +for that range of keys. The commit proxies provide this information to clients on-demand. The clients cache this mapping; if they ask a storage server for a key it does not have, they will clear their cache and get a more up-to-date list of servers from the commit proxies. @@ -95,9 +95,9 @@ Transaction Logs ~~~~~~~~~~~~~~~~ The transaction logs make mutations durable to disk for fast commit -latencies. The logs receive commits from the commit proxy in version order, +latencies. The logs receive commits from the commit proxy in version order, and only respond to the commit proxy once the data has been written and fsync’ed -to an append only mutation log on disk. Before the data is even written to +to an append-only mutation log on disk. Before the data is even written to disk we forward it to the storage servers responsible for that mutation. Once the storage servers have made the mutation durable, they pop it from the log. This generally happens roughly 6 seconds after the @@ -111,7 +111,7 @@ the log data bound for the failed server. Resolvers ~~~~~~~~~ -The resolvers are responsible determining conflicts between +The resolvers are responsible for determining conflicts between transactions. A transaction conflicts if it reads a key that has been written between the transaction’s read version and commit version. The resolver does this by holding the last 5 seconds of committed writes in @@ -122,13 +122,13 @@ Storage Servers ~~~~~~~~~~~~~~~ The vast majority of processes in a cluster are storage servers. Storage -servers are assigned ranges of key, and are responsible to storing all +servers are assigned ranges of keys and are responsible for storing all of the data for that range. They keep 5 seconds of mutations in memory, -and an on disk copy of the data as of 5 second ago. Clients must read at +and an on-disk copy of the data as of 5 seconds ago. Clients must read at a version within the last 5 seconds, or they will get a ``transaction_too_old`` error. The SSD storage engine stores the data in -a B-tree based on SQLite. The memory storage engine store the data in -memory with an append only log that is only read from disk if the +a B-tree based on SQLite. The memory storage engine stores the data in +memory with an append-only log that is only read from disk if the process is rebooted. In the upcoming FoundationDB 7.0 release, the B-tree storage engine will be replaced with a brand new *Redwood* engine. @@ -136,10 +136,10 @@ engine. Data Distributor ~~~~~~~~~~~~~~~~ -Data distributor manages the lifetime of storage servers, decides which +The data distributor manages the lifetime of storage servers, decides which storage server is responsible for which data range, and ensures data is -evenly distributed across all storage servers (SS). Data distributor as -a singleton in the cluster is recruited and monitored by Cluster +evenly distributed across all storage servers (SS). The data distributor, as +a singleton in the cluster, is recruited and monitored by the Cluster Controller. See `internal documentation `__. @@ -148,8 +148,8 @@ Ratekeeper Ratekeeper monitors system load and slows down client transaction rate when the cluster is close to saturation by lowering the rate at which -the proxy provides read versions. Ratekeeper as a singleton in the -cluster is recruited and monitored by Cluster Controller. +the proxy provides read versions. Ratekeeper, as a singleton in the +cluster, is recruited and monitored by the Cluster Controller. Clients ~~~~~~~ @@ -157,15 +157,15 @@ Clients A client links with specific language bindings (i.e., client libraries) in order to communicate with a FoundationDB cluster. The language bindings support loading multiple versions of C libraries, allowing the -client communicates with older version of the FoundationDB clusters. -Currently, C, Go, Python, Java, Ruby bindings are officially supported. +client to communicate with FoundationDB clusters running older versions. +Currently, C, Go, Python, Java, and Ruby bindings are officially supported. Transaction Processing ---------------------- A database transaction in FoundationDB starts by a client contacting one of the GRV proxies to obtain a read version, which is guaranteed to be -larger than any of commit version that client may know about (even +larger than any commit version that the client may know about (even through side channels outside the FoundationDB cluster). This is needed so that a client will see the result of previous commits that have happened. @@ -176,12 +176,12 @@ memory without contacting the cluster. By default, reading a key that was written in the same transaction will return the newly written value. At commit time, the client sends the transaction data (all reads and -writes) to one of the commit proxies and waits for commit or abort response +writes) to one of the commit proxies and waits for a commit or abort response from the commit proxy. If the transaction conflicts with another one and cannot commit, the client may choose to retry the transaction from the -beginning again. If the transaction commits, the commit proxy also returns -the commit version back to the client and to master so that GRV proxies can -get access to the latest committed version. Note this commit version is +beginning. If the transaction commits, the commit proxy also returns +the commit version to the client and to the master so that GRV proxies can +access the latest committed version. Note that this commit version is larger than the read version and is chosen by the master. The FoundationDB architecture separates the scaling of client reads and @@ -193,33 +193,33 @@ to Commit Proxies, Resolvers, and Log Servers in the transaction system. Determine Read Version ~~~~~~~~~~~~~~~~~~~~~~ -When a client requests a read version from a GRV proxy, the GRV proxy asks -master for the latest committed version, and checks a set of transaction -logs satisfying replication policy are live. Then the GRV proxy returns +When a client requests a read version from a GRV proxy, the GRV proxy asks +the master for the latest committed version and checks that a set of transaction +logs satisfying the replication policy is live. Then the GRV proxy returns the maximum committed version as the read version to the client. |image2| -The reason for the GRV proxy to contact master for the latest committed -versions is to because master is a central place to keep the largest of -all commit proxies' committed version. +The GRV proxy contacts the master for the latest committed versions because +the master is a central place for keeping the largest committed version +reported by all commit proxies. -The reason for checking a set of transaction logs satisfying replication -policy are live is to ensure the GRV proxy is not replaced with newer +Checking that a set of transaction logs satisfying the replication +policy is live ensures that the GRV proxy has not been replaced by a newer generation of GRV proxies. This is because GRV proxy is a stateless role recruited in each generation. If a recovery has happened and the old GRV proxy is still live, this old GRV proxy could still give out read versions. As a result, a *read-only* transaction may see stale results (a read-write transaction will be aborted). By checking a set of -transaction logs satisfying replication policy are live, the GRV proxy makes +transaction logs satisfying the replication policy is live, the GRV proxy makes sure no recovery has happened, thus the *read-only* transaction sees the latest data. Note that the client cannot simply ask the master for read versions because -this approach is putting more work towards the master, because the master -role can’t be scaled. Even though giving out read-versions isn’t very -expensive, it still requires the master to get a transaction budget from the -Ratekeeper, batches requests, and potentially maintains thousands of network +this approach puts more work on the master, whose role cannot be scaled. +Even though providing read versions is not very expensive, it still requires +the master to get a transaction budget from Ratekeeper, batch requests, and +potentially maintain thousands of network connections from clients. |image3| @@ -235,28 +235,28 @@ A client transaction commits in the following steps: version seen before. 4. The commit proxy sends the read and write conflict ranges to the resolver(s) with the commit version included. -5. The resolver responds back with whether the transaction has any +5. The resolver responds with whether the transaction has any conflicts with previous transactions by sorting transactions according to their commit versions and computing if such a serial execution order is conflict-free. - - If there are conflicts, the commit proxy responds back to the client with + - If there are conflicts, the commit proxy responds to the client with a not_committed error. - If there are no conflicts, the commit proxy sends the mutations and commit version of this transaction to the transaction logs. -6. Once the mutations are durable on the logs, the commit proxy responds back - success to the user. +6. Once the mutations are durable on the logs, the commit proxy responds + with success to the user. -Note the commit proxy sends each resolver their respective key ranges, if -any one of the resolvers detects a conflict then the transaction is not +Note that the commit proxy sends each resolver its respective key ranges. If +any one of the resolvers detects a conflict, then the transaction is not committed. This has the flaw that if only one of the resolvers detects a conflict, the other resolver will still think the transaction has succeeded and may fail future transactions with overlapping write -conflict ranges, even though these future transaction can commit. In -practice, a well designed workload will only have a very small +conflict ranges, even though these future transactions can commit. In +practice, a well-designed workload will only have a very small percentage of conflicts, so this amplification will not affect -performance. Additionally, each transaction has a five seconds window. +performance. Additionally, each transaction has a five-second window. After five seconds, resolvers will remove the conflict ranges of old transactions, which also limits the chance of this type of false conflict. @@ -268,22 +268,22 @@ conflict. Background Work ~~~~~~~~~~~~~~~ -There are a number of background work happening besides the transaction +There are a number of background tasks happening in addition to transaction processing: -- **Ratekeeper** collects statistic information from GRV proxies, Commit - proxies, transaction logs, and storage servers and compute the target +- **Ratekeeper** collects statistical information from GRV proxies, Commit + proxies, transaction logs, and storage servers and computes the target transaction rate for the cluster. -- **Data distribution** monitors all storage servers and perform load +- **Data distribution** monitors all storage servers and performs load balancing operations to evenly distribute data among all storage servers. -- **Storage servers** pull mutations from transaction logs, write them - into storage engine to persist on disks. +- **Storage servers** pull mutations from transaction logs and write them + into the storage engine to persist them on disk. - **Commit proxies** periodically send empty commits to transaction logs to - keep commit versions increasing, in case there is no client generated + keep commit versions increasing, in case there are no client-generated transactions. |image6| @@ -294,12 +294,12 @@ Transaction System Recovery The transaction system implements the write pipeline of the FoundationDB cluster and its performance is critical to the transaction commit latency. A typical recovery takes about a few hundred milliseconds, but -longer recovery time (usually a few seconds) can happen. Whenever there +longer recoveries (usually a few seconds) can happen. Whenever there is a failure in the transaction system, a recovery process is performed to restore the transaction system to a new configuration, i.e., a clean state. Specifically, the Master process monitors the health of GRV Proxies, -Commit Proxies, Resolvers, and Transaction Logs. If any one of the monitored -process failed, the Master process terminates. The Cluster Controller will +Commit Proxies, Resolvers, and Transaction Logs. If any one of the monitored +processes fails, the Master process terminates. The Cluster Controller will detect this event, and then recruits a new Master, which coordinates the recovery and recruits a new transaction system instance. In this way, the transaction processing is divided into a number of epochs, where @@ -308,23 +308,23 @@ unique Master process. For each epoch, the Master initiates recovery in several steps. First, the Master reads the previous transaction system states from -Coordinators and lock the coordinated states to prevent another Master +Coordinators and locks the coordinated states to prevent another Master process from recovering at the same time. Then the Master recovers previous transaction system states, including all Log Servers’ -Information, stops these Log Servers from accepting transactions, and -recruits a new set of GRV Proxies, Commit Proxies, Resolvers, and -Transaction Logs. After previous Log Servers are stopped and new transaction -system is recruited, the Master writes the coordinated states with current +information, stops these Log Servers from accepting transactions, and +recruits a new set of GRV Proxies, Commit Proxies, Resolvers, and +Transaction Logs. After the previous Log Servers are stopped and a new transaction +system is recruited, the Master writes the coordinated states with the current transaction system information. Finally, the Master accepts new transaction commits. See details in this `documentation `__. -Because GRV Proxies, Commit Proxies and Resolvers are stateless, their -recoveries have no extra work. In contrast, Transaction Logs save the -logs of committed transactions, and we need to ensure all previously -committed transactions are durable and retrievable by storage servers. +Because GRV Proxies, Commit Proxies, and Resolvers are stateless, their +recoveries require no extra work. In contrast, Transaction Logs save the +logs of committed transactions, and we need to ensure all previously +committed transactions are durable and retrievable by storage servers. That is, for any transactions that the Commit Proxies may have sent back -commit response, their logs are persisted in multiple Log Servers (e.g., +commit responses, their logs are persisted in multiple Log Servers (e.g., three servers if replication degree is 3). Finally, a recovery will *fast forward* time by 90 seconds, which would @@ -332,14 +332,14 @@ abort any in-progress client transactions with ``transaction_too_old`` error. During retry, these client transactions will find the new generation of transaction system and commit. -**``commit_result_unknown`` error:** If a recovery happened while a -transaction is committing (i.e., a commit proxy has sent mutations to -transaction logs). A client would have received -``commit_result_unknown``, and then retried the transaction. It’s -completely permissible for FDB to commit both the first attempt, and the +**``commit_result_unknown`` error:** If a recovery happens while a +transaction is committing (i.e., after a commit proxy has sent mutations to +transaction logs), a client may receive +``commit_result_unknown`` and then retry the transaction. It’s +completely permissible for FDB to commit both the first attempt and the second retry, as ``commit_result_unknown`` means the transaction may or may not have committed. This is why it’s strongly recommended that -transactions should be idempotent, so that they handle +transactions be idempotent, so that they handle ``commit_result_unknown`` correctly. Resources @@ -367,4 +367,3 @@ Documentation ` that we developed specifically for these use cases. +This is 50x faster than our first simple loop! We have achieved this without sacrificing any safety, and still only using a single core. This shows how efficiently FoundationDB handles real-world concurrent workloads—in this case 100 parallel clients. A lot of this can be attributed to the :doc:`Flow asynchronous runtime ` that we developed specifically for these use cases. Single-core Read Test ===================== diff --git a/documentation/sphinx/source/clang-format.rst b/documentation/sphinx/source/clang-format.rst index 52ecbf41ce4..c88e4902de6 100644 --- a/documentation/sphinx/source/clang-format.rst +++ b/documentation/sphinx/source/clang-format.rst @@ -35,7 +35,7 @@ If ``git clang-format`` complains about unstaged changes, or you want to format clang-format -i path/to/file.cpp # Format multiple files - clang-format -i fdbserver/DataDistribution.actor.cpp fdbclient/SystemData.cpp + clang-format -i fdbserver/datadistributor/DataDistribution.cpp fdbclient/SystemData.cpp .. note:: diff --git a/documentation/sphinx/source/clang-tidy.rst b/documentation/sphinx/source/clang-tidy.rst index 3dd4d21754c..28d11988756 100644 --- a/documentation/sphinx/source/clang-tidy.rst +++ b/documentation/sphinx/source/clang-tidy.rst @@ -10,15 +10,15 @@ This guide explains how to run ``clang-tidy`` locally so you can fix issues befo What clang-tidy checks ====================== -FoundationDB configures 40 named checks in the ``.clang-tidy`` file at the repository root. The +FoundationDB configures 43 named checks in the ``.clang-tidy`` file at the repository root. The active set depends on the clang-tidy version and can be inspected with ``clang-tidy --list-checks``. The intent is to enable more as we go forward. Here are some example rules: -* **22 Bugprone rules** -- catch potential runtime errors, including dangling returned references, incorrect forwarding, bytewise operations on non-trivially-copyable objects, incorrect erase/remove calls, and arithmetic widened after the calculation +* **24 Bugprone rules** -- catch potential runtime errors, including assignments in conditions, truncated string literals with embedded NUL bytes, dangling returned references, incorrect forwarding, bytewise operations on non-trivially-copyable objects, incorrect erase/remove calls, and arithmetic widened after the calculation * **1 C++ Core Guidelines rule** -- catch unsafe captures in coroutine lambdas (``cppcoreguidelines-avoid-capturing-lambda-coroutines``) * **2 Misc rules** -- catch redundant expressions and RAII objects held across coroutine suspension points * **4 Modernize rules** -- encourage modern C++ practices (e.g., ``modernize-use-auto``, ``modernize-use-override``) -* **4 Performance rules** -- avoid unnecessary copies, hidden range-loop conversions, pointless moves, and move constructors that copy movable members (``performance-for-range-copy``, ``performance-implicit-conversion-in-loop``, ``performance-move-const-arg``, ``performance-move-constructor-init``) +* **5 Performance rules** -- avoid unnecessary copies, hidden range-loop conversions, repeated vector growth in simple loops, pointless moves, and move constructors that copy movable members (``performance-for-range-copy``, ``performance-implicit-conversion-in-loop``, ``performance-inefficient-vector-operation``, ``performance-move-const-arg``, ``performance-move-constructor-init``) * **7 Readability rules** -- improve code clarity (e.g., ``readability-container-contains``, ``readability-container-size-empty``) ``misc-coroutine-hostile-raii`` checks Flow's blocking ``MutexHolder`` and @@ -91,7 +91,7 @@ Then symlink it to your source root: .. note:: - A full build is recommended so that generated files (e.g., ``.actor.g.cpp`` headers) exist and include paths resolve correctly. Without a build, ``clang-tidy`` may report false errors on files that depend on generated code. If your ``compile_commands.json`` was generated on a different machine (e.g., Okteto), fix the paths with: + A full build is recommended so that generated headers exist and include paths resolve correctly. Without a build, ``clang-tidy`` may report false errors on files that depend on generated code. If your ``compile_commands.json`` was generated on a different machine (e.g., Okteto), fix the paths with: .. code-block:: shell @@ -127,10 +127,10 @@ Check all changes between your branch and ``main``: .. code-block:: shell - git diff -U0 origin/main...HEAD | grep -v -E '\.actor\.cpp' | python3 "$TIDY_DIFF" -p 1 -path . + git diff -U0 origin/main...HEAD | python3 "$TIDY_DIFF" -p 1 -path . # Or with the alias: - git diff -U0 origin/main...HEAD | grep -v -E '\.actor\.cpp' | fdb-tidy + git diff -U0 origin/main...HEAD | fdb-tidy Check a specific commit: @@ -176,9 +176,7 @@ Optional CMake variables: Known limitations ----------------- -**``.actor.cpp`` files cannot be analyzed.** These files use FoundationDB's custom actor compiler syntax (``ACTOR``, ``wait()``, ``state``) that ``clang-tidy`` cannot parse. With CMake 3.27 or newer, build-integrated clang-tidy skips generated ``.actor.g.cpp`` outputs using CMake's ``SKIP_LINTING`` property. This automatic suppression is unavailable with the minimum supported CMake version, 3.24.2. Exclude actor inputs from your diff when running locally. - -Build-integrated clang-tidy also skips bundled external-library targets, including ``crc32``, ``libb64``, ``md5``, ``libeio``, and ``libcoroutine``. +Build-integrated clang-tidy skips bundled external-library targets, including ``crc32``, ``libb64``, ``md5``, ``libeio``, and ``libcoroutine``. ``readability-simplify-boolean-expr`` is not enabled because it diagnoses ordinary uses of FoundationDB's ``ASSERT`` macro after macro expansion. @@ -188,7 +186,7 @@ Quick reference .. code-block:: shell - git diff -U0 origin/main...HEAD | grep -v -E '\.actor\.cpp' | python3 "$TIDY_DIFF" -p 1 -path . + git diff -U0 origin/main...HEAD | python3 "$TIDY_DIFF" -p 1 -path . **GCC-built compile_commands.json with clang-tidy.** If your ``compile_commands.json`` was generated with GCC, it may contain GCC-specific flags that ``clang-tidy`` (which uses the clang frontend) does not recognize. Add ``-extra-arg=-Wno-unknown-warning-option`` to suppress these errors: diff --git a/documentation/sphinx/source/dtrace-probes.rst b/documentation/sphinx/source/dtrace-probes.rst index 7786176925c..7a0450991af 100644 --- a/documentation/sphinx/source/dtrace-probes.rst +++ b/documentation/sphinx/source/dtrace-probes.rst @@ -13,26 +13,28 @@ Probes ====== -Actors ------- +Legacy actor probes +------------------- + +These probes are emitted by the legacy source translator. The C++ coroutine runtime does +not emit them. .. code-block:: c - FDB_TRACE_PROBE(actor_create, "actorname") - FDB_TRACE_PROBE(actor_destroy, "actorname") + FDB_TRACE_PROBE(actor_create, "actorname", id) + FDB_TRACE_PROBE(actor_destroy, "actorname", id) -Gets called whenever an actor is created or gets destroyed. It provides one argument which is a -string and it is the name of the actor. +These record creation and destruction of a generated actor. Their arguments are the actor's +name and an ``unsigned long`` instance identifier. .. code-block:: c - FDB_TRACE_PROBE(actor_enter, "name", index) - FDB_TRACE_PROBE(actor_exit, "name", index) + FDB_TRACE_PROBE(actor_enter, "name", id, index) + FDB_TRACE_PROBE(actor_exit, "name", id, index) -Whenever we call into an actor (either directly through a function call or indirectly through a callback) -we call ``actor_enter``. Whenever we leave an actor (either because it returns or because it calls into -wait) we call ``actor_exit``. The first argument is a string of the name of the actor and the second is an -index. ``-1`` means that we entered/exited through in a main function call, otherwise it is a generated index. +These record entry to and exit from generated actor code. The arguments are the actor's name, +instance identifier, and an integer index. ``-1`` identifies the initial function invocation; +other indices identify generated callbacks. Main-Loop --------- diff --git a/documentation/sphinx/source/engineering.rst b/documentation/sphinx/source/engineering.rst index 3b1101949ab..8c1ecdb297d 100644 --- a/documentation/sphinx/source/engineering.rst +++ b/documentation/sphinx/source/engineering.rst @@ -7,12 +7,12 @@ When we built FoundationDB, we didn't just want it to make something that rivale Flow ==== -FoundationDB began with ambitious goals for both :doc:`high performance ` per node and :doc:`scalability `. We knew that to achieve these goals we would face serious engineering challenges that would require tool breakthroughs. We'd need efficient asynchronous communicating processes like in Erlang or the Async in .NET, but we'd also need the raw speed, I/O efficiency, and control of C++. To meet these challenges, we developed several new tools, the most important of which is :doc:`flow`, a new programming language that brings actor-based concurrency to C++11. Flow adds about 10 keywords to C++11 and is technically a trans-compiler: the Flow compiler reads Flow code and compiles it down to raw C++11, which is then compiled to a native binary with a traditional toolchain. One of Flow’s most important job is enabling Simulation. +FoundationDB began with ambitious goals for both :doc:`high performance ` per node and :doc:`scalability `. We knew that to achieve these goals we would face serious engineering challenges that would require tool breakthroughs. We'd need efficient asynchronous communicating processes like in Erlang or the Async in .NET, but we'd also need the raw speed, I/O efficiency, and control of C++. To meet these challenges, we developed :doc:`flow`, the asynchronous runtime used throughout FoundationDB. Flow combines standard C++ coroutines with futures, message streams, and cooperative scheduling. The same runtime supports efficient production execution and deterministic simulation. Simulation ========== -We wanted FoundationDB to survive failures of machines, networks, disks, clocks, racks, data centers, file systems, etc., so we created a simulation framework closely tied to Flow. By replacing physical interfaces with shims, replacing the main epoll-based run loop with a time-based simulation, and running multiple logical processes as concurrent Flow Actors, Simulation is able to conduct a deterministic simulation of an entire FoundationDB cluster within a single-thread! Even better, we are able to execute this simulation in a deterministic way, enabling us to reproduce problems and add instrumentation ex post facto. This incredible capability enabled us to build FoundationDB exclusively in simulation for the first 18 months and ensure exceptional fault tolerance long before it sent its first real network packet. For a database with as strong a contract as the FoundationDB, testing is crucial, and over the years we have run the equivalent of *a trillion CPU-hours* of simulated stress testing. Read more about our :doc:`Simulation and Testing `. +We wanted FoundationDB to survive failures of machines, networks, disks, clocks, racks, data centers, file systems, etc., so we created a simulation framework closely tied to Flow. By replacing physical interfaces with shims, replacing the main epoll-based run loop with a time-based simulation, and running multiple logical processes as concurrent Flow coroutines, Simulation is able to conduct a deterministic simulation of an entire FoundationDB cluster within a single-thread! Even better, we are able to execute this simulation in a deterministic way, enabling us to reproduce problems and add instrumentation ex post facto. This incredible capability enabled us to build FoundationDB exclusively in simulation for the first 18 months and ensure exceptional fault tolerance long before it sent its first real network packet. For a database with as strong a contract as the FoundationDB, testing is crucial, and over the years we have run the equivalent of *a trillion CPU-hours* of simulated stress testing. Read more about our :doc:`Simulation and Testing `. Ratekeeper ========== diff --git a/documentation/sphinx/source/flow.rst b/documentation/sphinx/source/flow.rst index da6cf1fd4aa..8fb2718971c 100644 --- a/documentation/sphinx/source/flow.rst +++ b/documentation/sphinx/source/flow.rst @@ -5,154 +5,121 @@ Flow Engineering challenges ====================== -FoundationDB began with ambitious goals for both :doc:`high performance ` per node and :doc:`scalability `. We knew that to achieve these goals we would face serious engineering challenges while developing the FoundationDB core. We'd need to implement efficient asynchronous communicating processes of the sort supported by `Erlang `_ or the `Async library in .NET `_, but we'd also need the raw speed and I/O efficiency of C++. Finally, we'd need to perform extensive simulation to engineer for reliability and fault tolerance on large clusters. +FoundationDB began with ambitious goals for both :doc:`high performance ` per node and :doc:`scalability `. We knew that to achieve these goals we would face serious engineering challenges while developing the FoundationDB core. We'd need efficient asynchronous communicating processes, the speed and I/O efficiency of C++, and extensive simulation to engineer for reliability and fault tolerance on large clusters. -To meet these challenges, we developed several new tools, the first of which is Flow, a new programming language that brings `actor-based concurrency `_ to C++11. To add this capability, Flow introduces a number of new keywords and control-flow primitives for managing concurrency. Flow is implemented as a compiler which analyzes an asynchronous function (actor) and rewrites it as an object with many different sub-functions that use callbacks to avoid blocking (see `streamlinejs `_ for a similar concept using JavaScript). The Flow compiler's output is normal C++11 code, which is then compiled to a binary using traditional tools. Flow also provides input to our simulation tool, which conducts deterministic simulations of the entire system, including its physical interfaces and failure modes. In short, Flow allows efficient concurrency within C++ in a maintainable and extensible manner, achieving all three major engineering goals: +Flow provides the asynchronous runtime for these processes. FoundationDB uses standard C++ coroutines with Flow futures, streams, and cooperative scheduling. The C++ compiler preserves each coroutine's execution state across suspension points, and Flow connects awaited operations to the event loop. The same runtime supports deterministic simulation of the system, including its physical interfaces and failure modes. -* high performance (by compiling to native code), -* actor-based concurrency (for high productivity development), -* simulation support (for testing). +Flow supports three major engineering goals: + +* high performance through native code, +* asynchronous concurrency through coroutines and message passing, +* simulation support for testing. A first look ============ -Actors in Flow receive asynchronous messages from each other using a data type called a *future*. When an actor requires a data value to continue computation, it waits for it without blocking other actors. The following simple actor performs asynchronous addition. It takes a future integer and a normal integer as an offset, waits on the future integer, and returns the sum of the value and the offset: +Coroutines receive asynchronous results through futures. When a coroutine needs a result, it can suspend without blocking other work on the event loop. This function waits for an integer, adds an offset, and returns the sum: + +.. code-block:: cpp -.. code-block:: c + #include "flow/flow.h" - ACTOR Future asyncAdd(Future f, int offset) { - int value = wait( f ); - return value + offset; + Future asyncAdd(Future f, int offset) { + int value = co_await f; + co_return value + offset; } +A coroutine returning ``Future`` starts immediately when called and runs until it completes or awaits an operation that is not ready. Its returned future represents the eventual result. Coroutine code lives in ordinary ``.cpp`` and ``.h`` files. + Flow features ============= -Flow's new keywords and control-flow primitives support the capability to pass messages asynchronously between components. Here's a brief overview. - Promise and Future ------------------------ -The data types that connect asynchronous senders and receivers are ``Promise`` and ``Future`` for some C++ type ``T``. When a sender holds a ``Promise``, it represents a promise to deliver a value of type ``T`` at some point in the future to the holder of the ``Future``. Conversely, a receiver holding a ``Future`` can asynchronously continue computation until the point at which it actually needs the ``T.`` - -Promises and futures can be used within a single process, but their real strength in a distributed system is that they can traverse the network. For example, one computer could create a promise/future pair, then send the promise to another computer over the network. The promise and future will still be connected, and when the promise is fulfilled by the remote computer, the original holder of the future will see the value appear. +``Promise`` and ``Future`` connect an asynchronous sender and receiver. A promise can deliver one value of type ``T`` or an error. The future lets its holder observe that result. A future may have multiple holders. -wait() ------- +These are local process handles. FoundationDB's RPC layer uses ``RequestStream`` and ``ReplyPromise`` for network requests and replies. A request can carry a reply promise to another process; sending the reply there makes the caller's future ready. -At the point when a receiver holding a ``Future`` needs the ``T`` to continue computation, it invokes the ``wait()`` statement with the ``Future`` as its parameter. The ``wait()`` statement allows the calling actor to pause execution until the value of the future is set, returning a value of type ``T``. During the wait, other actors can continue execution, providing asynchronous concurrency within a single process. +co_await and co_return +---------------------- -ACTOR ------ +``co_await`` waits for a future without blocking the event loop. Awaiting a ready future continues immediately; otherwise the coroutine suspends until it becomes ready. If the future contains an error, awaiting it throws that ``Error``. -Only functions labeled with the ``ACTOR`` tag can call ``wait()``. Actors are the essential unit of asynchronous work and can be composed to create complex message-passing systems. By composing actors, futures can be chained together so that the result of one depends on the output of another. +A coroutine returning ``Future`` completes with ``co_return value;``. For ``Future``, use ``co_return;`` to signal completion without a payload. Awaiting a ``Future`` does not produce a value to assign. -An actor is declared as returning a ``Future`` where ``T`` may be ``Void`` if the actor's return value is used only for signaling. Each actor is preprocessed into a C++11 class with internal callbacks and supporting functions. +Local variables and ownership +----------------------------- -State ------ +Local variables obey normal C++ scope rules and remain alive across suspension while their scope is active. Value parameters are stored in the coroutine frame. References, pointers, and views such as ``StringRef`` do not extend the lifetime of their referents or backing storage. Use owning types such as ``Reference`` and ``Standalone`` when the coroutine needs to retain an object or its bytes. -The ``state`` keyword is used to scope a variable so that it is visible across multiple ``wait()`` statements within an actor. The use of a ``state`` variable is illustrated in the example actor below. +PromiseStream and FutureStream +------------------------------------ -PromiseStream, FutureStream ---------------------------------- +``PromiseStream`` and ``FutureStream`` represent a series of asynchronous messages. Awaiting a ``FutureStream`` consumes its next value. If an item is already queued, execution continues without suspension: -When a component wants to work with a *stream* of asynchronous messages rather than a single message, it can use ``PromiseStream`` and ``FutureStream``. These constructs allow for two important features: multiplexing and reliable delivery of messages. They also play an important role in Flow design patterns. For example, many of the servers in FoundationDB expose their interfaces as a ``struct`` of promise streams—one for each request type. +.. code-block:: cpp -waitNext() ----------- + Future forwardWithOffset(FutureStream input, + PromiseStream output, + int offset) { + while (true) { + int value = co_await input; + output.send(value + offset); + } + } -``waitNext()`` is the counterpart of ``wait()`` for streams. It pauses program execution and waits for the next value in a ``FutureStream``. If there is a value ready in the stream, execution continues without delay. +Waiting for multiple inputs +--------------------------- -choose . . . when ------------------ +``race()`` from ``flow/CoroUtils.h`` waits for the first ready input. It returns a ``std::variant`` whose index matches the winning argument. Inputs can be futures or streams; a winning stream consumes one item. If several inputs are already ready, the lowest argument index wins. Errors propagate from the winning input. -The ``choose`` and ``when`` constructs allow an actor to wait for multiple futures at once in a ordered and predictable way. +Losing inputs are detached from the race, not explicitly cancelled. They may still be cancelled if releasing the race drops their last future reference. Retain a future separately when its operation must continue after losing. A completed non-stream future stays ready, so replace it or remove it from a repeated race after handling it. -Example: A Server Interface +Example: A server interface --------------------------- -Below is a actor that runs on single server communicating over the network. Its functionality is to maintain a count in response to asynchronous messages from other actors. It supports an interface implemented with a loop containing a ``choose`` statement with a ``when`` for each request type. Each ``when`` uses ``waitNext()`` to asynchronously wait for the next request in the stream. The add and subtract interfaces modify the count itself, stored with a state variable. The get interface takes a ``Promise`` instead of just an ``int`` to facilitate sending back the return message. - -To write the equivalent code directly in C++, a developer would have to implement a complex set of callbacks with exception-handling, requiring far more engineering effort. Flow makes it much easier to implement this sort of asynchronous coordination, with no loss of performance: - -.. code-block:: c - - ACTOR void serveCountingServerInterface( - CountingServerInterface csi) { - state int count = 0; - while (1) { - choose { - when (int x = waitNext(csi.addCount.getFuture())){ - count += x; - } - when (int x = waitNext(csi.subtractCount.getFuture())){ - count -= x; - } - when (Promise r = waitNext(csi.getCount.getFuture())){ - r.send( count ); // goes to client - } +This local server maintains a count in response to asynchronous messages. It has one promise stream per request type and races those streams in a loop. A request for the current count carries a ``Promise`` through which the server replies. A network interface uses the RPC types described above and also needs serialization. + +.. code-block:: cpp + + #include "flow/CoroUtils.h" + + struct CountingServerInterface { + PromiseStream addCount; + PromiseStream subtractCount; + PromiseStream> getCount; + }; + + Future serveCountingServerInterface(CountingServerInterface csi) { + int count = 0; + while (true) { + auto request = co_await race(csi.addCount.getFuture(), + csi.subtractCount.getFuture(), + csi.getCount.getFuture()); + switch (request.index()) { + case 0: + count += std::get<0>(request); + break; + case 1: + count -= std::get<1>(request); + break; + case 2: + std::get<2>(request).send(count); + break; } } } -Caveats -======= - -Even though Flow code looks a lot like C++, it is not. It has different rules and the files are preprocessed. It is always important to keep this in mind when programming flow. - -We still want to be able to use IDEs and modern editors (with language servers like cquery or clang-based completion engines like ycm). Because of this there is a header-file ``actorcompiler.h`` in flow which defines preprocessor definitions to make flow compile as normal C++ code. CMake even supports a special mode so that it doesn't preprocess flow files. This mode can be used by passing ``-DOPEN_FOR_IDE=ON`` to cmake. Additionally we generate a special ``compile_commands.json`` into the source-directory which will support opening the project in IDEs and editors that look for a compilation database. - -Some preprocessor definitions will not fix all issues though. When programming Flow the following things have to be taken care of by the programmer: - -- Local variables don't survive a call to ``wait``. So this would be legal Flow code, but NOT legal C++ code: - - .. code-block:: c - - ACTOR void foo { - int i = 0; - wait(someFuture); - int i = 2; - wait(someOtherFuture) - } - - - In order to make this not break IDE-support one can either rename the second occurrence of this variable or, if this is not desired as it might make the code unreadable, one can use scoping: - - - .. code-block:: c - - ACTOR void foo { - { - int i = 0; - wait(someFuture); - } - { - int i = 2; - wait(someOtherFuture) - } - } - -- An ``ACTOR`` is compiled into a class internally. Which means that within an actor-function, ``this`` is a valid pointer to this class. But using them explicitly (or as described later implicitly) will break IDE support. One can use ``THIS`` and ``THIS_ADDR`` instead. But be careful as ``THIS`` will be of type ``nullptr_t`` in IDE-mode and of the actor-type in normal compilation mode. -- Lambdas and state variables are weird in a sense. After actor compilation, a state variable is a member of the compiled actor class. In IDE mode it is considered a normal local variable. This can result in some surprising side-effects. So the following code will only compile if the method ``Foo::bar`` is defined as ``const``: - - .. code-block:: c - - ACTOR foo() { - state Foo f; - foo([=]() { f.bar(); }) - } - +The caller must keep the returned ``Future`` alive while the server is needed. The local ``count`` remains alive across each suspension. - If it is not, one has to pass the member explicitly as a reference: +Cancellation and lifetime pitfalls +================================== - .. code-block:: c +By default, dropping the last reference to a pending coroutine's returned future cancels it. An explicit ``Future::cancel()`` also requests cancellation. A suspended coroutine resumes by throwing ``actor_cancelled`` from its await. Local objects are destroyed as their scopes unwind. Keep the future alive, or await it directly, when the work must continue. - ACTOR foo() { - state Foo f; - auto x = &f; - foo([x]() { x->bar(); }) - } +Error and retry handlers must propagate ``actor_cancelled`` after any required synchronous cleanup. Do not silently swallow ``broken_promise`` or other errors unless the caller's contract explicitly handles them. C++ does not allow ``co_await`` inside a ``catch`` handler; save the error and await asynchronous recovery after leaving the handler. -- state variables in Flow don't follow the normal scoping rules. So in Flow a state variable can be defined in an inner scope and later it can be used in the outer scope. In order to not break compilation in IDE-mode, always define state variables in the outermost scope they will be used. +Captures in a coroutine lambda belong to the lambda's closure, which may be destroyed while the coroutine is suspended. Prefer a named coroutine with explicit value parameters when the work can outlive the call that starts it. +See the `Flow tutorial `_, the `coroutine design guide `_, and the `runnable coroutine tutorial `_ for further examples. diff --git a/documentation/sphinx/source/mr-status-json-schemas.rst.inc b/documentation/sphinx/source/mr-status-json-schemas.rst.inc index ca5d9384369..a06bf5411bb 100644 --- a/documentation/sphinx/source/mr-status-json-schemas.rst.inc +++ b/documentation/sphinx/source/mr-status-json-schemas.rst.inc @@ -325,6 +325,102 @@ "p99":0.0, "p99.9":0.0 }, + "commit_batch_transactions":{ // Number of transactions in the commit batch + "count":0, + "min":0.0, + "max":0.0, + "median":0.0, + "mean":0.0, + "p25":0.0, + "p90":0.0, + "p95":0.0, + "p99":0.0, + "p99.9":0.0 + }, + "commit_batch_bytes":{ // Total size of transactions in the commit batch (bytes) + "count":0, + "min":0.0, + "max":0.0, + "median":0.0, + "mean":0.0, + "p25":0.0, + "p90":0.0, + "p95":0.0, + "p99":0.0, + "p99.9":0.0 + }, + "commit_batching_waiting":{ // Time transactions spent waiting in queue before commit batch processing (seconds) + "count":0, + "min":0.0, + "max":0.0, + "median":0.0, + "mean":0.0, + "p25":0.0, + "p90":0.0, + "p95":0.0, + "p99":0.0, + "p99.9":0.0 + }, + "commit_preresolution_latency":{ // Latency of the pre-resolution phase (seconds) + "count":0, + "min":0.0, + "max":0.0, + "median":0.0, + "mean":0.0, + "p25":0.0, + "p90":0.0, + "p95":0.0, + "p99":0.0, + "p99.9":0.0 + }, + "commit_resolution_latency":{ // Latency of the resolution phase (seconds) + "count":0, + "min":0.0, + "max":0.0, + "median":0.0, + "mean":0.0, + "p25":0.0, + "p90":0.0, + "p95":0.0, + "p99":0.0, + "p99.9":0.0 + }, + "commit_postresolution_latency":{ // Latency of the post-resolution phase (seconds) + "count":0, + "min":0.0, + "max":0.0, + "median":0.0, + "mean":0.0, + "p25":0.0, + "p90":0.0, + "p95":0.0, + "p99":0.0, + "p99.9":0.0 + }, + "commit_tlog_logging_latency":{ // Latency of the TLog logging phase (seconds) + "count":0, + "min":0.0, + "max":0.0, + "median":0.0, + "mean":0.0, + "p25":0.0, + "p90":0.0, + "p95":0.0, + "p99":0.0, + "p99.9":0.0 + }, + "commit_reply_latency":{ // Latency of the reply phase (seconds) + "count":0, + "min":0.0, + "max":0.0, + "median":0.0, + "mean":0.0, + "p25":0.0, + "p90":0.0, + "p95":0.0, + "p99":0.0, + "p99.9":0.0 + }, "grv_latency_bands":{ // How many GRV requests belong to the latency (in seconds) band (e.g., How many requests belong to [0.01,0.1] latency band). The key is the upper bound of the band and the lower bound is the next smallest band (or 0, if none). Example: {0.01: 27, 0.1: 18, 1: 1, inf: 98,filtered: 10}, we have 18 requests in [0.01, 0.1) band. "$map_key=upperBoundOfBand": 1 }, diff --git a/documentation/sphinx/source/rangelock.rst b/documentation/sphinx/source/rangelock.rst index 80c69d449c0..ba38044dbd9 100644 --- a/documentation/sphinx/source/rangelock.rst +++ b/documentation/sphinx/source/rangelock.rst @@ -48,36 +48,36 @@ Put an exclusive read lock on a range. The range must be within the user key spa The locking request is rejected with a range_lock_reject error if the range contains any existing lock with a different range, user, or lock type. Currently, only the ExclusiveReadLock type is supported, but the design allows for future extension. -``ACTOR Future takeExclusiveReadLockOnRange(Database cx, KeyRange range, RangeLockOwnerName ownerUniqueID);`` +``Future takeExclusiveReadLockOnRange(Database cx, KeyRange range, RangeLockOwnerName ownerUniqueID);`` Release an exclusive read lock on a range. The range must be within the user key space, aka ``"" ~ \xff``. The release request is rejected with a range_lock_reject error if the range contains any existing lock with a different range, user, or lock type. -``ACTOR Future releaseExclusiveReadLockOnRange(Database cx, KeyRange range, RangeLockOwnerName ownerUniqueID);`` +``Future releaseExclusiveReadLockOnRange(Database cx, KeyRange range, RangeLockOwnerName ownerUniqueID);`` Note that takeExclusiveReadLockOnRange and releaseExclusiveReadLockOnRange are transactional. If the execution of the API is successful, all ranges are guaranteed to be locked/unlocked at a single version. If the execution is failed, no range is locked/unlocked. -Get exclusive read locks on the input range +Get exclusive read locks on the input range, optionally filtering by owner. -``ACTOR Future>> findExclusiveReadLockOnRange(Database cx, KeyRange range);`` +``Future>> findExclusiveReadLockOnRange(Database cx, KeyRange range, Optional ownerName = Optional());`` Register a range lock owner to database metadata. -``ACTOR Future registerRangeLockOwner(Database cx, RangeLockOwnerName ownerUniqueID, std::string description);`` +``Future registerRangeLockOwner(Database cx, RangeLockOwnerName ownerUniqueID, std::string description);`` Remove an owner from the database metadata -``ACTOR Future removeRangeLockOwner(Database cx, RangeLockOwnerName ownerUniqueID);`` +``Future removeRangeLockOwner(Database cx, RangeLockOwnerName ownerUniqueID);`` Get all registered range lock owners -``ACTOR Future> getAllRangeLockOwners(Database cx);`` +``AsyncResult> getAllRangeLockOwners(Database cx);`` Get a range lock owner by uniqueId -``ACTOR Future> getRangeLockOwner(Database cx, RangeLockOwnerName ownerUniqueID);`` +``Future> getRangeLockOwner(Database cx, RangeLockOwnerName ownerUniqueID);`` Using ``fdbcli`` @@ -103,11 +103,11 @@ Example usage ------------- When submitting a bulk load task on a range, we block user write traffic to the range. -``ACTOR Future setBulkLoadSubmissionTransaction(Transaction* tr, BulkLoadTaskState bulkLoadTask);`` +``Future setBulkLoadSubmissionTransaction(Transaction* tr, BulkLoadTaskState bulkLoadTask);`` Upon a bulk load task completes on a range, we unblock user write traffic on the range. -``ACTOR Future setBulkLoadFinalizeTransaction(Transaction* tr, KeyRange range, UID taskId);`` +``Future setBulkLoadFinalizeTransaction(Transaction* tr, KeyRange range, UID taskId);`` Range Lock Design (Exclusive Read Lock) ======================================= diff --git a/documentation/sphinx/source/read-write-path.rst b/documentation/sphinx/source/read-write-path.rst index 277950b832f..8abb232606d 100644 --- a/documentation/sphinx/source/read-write-path.rst +++ b/documentation/sphinx/source/read-write-path.rst @@ -382,19 +382,19 @@ Read write path of a transaction This section uses an example transaction to describe how a transaction with both read and write operation works in FDB. -Suppose application creates the following transaction, where *Future* is an object that holds an asynchronous call and -becomes ready when the async call returns, and *wait()* is a synchronous point when the code waits for futures to be ready. -The following code reads key k1 and k2 from database, increases k1’s value by 1 and write back k1’s new value into database. +Suppose an application creates the following transaction. In this pseudocode, ``Future`` represents an asynchronous +read result, and ``co_await`` suspends the coroutine until that result is ready without blocking other work. +The example reads integer values from keys k1 and k2 and writes their sum back to k1. **Example Transaction** :: Line1: Transaction tr; Line2: Future fv1 = tr.get(k1); Line3: Future fv2 = tr.get(k2); - Line4: v1 = wait(fv1); - Line5: v2 = wait(fv2); - Line6: tr.set(v1+v2); - Line7: tr.commit(); + Line4: v1 = co_await fv1; + Line5: v2 = co_await fv2; + Line6: tr.set(k1, v1+v2); + Line7: co_await tr.commit(); The transaction starts with the read path: diff --git a/documentation/sphinx/source/release-notes/release-notes-740.rst b/documentation/sphinx/source/release-notes/release-notes-740.rst index 88244098b7f..511c8710e64 100644 --- a/documentation/sphinx/source/release-notes/release-notes-740.rst +++ b/documentation/sphinx/source/release-notes/release-notes-740.rst @@ -4,6 +4,36 @@ Release Notes ############# +7.4.7 +===== + +AVX enabled release. + +* Avoided restarting transaction-system recovery when an old-epoch backup worker fails. `(PR #13939) `_ +* Moved ``waitForShardReady`` outside the transaction used to finish data moves, preventing slow destination servers from exhausting the transaction lifetime and stalling data distribution. `(PR #13642) `_ +* Fixed a use-after-free when cancelling S3 backup file copies. `(PR #13916) `_ +* Fixed server knob overrides being silently skipped when a knob name is shared with client knobs. `(PR #13932) `_ +* Added opt-in eviction of stale client connections to failed storage servers and proxies. `(PR #13912) `_ +* Disabled data distribution global admission control by default. `(PR #13917) `_ +* Fixed crashes and incorrect behavior in snapshot options, cancelled file reads, data distribution shutdown, and TLog minimum popped-version tracking. `(PR #13871) `_ +* Fixed an incorrect shard assignment after an excluded server failed. `(PR #13869) `_ +* Added a retry limit for repeated ``startMoveKeys`` failures to prevent infinite retries. `(PR #13782) `_ +* Fixed exclusion tracking for data distribution. `(PR #13766) `_ +* Fixed read and write overflow handling for snapshot manifests larger than 2 GB. `(PR #13820) `_ +* Added GCM authentication tags to encrypted backup files and fixed encrypted restore block-size initialization. `(PR #13777) `_ +* Dropped stale backup workers during recovery. `(PR #13655) `_ +* Fixed successful automatic-idempotency replays not populating the committed version. `(PR #13294) `_ +* Fixed read-ahead cache reads extending past the end of small or final blocks. `(PR #13292) `_ +* Improved RocksDB memory accounting by charging write buffers and selected memory usage to the block cache. `(PR #13371) `_, `(PR #13346) `_ +* Improved data distribution relocation backoff and admission control to reduce retry contention and pipeline pressure. `(PR #13145) `_, `(PR #13280) `_ +* Added mutation-log-type configuration for backups. `(PR #13127) `_ +* Fixed encrypted restore failures, including a ``platform_error`` crash. `(PR #13118) `_ +* Made cluster status requests asynchronous so status retrieval can return before the timeout. `(PR #13099) `_ +* Hardened S3 error handling for non-XML and HTTP 4xx responses. `(PR #12996) `_, `(PR #13021) `_ +* Fixed ``minRestorableVersion`` and ``maxRestorableVersion`` updates when mutation logs are missing. `(PR #12710) `_ +* Restricted ``backup_worker_enabled`` configuration through ``fdbcli``. `(PR #12711) `_ +* Enabled TLS hostname validation against certificate CN and SAN values. `(PR #12730) `_ + 7.4.6 ===== diff --git a/documentation/sphinx/source/technical-overview.rst b/documentation/sphinx/source/technical-overview.rst index c5950d11434..4fc222bd7e2 100644 --- a/documentation/sphinx/source/technical-overview.rst +++ b/documentation/sphinx/source/technical-overview.rst @@ -24,7 +24,7 @@ These documents explain the engineering design of FoundationDB, with detailed in * :doc:`fault-tolerance`: FoundationDB provides fault tolerance by intelligently replicating data across a distributed cluster of machines. Our architecture is designed to minimize service interruption and data loss in the event of machine failures. -* :doc:`flow`: FoundationDB faces rigorous engineering challenges for high performance and scalability. To meet these challenges, we implemented Flow, an extension to C++ that supports actor-based concurrency with new keywords and control-flow primitives while retaining speed and I/O efficiency. +* :doc:`flow`: FoundationDB faces rigorous engineering challenges for high performance and scalability. To meet these challenges, we implemented Flow, an asynchronous runtime that combines standard C++ coroutines with futures, message streams, and cooperative scheduling while retaining speed and I/O efficiency. * :doc:`testing`: FoundationDB uses a combined regime of robust simulation, live performance testing, and hardware-based failure testing to meet exacting standards of correctness and performance. diff --git a/documentation/sphinx/source/testing.rst b/documentation/sphinx/source/testing.rst index 4ab8b183e99..c3672fe15a2 100644 --- a/documentation/sphinx/source/testing.rst +++ b/documentation/sphinx/source/testing.rst @@ -10,7 +10,7 @@ Rigorous testing is central to our engineering process. The :doc:`features of ou Simulation ========== -Simulation is a powerful tool for testing system correctness. Our simulation technology, called Simulation, is enabled by and tightly integrated with :doc:`flow`, our programming language for actor-based concurrency. In addition to generating efficient production code, Flow works with Simulation for simulated execution. +Simulation is a powerful tool for testing system correctness. Our simulation technology, called Simulation, is enabled by and tightly integrated with :doc:`flow`, our asynchronous runtime for C++ coroutines. Flow supports both production execution and deterministic simulated execution. The major goal of Simulation is to make sure that we find and diagnose issues in simulation rather than the real world. Simulation runs tens of thousands of simulations every night, each one simulating large numbers of component failures. Based on the volume of tests that we run and the increased intensity of the failures in our scenarios, we estimate that we have run the equivalent of roughly one trillion CPU-hours of simulation on FoundationDB. diff --git a/documentation/tutorial/CMakeLists.txt b/documentation/tutorial/CMakeLists.txt deleted file mode 100644 index ce3458eaca7..00000000000 --- a/documentation/tutorial/CMakeLists.txt +++ /dev/null @@ -1,29 +0,0 @@ -set(TUTORIAL_SRCS tutorial.actor.cpp) - -add_flow_target(EXECUTABLE NAME tutorial SRCS "${TUTORIAL_SRCS}") -target_link_libraries(tutorial PUBLIC fdbclient) - -# Small exercise -set(PRINT_IN_ORDER_SRCS print_in_order.actor.cpp) -add_flow_target(EXECUTABLE NAME print_in_order SRCS "${PRINT_IN_ORDER_SRCS}") -target_link_libraries(print_in_order PUBLIC fdbclient) - -# Medium exercise -set(MAKE_H2O_SRCS make_h2o.actor.cpp) -add_flow_target(EXECUTABLE NAME make_h2o SRCS "${MAKE_H2O_SRCS}") -target_link_libraries(make_h2o PUBLIC fdbclient) - -# Extra medium exercise -set(DP_SRCS dining_philosophers.actor.cpp) -add_flow_target(EXECUTABLE NAME dining_philosophers SRCS "${DP_SRCS}") -target_link_libraries(dining_philosophers PUBLIC fdbclient) - -# Playground to learn flow -set(PLAY_SRC play.actor.cpp) -add_flow_target(EXECUTABLE NAME play SRCS "${PLAY_SRC}") -target_link_libraries(play PUBLIC fdbclient) - -# Playground to learn flow and its interaction over a network (client/server) -set(PLAY_NETWORK_SRC play_network.actor.cpp) -add_flow_target(EXECUTABLE NAME play_network SRCS "${PLAY_NETWORK_SRC}") -target_link_libraries(play_network PUBLIC fdbclient) diff --git a/documentation/tutorial/dining_philosophers.actor.cpp b/documentation/tutorial/dining_philosophers.actor.cpp deleted file mode 100644 index 5411c22f739..00000000000 --- a/documentation/tutorial/dining_philosophers.actor.cpp +++ /dev/null @@ -1,345 +0,0 @@ -/* - * dining_philosophers.actor.cpp - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2025 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -#include "fmt/format.h" -#include "flow/flow.h" -#include "flow/Platform.h" -#include "flow/DeterministicRandom.h" -#include "fdbclient/NativeAPI.h" -#include "fdbclient/ReadYourWrites.h" -#include "flow/TLSConfig.h" -#include -#include -#include -#include -#include "flow/actorcompiler.h" - -// Flow solution to Dining Philosophers problem -// (https://en.wikipedia.org/wiki/Dining_philosophers_problem; or -// https://leetcode.com/problems/the-dining-philosophers/description/, -// but note that that calls for a single process/threaded solution -// and here we implement a distributed solution). -// -// This uses most of the techniques illustrated in tutorial.actor.cpp. -// A server is used to track "fork ownership". The dining -// philosophers are modeled as clients who must request and obtain -// ownership of forks prior to eating. -// -// To do this exercise, delete the code below down to main(), then -// implement it using techniques you see in tutorial.actor.cpp. - -enum DPEndpoints { - WLTOKEN_DP_SERVER = WLTOKEN_FIRST_AVAILABLE, - DP_ENDPOINT_COUNT, -}; - -struct DPServerInterface { - constexpr static FileIdentifier file_identifier = 9957031; - RequestStream getInterface; - RequestStream getFork; - RequestStream releaseFork; - - template - void serialize(Ar& ar) { - serializer(ar, getInterface, getFork, releaseFork); - } -}; - -struct GetInterfaceRequest { - constexpr static FileIdentifier file_identifier = 13789052; - ReplyPromise reply; - - template - void serialize(Ar& ar) { - serializer(ar, reply); - } -}; - -// This is sent in both requests and responses. -// NOTE: it seems better to have reply types be structs with -// file_identifier members and serialize() overrides. -// tutorial.actor.cpp has an example where a std::string is sent -// directly. Attempts to do similar things with base types like int -// will run into trouble. (Caveat: I didn't try uint32_t or the like; -// maybe those work.) -struct ForkState { - constexpr static FileIdentifier file_identifier = 998236; - // ID [0, N) of the philospher requesting this fork. - uint32_t clientId; - // The number of the fork we are requesting, also [0, N). - // Philosophers numbered i request forks i and (i + 1) % N, - // not necessarily in that order. - uint32_t forkNumber; - ForkState() : clientId(0), forkNumber(0) {} - ForkState(int c, int f) : clientId(c), forkNumber(f) {} - - template - void serialize(Ar& ar) { - serializer(ar, clientId, forkNumber); - } -}; - -struct GetForkRequest { - constexpr static FileIdentifier file_identifier = 14904213; - - ForkState forkState; - ReplyPromise reply; - - explicit GetForkRequest(ForkState fork_state) : forkState(fork_state) {} - GetForkRequest() = default; - - template - void serialize(Ar& ar) { - serializer(ar, forkState, reply); - } -}; - -struct ReleaseForkRequest { - constexpr static FileIdentifier file_identifier = 5914324; - ForkState forkState; - ReplyPromise reply; - - explicit ReleaseForkRequest(ForkState fork_state) : forkState(fork_state) {} - ReleaseForkRequest() = default; - - template - void serialize(Ar& ar) { - serializer(ar, forkState, reply); - } -}; - -ACTOR Future dpClient(NetworkAddress serverAddress, int idnum, int numEaters) { - std::cout << format( - "dpClient: starting philosopher #%d, server address [%s]\n", idnum, serverAddress.toString().c_str()); - - state DPServerInterface server; - server.getInterface = RequestStream(Endpoint::wellKnown({ serverAddress }, WLTOKEN_DP_SERVER)); - DPServerInterface s = wait(server.getInterface.getReply(GetInterfaceRequest())); - server = s; - - state int firstfork; - state int secondfork; - state GetForkRequest gf1; - state GetForkRequest gf2; - state ReleaseForkRequest rf1; - state ReleaseForkRequest rf2; - - // Change this to true and it should deadlock pretty quickly. - bool CAUSE_DEADLOCK = false; - - if (CAUSE_DEADLOCK) { - firstfork = idnum; - secondfork = (idnum + 1) % numEaters; - } else { - // The deadlock we must avoid is where each eater has 1 fork and is blocked - // trying to get one held by a person next to them. The protocol is is that - // odd numbered eaters get the fork to the left of them first, then the one - // to the right. Even numbered eaters get the fork to the right of them first, - // then the one to the left. - if (idnum % 2) { - firstfork = idnum; - secondfork = (idnum + 1) % numEaters; - } else { - firstfork = (idnum + 1) % numEaters; - secondfork = idnum; - } - } - - state double msec; - state int meals_eaten = 0; - - try { - loop { - std::cout << format("dpClient: eater [%d] WANTS TO EAT...\n", idnum); - - gf1 = GetForkRequest(ForkState(idnum, firstfork)); - ForkState reply = wait(server.getFork.getReply(gf1)); - ASSERT(reply.clientId == idnum); - ASSERT(reply.forkNumber == firstfork); - - gf2 = GetForkRequest(ForkState(idnum, secondfork)); - ForkState reply2 = wait(server.getFork.getReply(gf2)); - ASSERT(reply2.clientId == idnum); - ASSERT(reply2.forkNumber == secondfork); - - std::cout << format("dpClient: eater [%d] NOW EATING...\n", idnum); - msec = deterministicRandom()->randomInt(0, 1000) / 1000.0 + 1; - wait(delay(msec)); - meals_eaten++; - - rf1 = ReleaseForkRequest(ForkState(idnum, firstfork)); - ForkState reply3 = wait(server.releaseFork.getReply(rf1)); - - rf2 = ReleaseForkRequest(ForkState(idnum, secondfork)); - ForkState reply4 = wait(server.releaseFork.getReply(rf2)); - - // Elvis has left the bulding. - std::cout << format("dpClient: eater [%d] HAS RELEASED ITS FORKS (%d meals eaten)\n", idnum, meals_eaten); - - msec = deterministicRandom()->randomInt(0, 1000) / 1000.0 + 1; - wait(delay(msec)); - } - } catch (Error& e) { - std::cerr << format("dpClient: caught Error code %s, %s\n", e.code(), e.what()); - } - - std::cout << format("dpClient: philosopher #%d finished.\n", idnum); - return Void(); -} - -ACTOR Future dpServerLoop() { - state DPServerInterface dpServer; - dpServer.getInterface.makeWellKnownEndpoint(WLTOKEN_DP_SERVER, TaskPriority::DefaultEndpoint); - - std::cout << format("dpServer: starting...\n"); - - // Maps int fork to int eaterId who owns it. - state std::map forkOwners; - - // Maps to eaters who are waiting for a fork. In a more general problem - // of waiting for a shared resource, this might be a queue. Because of - // the specifics of Dining Philosophers, at most one eater is waiting for - // an in-use fork, so it can be a singleton. - // Maps int fork to pending reply to an eater who is waiting for it to be freed. - state std::map pending; - - loop { - try { - choose { - when(GetInterfaceRequest req = waitNext(dpServer.getInterface.getFuture())) { - req.reply.send(dpServer); - } - when(GetForkRequest req = waitNext(dpServer.getFork.getFuture())) { - int clientId = req.forkState.clientId; - int forkNo = req.forkState.forkNumber; - auto it = forkOwners.find(forkNo); - if (it == forkOwners.end()) { - // Available immediately, give it. - std::cout << format("dpServerLoop: eater %d gets fork %d\n", clientId, forkNo); - forkOwners[forkNo] = clientId; - req.reply.send(req.forkState); - } else { - auto it2 = pending.find(forkNo); - ASSERT(it2 == pending.end()); - std::cout << format("dpServerLoop: eater %d has to wait for fork %d\n", clientId, forkNo); - pending[forkNo] = req; - } - } - when(ReleaseForkRequest req = waitNext(dpServer.releaseFork.getFuture())) { - int clientId = req.forkState.clientId; - int forkNo = req.forkState.forkNumber; - auto it = forkOwners.find(forkNo); - if (it == forkOwners.end()) { - std::cerr << format( - "dpServerLoop: request from clientId %d to free fork %d which is not owned by anybody\n", - clientId, - forkNo); - } else if (it->second != clientId) { - std::cerr << format("dpServerLoop: request from clientId %d to free fork %d whichis owned by " - "somebody else [%d]\n ", - clientId, - forkNo, - it->second); - } else { - std::cout << format("dpServerLoop: eater %d is freeing fork %d\n", clientId, forkNo); - forkOwners.erase(it); - auto it2 = pending.find(forkNo); - if (it2 == pending.end()) { - std::cout << format( - "dpServerLoop: eater %d, fork %d: nobody is waiting on this fork\n", clientId, forkNo); - req.reply.send(req.forkState); - } else { - GetForkRequest pending_req = it2->second; - pending.erase(it2); - int nextClient = pending_req.forkState.clientId; - std::cout << format("dpServerLoop: eater %d fork %d: giving to waiting eater %d\n", - clientId, - forkNo, - nextClient); - forkOwners[forkNo] = nextClient; - req.reply.send(req.forkState); - pending_req.reply.send(pending_req.forkState); - } - } - } - } - } catch (Error& e) { - // XXX this is cargo-culted from tutorial.actor.cpp - if (e.code() != error_code_operation_obsolete) { - std::cerr << format("dpServerLoop: Error %d / %s\n", e.code(), e.what()); - throw e; - } - } - } -} - -static void usage(const char* argv0) { - std::cerr << format("Usage: %s -p portNum | -s serverAddress\n", argv0); -} - -int main(int argc, char** argv) { - // Cargo-culted from tutorial.actor.cpp. - platformInit(); - g_network = newNet2(TLSConfig(), /*useThreadPool=*/false, /*useMetrics=*/true); - - if (argc != 3) { - usage(argv[0]); - return 1; - } - - NetworkAddress serverAddress; - bool isServer = false; - - if (0 == strcmp(argv[1], "-p")) { - isServer = true; - serverAddress = NetworkAddress::parse("0.0.0.0:" + std::string(argv[2])); - } else if (0 == strcmp(argv[1], "-s")) { - serverAddress = NetworkAddress::parse(argv[2]); - } else { - usage(argv[0]); - return 1; - } - - FlowTransport::createInstance(!isServer, 0, DP_ENDPOINT_COUNT); - - std::vector> all; - if (isServer) { - try { - auto listenError = FlowTransport::transport().bind(serverAddress, serverAddress); - if (listenError.isError()) { - listenError.get(); - } - } catch (Error& e) { - std::cerr << format( - "Error binding to address [%s]: %d, %s\n", serverAddress.toString().c_str(), e.code(), e.what()); - return 2; - } - all.emplace_back(dpServerLoop()); - } else { - for (int i = 0; i < 5; i++) { - all.emplace_back(dpClient(serverAddress, i, 5)); - } - } - - auto f = stopAfter(waitForAll(all)); - g_network->run(); - - return 0; -} diff --git a/documentation/tutorial/make_h2o.actor.cpp b/documentation/tutorial/make_h2o.actor.cpp deleted file mode 100644 index dea1fb7998c..00000000000 --- a/documentation/tutorial/make_h2o.actor.cpp +++ /dev/null @@ -1,190 +0,0 @@ -/* - * make_h2o.actor.cpp - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2025 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -#include "fmt/format.h" -#include "flow/flow.h" -#include "flow/Platform.h" -#include "flow/DeterministicRandom.h" -#include "fdbclient/NativeAPI.h" -#include "fdbclient/ReadYourWrites.h" -#include "flow/TLSConfig.h" - -#include -#include -#include -#include -#include - -#include "flow/actorcompiler.h" - -// Flow solution to https://leetcode.com/problems/building-h2o/ -// -// The description of this problem calls for threads running as -// hydrogen and oxygen. To do this yourself, delete the code from -// this file other than main(), read the problem description above, -// and implement something in the spirit of what is requested. -// -// Caveat: this version was written on like day 2 of programming with -// Flow. This seems to work but may have bugs, non-idiomatic code, or -// generally do naive stuff. -// -// Notes on this solution: we model these as actors and we start a lot -// of them simultaneously in batches from orchestrate() below. Bonded -// sets of 2 H's and 1 O are supposed to complete together. We -// arrange this by having the H's store off a Promise, then block -// on the future. The O's pass their ID and the H's simply log it. -// An O finishes when it has seen two H's. - -static int h_seqno; -static int o_seqno; -std::queue h_queue; -std::map*> wakeup; - -ACTOR Future hydrogen(AsyncTrigger* h_ready_trigger) { - state int h_id = h_seqno++; - std::cout << format("Hydrogen %d starting\n", h_id); - state Promise promise; - state Future future = promise.getFuture(); - ASSERT(wakeup.find(h_id) == wakeup.end()); - wakeup[h_id] = &promise; - h_queue.push(h_id); - h_ready_trigger->trigger(); - - loop choose { - when(int o = wait(future)) { - std::cout << format("Hydrogen %d bound to Oxygen %d, returning\n", h_id, o); - return Void(); - } - } -} - -ACTOR Future oxygen(AsyncTrigger* h_ready_trigger) { - state int h_bound = 0; - state int o_id = o_seqno++; - std::cout << format("Oxygen %d starting\n", o_id); - - loop choose { - when(wait(h_ready_trigger->onTrigger())) { - while (h_queue.size() > 0) { - state int h = h_queue.front(); - h_queue.pop(); - auto it = wakeup.find(h); - ASSERT(it != wakeup.end()); - std::cout << format("Oxygen %d bound to Hydrogen %d\n", o_id, h); - wakeup.erase(it); - it->second->send(o_id); - h_bound++; - if (h_bound == 2) { - std::cout << format(" Oxygen %d returning\n", o_id); - return Void(); - } - } - } - } -} - -ACTOR Future orchestrate() { - state int h_threads = 0; - state int o_threads = 0; - state int total_h_threads = 0; - state int total_o_threads = 0; - state int bigloops = 0; - state std::vector> all; - state AsyncTrigger h_ready_trigger; - - // Use much smaller numbers, like say 10, to debug assertion failures, - // deadlocks, or other weirdness. - while (bigloops < 1000) { - int n = deterministicRandom()->randomInt(0, 3); - if (n < 2) { - all.emplace_back(hydrogen(&h_ready_trigger)); - h_threads++; - } else { - ASSERT(n < 3); - all.emplace_back(oxygen(&h_ready_trigger)); - o_threads++; - } - - // FIXME: this is going to create some stop-and-go dynamics. - // Probably there is a way to continuously generate new - // H and O workers subject to some kind of low water mark - // type of condition. - if (h_threads + o_threads >= 1000) { - while (2 * o_threads < h_threads) { - all.emplace_back(oxygen(&h_ready_trigger)); - o_threads++; - } - while (h_threads < 2 * o_threads) { - all.emplace_back(hydrogen(&h_ready_trigger)); - h_threads++; - } - if (h_threads != 2 * o_threads) { - std::cout << format("WEIRDNESS: h_threads [%d] != 2*o_threads [%d]\n", h_threads, o_threads); - ASSERT(false); - } - - std::cout << format( - "orchestrate: blocking with %d H threads, %d O threads; %d bigloops\n", h_threads, o_threads, bigloops); - - // Without the following two lines, or if you just call - // waitForAll(all) without the wait (e.g. due to cargo - // culting it wrong), typically either you will get a - // deadlock on the wait() or, without it, wakeup.size() - // will be non-zero and you'll get WEIRDNESS output. - h_ready_trigger.trigger(); - wait(waitForAll(all)); - - if (wakeup.size() != 0) { - std::cout << format("WEIRDNESS: wakeup.size() != 0 [%d]\n", wakeup.size()); - - for (auto it = wakeup.begin(); it != wakeup.end(); it++) { - std::cout << format(" %d -> %p\n", it->first, it->second); - } - - // wait(delay(30.0)); - } - - all.clear(); - total_h_threads += h_threads; - total_o_threads += o_threads; - h_threads = 0; - o_threads = 0; - // wait(delay(2.0)); - bigloops++; - } - } - - return Void(); -} - -int main(int argc, char** argv) { - // Cargo-culted from tutorial.actor.cpp. - platformInit(); - g_network = newNet2(TLSConfig(), false, true); - - std::vector> all; - - all.emplace_back(orchestrate()); - - auto f = stopAfter(waitForAll(all)); - g_network->run(); - - return 0; -} diff --git a/documentation/tutorial/play.actor.cpp b/documentation/tutorial/play.actor.cpp deleted file mode 100644 index 621cf5c9d61..00000000000 --- a/documentation/tutorial/play.actor.cpp +++ /dev/null @@ -1,55 +0,0 @@ -/* - * play.actor.cpp - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2025 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -#include "fdbclient/NativeAPI.h" -#include "flow/Arena.h" -#include "flow/flow.h" -#include "flow/Platform.h" -#include "flow/TLSConfig.h" - -#include -#include - -#include "flow/actorcompiler.h" - -/// Use this file as means to play with flow code -/// The goal is to give you a starting point with a boilerplate template -/// Don't expect frequent changes to this file unless we want to change the base template -/// The use-case would be for people to have ephemeral code (on top of this template) that never gets checked in - -ACTOR Future foo() { - std::cout << "foo enter" << std::endl; - wait(delay(1)); - std::cout << "foo exit" << std::endl; - return Void(); -} - -int main(int argc, char** argv) { - platformInit(); - g_network = newNet2(TLSConfig(), false, true); - - std::vector> all; - all.emplace_back(foo()); - - auto f = stopAfter(waitForAll(all)); - g_network->run(); - - return 0; -} diff --git a/documentation/tutorial/play_network.actor.cpp b/documentation/tutorial/play_network.actor.cpp deleted file mode 100644 index c6948bcb633..00000000000 --- a/documentation/tutorial/play_network.actor.cpp +++ /dev/null @@ -1,221 +0,0 @@ -/* - * play_network.actor.cpp - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2025 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -#include "fdbclient/NativeAPI.h" -#include "flow/Arena.h" -#include "flow/flow.h" -#include "flow/Platform.h" -#include "flow/TLSConfig.h" - -#include -#include - -#include "flow/actorcompiler.h" - -/// This file has code similar to play.actor.cpp, but for client/server actors talking over a network -/// It's also similar to tutorial.actor.cpp but only has the minimal code needed for testing client/server actor -/// interactions over the network. -/// Use this file as means to play with flow code and test network behavior -/// The goal is to give you a starting point with a boilerplate template -/// Don't expect frequent changes to this file unless we want to change the base template -/// The use-case would be for people to have ephemeral code (on top of this template) that never gets checked in -/// An example use-case is in the description of https://github.com/apple/foundationdb/pull/12484 - -/// Usage below - -/// Start server: -/// ~/c/bin $ ./play_network -s 6666 serverActor - -/// Then client: -/// ~/c/bin $ ./play_network -c 127.0.0.1:6666 clientActor -/// req msg: Hello World -/// rsp msg: dlroW olleH - -/// Now server prints: -/// ~/c/bin $ ./play_network -s 6666 serverActor -/// got interface request from client -/// sent interface back to client -/// got play request with msg: Hello World -/// sending this play response back: dlroW olleH -/// sent this play response back: dlroW olleH - -NetworkAddress serverAddress; - -enum TutorialWellKnownEndpoints { WLTOKEN_PLAY_SERVER = WLTOKEN_FIRST_AVAILABLE, WLTOKEN_COUNT_IN_TUTORIAL }; - -struct PlayServerInterface { - constexpr static FileIdentifier file_identifier = 3152015; - RequestStream getInterface; - RequestStream play; - - template - void serialize(Ar& ar) { - serializer(ar, getInterface, play); - } -}; - -struct GetInterfaceRequest { - constexpr static FileIdentifier file_identifier = 12004156; - ReplyPromise reply; - - template - void serialize(Ar& ar) { - serializer(ar, reply); - } -}; - -struct PlayRequest { - constexpr static FileIdentifier file_identifier = 10624019; - std::string msg; - // For now, returns a reverse of msg - ReplyPromise reply; - - template - void serialize(Ar& ar) { - serializer(ar, msg, reply); - } -}; - -ACTOR Future server() { - // Setup - state PlayServerInterface playServer; - playServer.getInterface.makeWellKnownEndpoint(WLTOKEN_PLAY_SERVER, TaskPriority::DefaultEndpoint); - - // Main server - loop { - try { - choose { - when(GetInterfaceRequest req = waitNext(playServer.getInterface.getFuture())) { - std::cout << "got interface request from client" << std::endl; - req.reply.send(playServer); - std::cout << "sent interface back to client" << std::endl; - } - when(PlayRequest req = waitNext(playServer.play.getFuture())) { - std::cout << "got play request with msg: " << req.msg << std::endl; - std::string rsp(req.msg.rbegin(), req.msg.rend()); - std::cout << "sending this play response back: " << rsp << std::endl; - req.reply.send(rsp); - std::cout << "sent this play response back: " << rsp << std::endl; - } - } - } catch (Error& e) { - if (e.code() != error_code_operation_obsolete) { - fprintf(stderr, "Error: %s\n", e.what()); - throw e; - } - } - } -} - -ACTOR Future client() { - // Setup - state PlayServerInterface server; - server.getInterface = - RequestStream(Endpoint::wellKnown({ serverAddress }, WLTOKEN_PLAY_SERVER)); - PlayServerInterface s = wait(server.getInterface.getReply(GetInterfaceRequest())); - server = s; - - // Create and print req - PlayRequest playReq; - playReq.msg = "Hello World"; - std::cout << "req msg: " << playReq.msg << std::endl; - - // Send req and wait for rsp - std::string playRsp = wait(server.play.getReply(playReq)); - - // Print rsp - std::cout << "rsp msg: " << playRsp << std::endl; - - return Void(); -} - -std::unordered_map()>> actors = { - { "serverActor", &server }, // ./play_network -s 6666 serverActor - { "clientActor", &client }, // ./play_network -c 127.0.0.1:6666 clientActor -}; - -int main(int argc, char* argv[]) { - bool isServer = false; - std::string port; - std::vector()>> toRun; - - // parse arguments - for (int i = 1; i < argc; ++i) { - std::string arg(argv[i]); - if (arg == "-s") { - isServer = true; - if (i + 1 >= argc) { - std::cout << "Expecting an argument after -p\n"; - return 1; - } - port = std::string(argv[++i]); - continue; - } else if (arg == "-c") { - if (i + 1 >= argc) { - std::cout << "Expecting an argument after -s\n"; - return 1; - } - serverAddress = NetworkAddress::parse(argv[++i]); - continue; - } else { - assert(false); - } - - auto actor = actors.find(arg); - if (actor == actors.end()) { - std::cout << format("Error: actor %s does not exist\n", arg.c_str()); - return 1; - } - toRun.push_back(actor->second); - } - - // platform init - platformInit(); - g_network = newNet2(TLSConfig(), false, true); - FlowTransport::createInstance(!isServer, 0, WLTOKEN_COUNT_IN_TUTORIAL); - NetworkAddress publicAddress = NetworkAddress::parse("0.0.0.0:0"); - if (isServer) { - publicAddress = NetworkAddress::parse("0.0.0.0:" + port); - } - - try { - if (isServer) { - auto listenError = FlowTransport::transport().bind(publicAddress, publicAddress); - if (listenError.isError()) { - listenError.get(); - } - } - } catch (Error& e) { - std::cout << format("Error while binding to address (%d): %s\n", e.code(), e.what()); - } - - // now we start the actors - std::vector> all; - all.reserve(toRun.size()); - for (auto& f : toRun) { - all.emplace_back(f()); - } - - auto f = stopAfter(waitForAll(all)); - - g_network->run(); - - return 0; -} diff --git a/documentation/tutorial/print_in_order.actor.cpp b/documentation/tutorial/print_in_order.actor.cpp deleted file mode 100644 index 50bea24d196..00000000000 --- a/documentation/tutorial/print_in_order.actor.cpp +++ /dev/null @@ -1,95 +0,0 @@ -/* - * print_in_order.actor.cpp - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2025 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -#include "fmt/format.h" -#include "flow/flow.h" -#include "flow/Platform.h" -#include "flow/DeterministicRandom.h" -#include "fdbclient/NativeAPI.h" -#include "fdbclient/ReadYourWrites.h" -#include "flow/TLSConfig.h" - -#include -#include -#include -#include -#include - -#include "flow/actorcompiler.h" - -// Solution to https://leetcode.com/problems/print-in-order/description/ -// -// This is a super basic concurrency exercise useful as a day 1 -// exercise in a new environment. To try this yourself, delete the -// next two functions, then write a solution from scratch. - -ACTOR Future print_msg_when_ready(Future ready, std::string msg) { - int delay_msec = deterministicRandom()->randomInt(0, 1000); - double delay_sec = static_cast(delay_msec) / 1000.0; - wait(delay(delay_sec)); - - wait(ready); - std::cout << msg << std::endl; - wait(delay(0.1)); - return Void(); -} - -ACTOR Future orchestrate() { - state Promise p_first, p_second, p_third; - state Future first_ready = p_first.getFuture(); - state Future second_ready = p_second.getFuture(); - state Future third_ready = p_third.getFuture(); - - state Future first = print_msg_when_ready(first_ready, "First"); - state Future second = print_msg_when_ready(second_ready, "Second"); - state Future third = print_msg_when_ready(third_ready, "Third"); - - // If we do the following, the order of output varies from run to run based on - // random seed chosen. This is expected. - // p_first.send(0); - // p_second.send(0); - // p_third.send(0); - - // So what we have to do is signal in order and wait before signaling the - // next. - p_first.send(Void()); - wait(first); - p_second.send(Void()); - wait(second); - p_third.send(Void()); - wait(third); - - return Void(); -} - -int main(int argc, char** argv) { - // Cargo-culted from tutorial.actor.cpp. - platformInit(); - g_network = newNet2(TLSConfig(), false, true); - - std::vector> all; - - all.emplace_back(orchestrate()); - - auto f = stopAfter(waitForAll(all)); - g_network->run(); - - return 0; -} diff --git a/documentation/tutorial/tutorial.actor.cpp b/documentation/tutorial/tutorial.actor.cpp deleted file mode 100644 index 5111bfdc636..00000000000 --- a/documentation/tutorial/tutorial.actor.cpp +++ /dev/null @@ -1,597 +0,0 @@ -/* -* tutorial.actor.cpp - -* -* This source file is part of the FoundationDB open source project -* -* Copyright 2013-2026 Apple Inc. and the FoundationDB project authors -* -* Licensed under the Apache License, Version 2.0 (the "License"); -* you may not use this file except in compliance with the License. -* You may obtain a copy of the License at -* -* http://www.apache.org/licenses/LICENSE-2.0 -* -* Unless required by applicable law or agreed to in writing, software -* distributed under the License is distributed on an "AS IS" BASIS, -* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -* See the License for the specific language governing permissions and -* limitations under the License. -*/ - -#include "fmt/format.h" -#include "flow/flow.h" -#include "flow/Platform.h" -#include "flow/DeterministicRandom.h" -#include "fdbclient/NativeAPI.h" -#include "fdbclient/ReadYourWrites.h" -#include "flow/TLSConfig.h" -#include -#include -#include -#include -#include "flow/actorcompiler.h" - -NetworkAddress serverAddress; - -enum TutorialWellKnownEndpoints { - WLTOKEN_SIMPLE_KV_SERVER = WLTOKEN_FIRST_AVAILABLE, - WLTOKEN_ECHO_SERVER, - WLTOKEN_COUNT_IN_TUTORIAL -}; - -// this is a simple actor that will report how long -// it is already running once a second. -ACTOR Future simpleTimer() { - // we need to remember the time when we first - // started. - // This needs to be a state-variable because - // we will use it in different parts of the - // actor. If you don't understand how state - // variables work, it is a good idea to remove - // the state keyword here and look at the - // generated C++ code from the actor compiler. - state double start_time = g_network->now(); - loop { - wait(delay(1.0)); - std::cout << format("Time: %.2f\n", g_network->now() - start_time); - } -} - -// A actor that demonstrates how choose-when blocks work. -ACTOR Future someFuture(Future ready) { - // loop choose {} works as well here - the braces are optional - loop choose { - when(wait(delay(0.5))) { - std::cout << "Still waiting...\n"; - } - when(int r = wait(ready)) { - std::cout << format("Ready %d\n", r); - wait(delay(double(r))); - std::cout << "Done\n"; - return Void(); - } - } -} - -ACTOR Future promiseDemo() { - state Promise promise; - state Future f = someFuture(promise.getFuture()); - wait(delay(3.0)); - promise.send(2); - wait(f); - return Void(); -} - -ACTOR Future eventLoop(AsyncTrigger* trigger) { - loop choose { - when(wait(delay(0.5))) { - std::cout << "Still waiting...\n"; - } - when(wait(trigger->onTrigger())) { - std::cout << "Triggered!\n"; - } - } -} - -ACTOR Future triggerDemo() { - state int runs = 1; - state AsyncTrigger trigger; - state Future triggerLoop = eventLoop(&trigger); - while (++runs < 10) { - wait(delay(1.0)); - std::cout << "trigger.."; - trigger.trigger(); - } - std::cout << "Done."; - return Void(); -} - -struct EchoServerInterface { - constexpr static FileIdentifier file_identifier = 3152015; - RequestStream getInterface; - RequestStream echo; - RequestStream reverse; - RequestStream stream; - - template - void serialize(Ar& ar) { - serializer(ar, getInterface, echo, reverse, stream); - } -}; - -struct GetInterfaceRequest { - constexpr static FileIdentifier file_identifier = 12004156; - ReplyPromise reply; - - template - void serialize(Ar& ar) { - serializer(ar, reply); - } -}; - -struct EchoRequest { - constexpr static FileIdentifier file_identifier = 10624019; - std::string message; - - // NOTES: - // 1) This variable has to be called reply! - // 2) Do not be mislead into thinking that just because we say - // ReplyPromise below that you can also say - // ReplyPromise elsewhere. - // 3) There is no good documentation about this. You'll have to - // just read the code or ask your local team. - ReplyPromise reply; - - template - void serialize(Ar& ar) { - serializer(ar, message, reply); - } -}; - -struct ReverseRequest { - constexpr static FileIdentifier file_identifier = 10765955; - std::string message; - // this variable has to be called reply! - ReplyPromise reply; - - template - void serialize(Ar& ar) { - serializer(ar, message, reply); - } -}; - -struct StreamReply : ReplyPromiseStreamReply { - constexpr static FileIdentifier file_identifier = 440804; - - int index = 0; - StreamReply() = default; - explicit StreamReply(int index) : index(index) {} - - size_t expectedSize() const { return 2e6; } - - template - void serialize(Ar& ar) { - serializer(ar, ReplyPromiseStreamReply::acknowledgeToken, ReplyPromiseStreamReply::sequence, index); - } -}; - -struct StreamRequest { - constexpr static FileIdentifier file_identifier = 5410805; - ReplyPromiseStream reply; - - template - void serialize(Ar& ar) { - serializer(ar, reply); - } -}; - -uint64_t tokenCounter = 1; - -ACTOR Future echoServer() { - state EchoServerInterface echoServer; - echoServer.getInterface.makeWellKnownEndpoint(WLTOKEN_ECHO_SERVER, TaskPriority::DefaultEndpoint); - loop { - try { - choose { - when(GetInterfaceRequest req = waitNext(echoServer.getInterface.getFuture())) { - req.reply.send(echoServer); - } - when(EchoRequest req = waitNext(echoServer.echo.getFuture())) { - req.reply.send(req.message); - } - when(ReverseRequest req = waitNext(echoServer.reverse.getFuture())) { - req.reply.send(std::string(req.message.rbegin(), req.message.rend())); - } - when(state StreamRequest req = waitNext(echoServer.stream.getFuture())) { - req.reply.setByteLimit(1024); - state int i = 0; - for (; i < 100; ++i) { - wait(req.reply.onReady()); - std::cout << "Send " << i << std::endl; - req.reply.send(StreamReply{ i }); - } - req.reply.sendError(end_of_stream()); - } - } - } catch (Error& e) { - if (e.code() != error_code_operation_obsolete) { - fprintf(stderr, "Error: %s\n", e.what()); - throw e; - } - } - } -} - -ACTOR Future echoClient() { - state EchoServerInterface server; - server.getInterface = - RequestStream(Endpoint::wellKnown({ serverAddress }, WLTOKEN_ECHO_SERVER)); - EchoServerInterface s = wait(server.getInterface.getReply(GetInterfaceRequest())); - server = s; - EchoRequest echoRequest; - echoRequest.message = "Hello World"; - std::string echoMessage = wait(server.echo.getReply(echoRequest)); - std::cout << format("Sent %s to echo, received %s\n", "Hello World", echoMessage.c_str()); - ReverseRequest reverseRequest; - reverseRequest.message = "Hello World"; - std::string reverseString = wait(server.reverse.getReply(reverseRequest)); - std::cout << format("Sent %s to reverse, received %s\n", "Hello World", reverseString.c_str()); - - state ReplyPromiseStream stream = server.stream.getReplyStream(StreamRequest{}); - state int j = 0; - try { - loop { - StreamReply rep = waitNext(stream.getFuture()); - std::cout << "Rep: " << rep.index << std::endl; - ASSERT(rep.index == j++); - } - } catch (Error& e) { - ASSERT(e.code() == error_code_end_of_stream || e.code() == error_code_connection_failed); - } - return Void(); -} - -struct SimpleKeyValueStoreInterface { - constexpr static FileIdentifier file_identifier = 8226647; - RequestStream connect; - RequestStream get; - RequestStream set; - RequestStream clear; - - template - void serialize(Ar& ar) { - serializer(ar, connect, get, set, clear); - } -}; - -struct GetKVInterface { - constexpr static FileIdentifier file_identifier = 8062308; - ReplyPromise reply; - - template - void serialize(Ar& ar) { - serializer(ar, reply); - } -}; - -struct GetRequest { - constexpr static FileIdentifier file_identifier = 6983506; - std::string key; - ReplyPromise reply; - - template - void serialize(Ar& ar) { - serializer(ar, key, reply); - } -}; - -struct SetRequest { - constexpr static FileIdentifier file_identifier = 7554186; - std::string key; - std::string value; - ReplyPromise reply; - - template - void serialize(Ar& ar) { - serializer(ar, key, value, reply); - } -}; - -struct ClearRequest { - constexpr static FileIdentifier file_identifier = 8500026; - std::string from; - std::string to; - ReplyPromise reply; - - template - void serialize(Ar& ar) { - serializer(ar, from, to, reply); - } -}; - -ACTOR Future kvStoreServer() { - state SimpleKeyValueStoreInterface inf; - state std::map store; - inf.connect.makeWellKnownEndpoint(WLTOKEN_SIMPLE_KV_SERVER, TaskPriority::DefaultEndpoint); - loop { - choose { - when(GetKVInterface req = waitNext(inf.connect.getFuture())) { - std::cout << "Received connection attempt\n"; - req.reply.send(inf); - } - when(GetRequest req = waitNext(inf.get.getFuture())) { - auto iter = store.find(req.key); - if (iter == store.end()) { - req.reply.sendError(io_error()); - } else { - req.reply.send(iter->second); - } - } - when(SetRequest req = waitNext(inf.set.getFuture())) { - store[req.key] = req.value; - req.reply.send(Void()); - } - when(ClearRequest req = waitNext(inf.clear.getFuture())) { - auto from = store.lower_bound(req.from); - auto to = store.lower_bound(req.to); - while (from != store.end() && from != to) { - auto next = from; - ++next; - store.erase(from); - from = next; - } - req.reply.send(Void()); - } - } - } -} - -ACTOR Future connect() { - std::cout << format("%llu: Connect...\n", uint64_t(g_network->now())); - auto reqStream = RequestStream(Endpoint::wellKnown({ serverAddress }, WLTOKEN_SIMPLE_KV_SERVER)); - SimpleKeyValueStoreInterface result = wait(reqStream.getReply(GetKVInterface())); - std::cout << format("%llu: done..\n", uint64_t(g_network->now())); - return result; -} - -ACTOR Future kvSimpleClient() { - state SimpleKeyValueStoreInterface server = wait(connect()); - std::cout << format("Set %s -> %s\n", "foo", "bar"); - SetRequest setRequest; - setRequest.key = "foo"; - setRequest.value = "bar"; - wait(server.set.getReply(setRequest)); - GetRequest getRequest; - getRequest.key = "foo"; - std::string value = wait(server.get.getReply(getRequest)); - std::cout << format("get(%s) -> %s\n", "foo", value.c_str()); - return Void(); -} - -ACTOR Future kvClient(SimpleKeyValueStoreInterface server, std::shared_ptr ops) { - state Future timeout = delay(20); - state int rangeSize = 2 << 12; - loop { - SetRequest setRequest; - setRequest.key = std::to_string(deterministicRandom()->randomInt(0, rangeSize)); - setRequest.value = "foo"; - wait(server.set.getReply(setRequest)); - ++(*ops); - try { - GetRequest getRequest; - getRequest.key = std::to_string(deterministicRandom()->randomInt(0, rangeSize)); - std::string _ = wait(server.get.getReply(getRequest)); - ++(*ops); - } catch (Error& e) { - if (e.code() != error_code_io_error) { - throw e; - } - } - int from = deterministicRandom()->randomInt(0, rangeSize); - ClearRequest clearRequest; - clearRequest.from = std::to_string(from); - clearRequest.to = std::to_string(from + 100); - wait(server.clear.getReply(clearRequest)); - ++(*ops); - if (timeout.isReady()) { - // we are done - return Void(); - } - } -} - -ACTOR Future throughputMeasurement(std::shared_ptr operations) { - loop { - wait(delay(1.0)); - std::cout << format("%llu op/s\n", *operations); - *operations = 0; - } -} - -ACTOR Future multipleClients() { - SimpleKeyValueStoreInterface server = wait(connect()); - auto ops = std::make_shared(0); - std::vector> clients(100); - for (auto& f : clients) { - f = kvClient(server, ops); - } - auto done = waitForAll(clients); - wait(done || throughputMeasurement(ops)); - return Void(); -} - -std::string clusterFile = "fdb.cluster"; - -ACTOR Future logThroughput(int64_t* v, Key* next) { - loop { - state int64_t last = *v; - wait(delay(1)); - fmt::print("throughput: {} bytes/s, next: {}\n", *v - last, printable(*next).c_str()); - } -} - -ACTOR Future fdbClientStream() { - state Database db = Database::createDatabase(clusterFile, 300); - state Transaction tx(db); - state Key next; - state int64_t bytes = 0; - state Future logFuture = logThroughput(&bytes, &next); - loop { - state PromiseStream> results; - try { - state Future stream = tx.getRangeStream(results, - KeySelector(firstGreaterOrEqual(next), next.arena()), - KeySelector(firstGreaterOrEqual(normalKeys.end)), - GetRangeLimits()); - loop { - Standalone range = waitNext(results.getFuture()); - if (range.size()) { - bytes += range.expectedSize(); - next = keyAfter(range.back().key); - } - } - } catch (Error& e) { - if (e.code() == error_code_end_of_stream) { - break; - } - wait(tx.onError(e)); - } - } - return Void(); -} - -ACTOR Future fdbClientGetRange() { - state Database db = Database::createDatabase(clusterFile, 300); - state Transaction tx(db); - state Key next; - state int64_t bytes = 0; - state Future logFuture = logThroughput(&bytes, &next); - loop { - try { - Standalone range = - wait(tx.getRange(KeySelector(firstGreaterOrEqual(next), next.arena()), - KeySelector(firstGreaterOrEqual(normalKeys.end)), - GetRangeLimits(GetRangeLimits::ROW_LIMIT_UNLIMITED, CLIENT_KNOBS->REPLY_BYTE_LIMIT))); - bytes += range.expectedSize(); - if (!range.more) { - break; - } - next = keyAfter(range.back().key); - } catch (Error& e) { - wait(tx.onError(e)); - } - } - return Void(); -} - -ACTOR Future fdbClient() { - wait(delay(30)); - state Database db = Database::createDatabase(clusterFile, 300); - state Transaction tx(db); - state std::string keyPrefix = "/tut/"; - state Key startKey; - state KeyRef endKey = "/tut0"_sr; - state int beginIdx = 0; - loop { - try { - tx.reset(); - // this workload is stupidly simple: - // 1. select a random key between 1 - // and 1e8 - // 2. select this key plus the 100 - // next ones - // 3. write 10 values in [k, k+100] - beginIdx = deterministicRandom()->randomInt(0, 1e8 - 100); - startKey = keyPrefix + std::to_string(beginIdx); - RangeResult range = wait(tx.getRange(KeyRangeRef(startKey, endKey), 100)); - for (int i = 0; i < 10; ++i) { - Key k = Key(keyPrefix + std::to_string(beginIdx + deterministicRandom()->randomInt(0, 100))); - tx.set(k, "foo"_sr); - } - wait(tx.commit()); - std::cout << "Committed\n"; - wait(delay(2.0)); - } catch (Error& e) { - wait(tx.onError(e)); - } - } -} - -std::unordered_map()>> actors = { - { "timer", &simpleTimer }, // ./tutorial timer - { "promiseDemo", &promiseDemo }, // ./tutorial promiseDemo - { "triggerDemo", &triggerDemo }, // ./tutorial triggerDemo - { "echoServer", &echoServer }, // ./tutorial -p 6666 echoServer - { "echoClient", &echoClient }, // ./tutorial -s 127.0.0.1:6666 echoClient - { "kvStoreServer", &kvStoreServer }, // ./tutorial -p 6666 kvStoreServer - { "kvSimpleClient", &kvSimpleClient }, // ./tutorial -s 127.0.0.1:6666 kvSimpleClient - { "multipleClients", &multipleClients }, // ./tutorial -s 127.0.0.1:6666 multipleClients - { "fdbClientStream", &fdbClientStream }, // ./tutorial -C $CLUSTER_FILE_PATH fdbClientStream - { "fdbClientGetRange", &fdbClientGetRange }, // ./tutorial -C $CLUSTER_FILE_PATH fdbClientGetRange - { "fdbClient", &fdbClient } // ./tutorial -C $CLUSTER_FILE_PATH fdbClient -}; - -int main(int argc, char* argv[]) { - bool isServer = false; - std::string port; - std::vector()>> toRun; - // parse arguments - for (int i = 1; i < argc; ++i) { - std::string arg(argv[i]); - if (arg == "-p") { - isServer = true; - if (i + 1 >= argc) { - std::cout << "Expecting an argument after -p\n"; - return 1; - } - port = std::string(argv[++i]); - continue; - } else if (arg == "-s") { - if (i + 1 >= argc) { - std::cout << "Expecting an argument after -s\n"; - return 1; - } - serverAddress = NetworkAddress::parse(argv[++i]); - continue; - } else if (arg == "-C") { - clusterFile = argv[++i]; - std::cout << "Using cluster file " << clusterFile << std::endl; - continue; - } - auto actor = actors.find(arg); - if (actor == actors.end()) { - std::cout << format("Error: actor %s does not exist\n", arg.c_str()); - return 1; - } - toRun.push_back(actor->second); - } - platformInit(); - g_network = newNet2(TLSConfig(), false, true); - FlowTransport::createInstance(!isServer, 0, WLTOKEN_COUNT_IN_TUTORIAL); - NetworkAddress publicAddress = NetworkAddress::parse("0.0.0.0:0"); - if (isServer) { - publicAddress = NetworkAddress::parse("0.0.0.0:" + port); - } - // openTraceFile(publicAddress, TRACE_DEFAULT_ROLL_SIZE, - // TRACE_DEFAULT_MAX_LOGS_SIZE); - try { - if (isServer) { - auto listenError = FlowTransport::transport().bind(publicAddress, publicAddress); - if (listenError.isError()) { - listenError.get(); - } - } - } catch (Error& e) { - std::cout << format("Error while binding to address (%d): %s\n", e.code(), e.what()); - } - // now we start the actors - std::vector> all; - all.reserve(toRun.size()); - for (auto& f : toRun) { - all.emplace_back(f()); - } - auto f = stopAfter(waitForAll(all)); - g_network->run(); - return 0; -} diff --git a/fdbbackup/FileConverter.cpp b/fdbbackup/FileConverter.cpp index d95fe9c926f..2a2052c08b3 100644 --- a/fdbbackup/FileConverter.cpp +++ b/fdbbackup/FileConverter.cpp @@ -308,6 +308,7 @@ struct MutationFilesReadProgress : public ReferenceCounted openLogFilesImpl(MutationFilesReadProgress* progress, Reference container) { std::vector>> asyncFiles; + asyncFiles.reserve(progress->files.size()); for (const auto& file : progress->files) { asyncFiles.push_back(container->readFile(file.fileName)); } diff --git a/fdbbackup/backup.cpp b/fdbbackup/backup.cpp index 3d2d24eea96..14b8d8693a2 100644 --- a/fdbbackup/backup.cpp +++ b/fdbbackup/backup.cpp @@ -4526,6 +4526,7 @@ int main() { int argc = static_cast(persistentArgs.size()); std::vector argv; + argv.reserve(persistentArgs.size() + 1); for (auto& arg : persistentArgs) { argv.push_back(arg.data()); } diff --git a/fdbcli/LocationMetadataCommand.cpp b/fdbcli/LocationMetadataCommand.cpp index e6e0c7ef443..0607b4e1ca6 100644 --- a/fdbcli/LocationMetadataCommand.cpp +++ b/fdbcli/LocationMetadataCommand.cpp @@ -29,6 +29,7 @@ namespace { Future describeServers(Reference tr, std::vector ids) { std::vector>> serverListEntries; + serverListEntries.reserve(ids.size()); for (const UID& id : ids) { serverListEntries.push_back(tr->get(serverListKeyFor(id))); } diff --git a/fdbcli/RangeLockCommand.cpp b/fdbcli/RangeLockCommand.cpp index 0db65611143..bd23e1fd1fc 100644 --- a/fdbcli/RangeLockCommand.cpp +++ b/fdbcli/RangeLockCommand.cpp @@ -137,6 +137,11 @@ Future rangeLockCommandActor(Database cx, std::vector tokens) { if (e.code() == error_code_actor_cancelled) { throw; } + if (e.code() == error_code_range_lock_reject) { + fmt::println("ERROR: cannot unregister owner: owner still holds range locks; " + "stop acquisitions, release the locks, and retry"); + co_return false; + } co_return reportRangeLockError(e, "unregister owner"); } fmt::println("Unregistered range lock owner: {}", ownerId); diff --git a/fdbcli/StatusCommand.cpp b/fdbcli/StatusCommand.cpp index f43d5e6e4f4..1cff7ae0213 100644 --- a/fdbcli/StatusCommand.cpp +++ b/fdbcli/StatusCommand.cpp @@ -714,10 +714,20 @@ void printStatus(StatusObjectReader statusObj, ASSERT_WE_THINK(availLoss == -1); const bool possiblyLosingData = logEpochsMayBeLosingData(statusObjCluster); if (possiblyLosingData) { - outputString += format( - "\n\n Warning: the database may have data loss and availability loss. Please " - "restart following tlog interfaces, otherwise storage servers may never be able " - "to catch up.\n"); + const std::string baseMessage = + "Please restart following tlog interfaces, otherwise storage servers " + "may never be able to catch up.\n"; + + bool degradedMultiRegion = false; + statusObjCluster.get("degraded_multi_region", degradedMultiRegion); + + const std::string header = + !degradedMultiRegion + ? "\n\n Warning: the database may have data loss and availability loss. " + : "\n\n Warning: one region is unavailable; committed data remains safe in " + "the surviving region. "; + + outputString += header + baseMessage; } else { outputString += format( "\n\n Warning: the database may have availability loss. The current log state " diff --git a/fdbclient/BackupContainer.cpp b/fdbclient/BackupContainer.cpp index e79eeccc9a9..515cac35f2d 100644 --- a/fdbclient/BackupContainer.cpp +++ b/fdbclient/BackupContainer.cpp @@ -25,9 +25,6 @@ #include "flow/Arena.h" #include "flow/Trace.h" #include "flow/Platform.h" -#ifdef BUILD_AZURE_BACKUP -#include "fdbclient/BackupContainerAzureBlobStore.h" -#endif #include "BackupContainerLocalDirectory.h" #include "BackupContainerBlobStore.h" #include "fdbclient/SystemData.h" @@ -259,9 +256,6 @@ std::string IBackupContainer::lastOpenError; std::vector IBackupContainer::getURLFormats() { return { -#ifdef BUILD_AZURE_BACKUP - BackupContainerAzureBlobStore::getURLFormat(), -#endif BackupContainerLocalDirectory::getURLFormat(), BackupContainerBlobStore::getURLFormat(), }; @@ -321,53 +315,7 @@ Reference IBackupContainer::openContainer(const std::string& u BackupContainerBlobStore::validateBackupUrl(resource); r = makeReference( bstore, resource, backupParams, encryptionKeyFileName, encryptionBlockSize, /*isBackup=*/true); - } -#ifdef BUILD_AZURE_BACKUP - else if (u.startsWith("azure://"_sr)) { - u.eat("azure://"_sr); - auto address = u.eat("/"_sr); - if (address.endsWith(std::string(azure::storage_lite::constants::default_endpoint_suffix))) { - CODE_PROBE(true, "Azure backup url with standard azure storage account endpoint"); - // ..core.windows.net/ - auto endPoint = address.toString(); - auto accountName = address.eat("."_sr).toString(); - auto containerName = u.eat("/"_sr).toString(); - r = makeReference( - endPoint, accountName, containerName, encryptionKeyFileName); - } else { - // resolve the network address if necessary - std::string endpoint(address.toString()); - Optional parsedAddress = NetworkAddress::parseOptional(endpoint); - if (!parsedAddress.present()) { - try { - auto hostname = Hostname::parse(endpoint); - auto resolvedAddress = hostname.resolveBlocking(); - if (resolvedAddress.present()) { - CODE_PROBE(true, "Azure backup url with hostname in the endpoint"); - parsedAddress = resolvedAddress.get(); - } - } catch (Error& e) { - TraceEvent(SevError, "InvalidAzureBackupUrl").error(e).detail("Endpoint", endpoint); - throw backup_invalid_url(); - } - } - if (!parsedAddress.present()) { - TraceEvent(SevError, "InvalidAzureBackupUrl").detail("Endpoint", endpoint); - throw backup_invalid_url(); - } - auto accountName = u.eat("/"_sr).toString(); - // Avoid including ":tls" and "(fromHostname)" - // note: the endpoint needs to contain the account name - // so either ".blob.core.windows.net" or ":/" - endpoint = - fmt::format("{}/{}", formatIpPort(parsedAddress.get().ip, parsedAddress.get().port), accountName); - auto containerName = u.eat("/"_sr).toString(); - r = makeReference( - endpoint, accountName, containerName, encryptionKeyFileName); - } - } -#endif - else { + } else { lastOpenError = "invalid URL prefix"; throw backup_invalid_url(); } @@ -422,15 +370,7 @@ Future> listContainers_impl(std::string baseURL, Option std::vector results = co_await BackupContainerBlobStore::listURLs(bstore, dummy.getBucket()); co_return results; - } - // TODO: Enable this when Azure backups are ready - /* - else if (u.startsWith("azure://"_sr)) { - std::vector results = wait(BackupContainerAzureBlobStore::listURLs(baseURL)); - return results; - } - */ - else { + } else { IBackupContainer::lastOpenError = "invalid URL prefix"; throw backup_invalid_url(); } diff --git a/fdbclient/BackupContainerFileSystem.cpp b/fdbclient/BackupContainerFileSystem.cpp index fe7a1dd3071..eec970ef7c9 100644 --- a/fdbclient/BackupContainerFileSystem.cpp +++ b/fdbclient/BackupContainerFileSystem.cpp @@ -22,9 +22,6 @@ #include "fdbclient/BackupFileFormat.h" #include "fdbclient/BackupContainer.h" #include "flow/BooleanParam.h" -#ifdef BUILD_AZURE_BACKUP -#include "fdbclient/BackupContainerAzureBlobStore.h" -#endif #include "fdbclient/BackupContainerFileSystem.h" #include "BackupContainerLocalDirectory.h" #include "BackupContainerBlobStore.h" diff --git a/fdbclient/CMakeLists.txt b/fdbclient/CMakeLists.txt index 1444da83990..66178067e95 100644 --- a/fdbclient/CMakeLists.txt +++ b/fdbclient/CMakeLists.txt @@ -18,7 +18,7 @@ set(FDBOPTIONS_GEN_H ${FDBOPTIONS_GEN_BASE}.h) set(FDBOPTIONS_GEN_CPP ${FDBOPTIONS_GEN_BASE}.cpp) # Generate the C++ option bindings directly into the final include directory so -# the header and source exist before any staging or actor compilation. +# the header and source exist before their consumers compile. add_custom_command( OUTPUT ${FDBOPTIONS_GEN_H} ${FDBOPTIONS_GEN_CPP} COMMAND ${VEXILLOGRAPHER_COMMAND} @@ -53,37 +53,8 @@ configure_file(${CMAKE_CURRENT_SOURCE_DIR}/BuildFlags.h.in ${CMAKE_CURRENT_BINAR set(BUILD_AZURE_BACKUP OFF CACHE BOOL "Build Azure backup client") if(BUILD_AZURE_BACKUP) - add_compile_definitions(BUILD_AZURE_BACKUP) - set(FDBCLIENT_SRCS - ${FDBCLIENT_SRCS} - azure_backup/BackupContainerAzureBlobStore.actor.cpp) - - configure_file(azurestorage.cmake azurestorage-download/CMakeLists.txt) - - execute_process( - COMMAND ${CMAKE_COMMAND} -G "${CMAKE_GENERATOR}" . - RESULT_VARIABLE results - WORKING_DIRECTORY ${CMAKE_CURRENT_BINARY_DIR}/azurestorage-download - ) - - if(results) - message(FATAL_ERROR "Configuration step for AzureStorage has Failed. ${results}") - endif() - - execute_process( - COMMAND ${CMAKE_COMMAND} --build . --config Release - RESULT_VARIABLE results - WORKING_DIRECTORY ${CMAKE_CURRENT_BINARY_DIR}/azurestorage-download - ) - - if(results) - message(FATAL_ERROR "Build step for AzureStorage has Failed. ${results}") - endif() - - add_subdirectory( - ${CMAKE_CURRENT_BINARY_DIR}/azurestorage-src - ${CMAKE_CURRENT_BINARY_DIR}/azurestorage-build - ) + message(FATAL_ERROR + "BUILD_AZURE_BACKUP is unsupported: this source tree does not contain the Azure backup implementation.") endif() @@ -168,9 +139,9 @@ if(WITH_SWIFT) AsyncFileBlobStore.h RYWIterator.h SnapshotCache.h WriteMap.h) # TODO: the TBD validation skip is because of swift_job_run_generic, though it seems weird why we need to do that? - target_compile_options(fdbclient_swift PRIVATE "$<$:SHELL:-Xcc -std=c++20 -Xfrontend -validate-tbd-against-ir=none -Xcc -DNO_INTELLISENSE -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/flow/include/headeroverlay.yaml -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/fdbclient/include/headeroverlay.yaml>") + target_compile_options(fdbclient_swift PRIVATE "$<$:SHELL:-Xcc -std=c++20 -Xfrontend -validate-tbd-against-ir=none -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/flow/include/headeroverlay.yaml -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/fdbclient/include/headeroverlay.yaml>") - add_dependencies(fdbclient_swift flow_swift fdbclient_actors fdbrpc_actors fdboptions) + add_dependencies(fdbclient_swift flow_swift fdboptions) # This does not work! (see rdar://99107402) # target_link_libraries(flow PRIVATE flow_swift) add_dependencies(fdbclient fdbclient_swift) @@ -194,9 +165,6 @@ target_include_directories(fdbclient_sampling add_dependencies(fdbclient_sampling fdboptions) target_link_libraries(fdbclient_sampling PUBLIC fdbrpc_sampling msgpack PRIVATE rapidxml) target_compile_definitions(fdbclient_sampling PRIVATE -DENABLE_SAMPLING) -if(WIN32) - add_dependencies(fdbclient_sampling_actors fdbclient_actors) -endif() add_flow_target(LINK_TEST NAME fdbclientlinktest SRCS LinkTest.cpp) target_link_libraries(fdbclientlinktest PRIVATE "$" rapidxml) # re-link rapidxml due to private link interface @@ -210,11 +178,6 @@ if(NOT FOUNDATIONDB_CROSS_COMPILING) # FIXME(swift): make this work when add_subdirectory(bench EXCLUDE_FROM_ALL) endif() -if(BUILD_AZURE_BACKUP) - target_link_libraries(fdbclient PRIVATE curl azure-storage-lite) - target_link_libraries(fdbclient_sampling PRIVATE curl azure-storage-lite) -endif() - if(WITH_AWS_BACKUP) target_link_libraries(fdbclient PRIVATE awssdk_target) target_link_libraries(fdbclient_sampling PRIVATE awssdk_target) diff --git a/fdbclient/FileBackupAgent.cpp b/fdbclient/FileBackupAgent.cpp index bdf010858e2..d360d33686b 100644 --- a/fdbclient/FileBackupAgent.cpp +++ b/fdbclient/FileBackupAgent.cpp @@ -71,9 +71,21 @@ std::atomic g_bulkDumpTaskCompleteCount(0); std::atomic g_bulkLoadRestoreTaskCompleteCount(0); // Helper function to monitor BulkDump job completion -// Returns true if job completed successfully, false if timed out +// Returns true if the job completed, false if it stopped making progress for timeoutDuration. +// +// The budget is a no-progress window, not a total duration. A dump advances one bounded slice per +// scheduling round (the SS returns after SS_BULKDUMP_BATCH_COUNT_MAX_PER_REQUEST batches and the +// remainder is re-dispatched on a later round), so total time scales with the size of the range and +// with how finely it is sharded -- not with the health of the job. A wall-clock budget therefore fails +// dumps that are advancing perfectly well, while still not catching a job that is truly wedged sooner +// than the budget. Measuring absence of progress separates the two. Future monitorBulkDumpJobCompletion(Database cx, UID jobId, double timeoutDuration, double pollInterval) { - double timeoutStart = now(); + double lastProgressTime = now(); + size_t lastCompleteCount = 0; + // Counting completed ranges walks the job's bulkdump metadata, so it must be sampled far more + // coarsely than the liveness poll. + double progressCheckInterval = std::max(pollInterval, timeoutDuration / 10); + double lastProgressCheck = now(); Transaction tr(cx); while (true) { @@ -86,8 +98,26 @@ Future monitorBulkDumpJobCompletion(Database cx, UID jobId, double timeout co_return true; // Job completed successfully } - if (now() - timeoutStart > timeoutDuration) { - co_return false; // Timed out + if (now() - lastProgressCheck >= progressCheckInterval) { + lastProgressCheck = now(); + // A sample that fails carries no information: leave the window running rather than + // letting a transient read error either fail the backup or extend its deadline. + try { + size_t completeCount = co_await getBulkDumpCompleteTaskCount(cx, currentJob.get().getJobRange()); + if (completeCount > lastCompleteCount) { + lastCompleteCount = completeCount; + lastProgressTime = now(); + } + } catch (Error& e) { + if (e.code() == error_code_actor_cancelled) { + throw; + } + TraceEvent(SevWarn, "BulkDumpProgressCheckFailed").error(e).detail("BulkDumpJobId", jobId); + } + } + + if (now() - lastProgressTime > timeoutDuration) { + co_return false; // No progress within the window } co_await delay(pollInterval); @@ -242,6 +272,7 @@ Future> TagUidMap::getAll_impl(TagUidMap* tagsMap, Key prefix = tagsMap->prefix; // Copying it here as tagsMap lifetime is not tied to this actor TagMap::RangeResultType tagPairs = co_await tagsMap->getRange(tr, std::string(), {}, 1e6, snapshot); std::vector results; + results.reserve(tagPairs.results.size()); for (auto& p : tagPairs.results) results.push_back(KeyBackedTag(p.first, prefix)); co_return results; @@ -258,6 +289,7 @@ Future anyPartitionedBackupRunning(Reference tr std::vector tags = co_await getAllBackupTags(tr); std::vector>> futures; + futures.reserve(tags.size()); for (const auto& tag : tags) { futures.push_back(tag.get(tr)); } @@ -288,6 +320,7 @@ Future anyRangePartitionedBackupRunning(Reference tags = co_await getAllBackupTags(tr); std::vector>> futures; + futures.reserve(tags.size()); for (const auto& tag : tags) { futures.push_back(tag.get(tr)); } @@ -630,8 +663,48 @@ Future> getBulk co_return std::make_tuple(completedTasks, submittedTasks, triggeredTasks, runningTasks, totalTasks, completedBytes); } +// Record a terminal bulkload-restore failure in the restore's own state and abort it. +// +// ABORTED is what makes the failure stick: throwing alone leaves the restore retryable, so the retry +// re-reads the same finished job and reaches the same terminal condition. ABORTED also puts the restore +// outside isRunnable(), which makes abortRestore() return before reaching its unlockDatabase() call -- so +// this must release the lock itself, or a failed restore leaves the database locked with no route out +// through the restore API. The commit retries because an unpersisted ABORTED loses both properties. +Future abortBulkLoadRestore(Database cx, RestoreConfig restore, std::string message) { + co_await restore.logError(cx, restore_bulkload_failed(), message); + Reference abortTr(new ReadYourWritesTransaction(cx)); + while (true) { + Error err; + try { + abortTr->setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); + abortTr->setOption(FDBTransactionOptions::LOCK_AWARE); + restore.stateEnum().set(abortTr, ERestoreState::ABORTED); + restore.clearApplyMutationsKeys(abortTr); + bool unlockDB = co_await restore.unlockDBAfterRestore().getD(abortTr, Snapshot::False, true); + if (unlockDB) { + co_await unlockDatabase(abortTr, restore.getUid()); + } + co_await abortTr->commit(); + co_return; + } catch (Error& e) { + if (e.code() == error_code_actor_cancelled) { + throw; + } + err = e; + } + co_await abortTr->onError(err); + } +} + // Monitor BulkLoad job completion and update restore progress counters // restoreUid is used to update the RestoreConfig progress +// +// Returns false when the job stops making progress for timeoutDuration, not when it has simply been +// running that long. A restore's duration is a function of how much data it moves, so a wall-clock budget +// aborts large restores that are advancing normally -- BULKLOAD_JOB_TIMEOUT's own comment concedes as much +// ("large DBs may take days" against a 24 hour default). The task counters this loop already maintains for +// the status display are exactly the progress signal needed, so measuring absence of progress costs +// nothing extra here. Future monitorBulkLoadJobCompletionWithProgress(Database cx, UID jobId, UID restoreUid, @@ -639,7 +712,9 @@ Future monitorBulkLoadJobCompletionWithProgress(Database cx, double timeoutDuration, double pollInterval, bool lockAware) { - double timeoutStart = now(); + double lastProgressTime = now(); + int64_t lastCompletedTasks = -1; + int64_t lastCompletedBytes = -1; RestoreConfig restore(restoreUid); while (true) { @@ -647,12 +722,91 @@ Future monitorBulkLoadJobCompletionWithProgress(Database cx, bool stillRunning = currentJob.present() && currentJob.get().getJobId() == jobId; if (!stillRunning) { + // The job's live metadata is gone, which says the job manager finished walking the job + // range -- not that every task succeeded. A job that ends with a task in Error is + // archived to history in the same transaction that clears the live metadata, so treating + // "no longer running" as success reports a restore complete while part of the key space + // was never ingested. Consult the archived phase before declaring success. + // lockAware: the restore holds the database lock while it runs, so a plain read here + // retries on database_locked until its transaction gives up, and the caller waits forever. + std::vector history; + Optional historyReadError; + try { + history = co_await getBulkLoadJobFromHistory(cx, lockAware); + } catch (Error& e) { + if (e.code() == error_code_actor_cancelled) { + throw; + } + historyReadError = e; + } + if (historyReadError.present()) { + // Fail closed. The tempting argument is that every task reported done, so the restore is + // very likely fine -- but the task counters reporting completion while a task's range was + // never ingested is the exact failure this check exists to catch, so they cannot be the + // grounds for skipping it. getBulkLoadJobFromHistory retries every retryable error + // indefinitely, so reaching here means a non-retryable one, not a transient blip: the + // least safe moment to assume success. + // + // The cost is real and is why this is logged loudly rather than quietly: a restore whose + // data is intact can be failed because a system-keyspace read was unavailable. Retrying + // the restore is the remedy. + TraceEvent(SevWarnAlways, "BulkLoadRestoreJobOutcomeUnknown") + .error(historyReadError.get()) + .detail("RestoreUID", restoreUid) + .detail("BulkLoadJobId", jobId) + .detail("Reason", + "Could not read the archived bulkload job phase, so the restore's outcome is " + "unverifiable; failing closed rather than reporting success") + .detail("Cost", "A restore whose data is intact may be failed by this; retry the restore") + .detail("TasksReportedComplete", lastCompletedTasks) + .detail("BytesReportedComplete", lastCompletedBytes); + co_await abortBulkLoadRestore( + cx, + restore, + "BulkLoad restore failed: the archived bulkload job phase could not be read, so the " + "restore's outcome could not be verified"); + throw restore_bulkload_failed(); + } + // Complete is the only phase that attests every task was ingested: the job manager sets it + // via setCompletePhase() only after walking the whole job range with no task in Error. + // Error and Cancelled both reach history with the live metadata already cleared -- a cancel + // wipes job and task metadata in the same transaction that archives -- so anything other + // than Complete, including a job missing from history altogether, leaves part of the key + // space unaccounted for and must fail the restore rather than report success. + Optional archivedPhase; + for (const auto& job : history) { + if (job.getJobId() == jobId) { + archivedPhase = job.getPhase(); + break; + } + } + if (!archivedPhase.present() || archivedPhase.get() != BulkLoadJobPhase::Complete) { + // A task that could not be loaded is a property of this restore, not a defect in the + // code, so this is not SevError: the restore failing is the signal, and a SevError + // would additionally fail any simulation that provokes the condition. + TraceEvent(SevWarnAlways, "BulkLoadRestoreJobDidNotComplete") + .detail("RestoreUID", restoreUid) + .detail("BulkLoadJobId", jobId) + .detail("JobPhase", + archivedPhase.present() ? convertBulkLoadJobPhaseToString(archivedPhase.get()) + : "AbsentFromHistory") + .detail("TasksReportedComplete", lastCompletedTasks) + .detail("BytesReportedComplete", lastCompletedBytes); + co_await abortBulkLoadRestore( + cx, restore, "BulkLoad restore failed: the bulkload job did not complete"); + throw restore_bulkload_failed(); + } co_return true; } // Update progress based on completed bulkload tasks try { auto [completed, submitted, triggered, running, total, bytes] = co_await getBulkLoadTaskProgress(cx, jobId); + if (completed > lastCompletedTasks || bytes > lastCompletedBytes) { + lastCompletedTasks = completed; + lastCompletedBytes = bytes; + lastProgressTime = now(); + } if (total > 0) { // For bulkload restores, fileBlockCount is 0, so use task count as "blocks" // This provides meaningful progress tracking for the restore status display @@ -700,8 +854,8 @@ Future monitorBulkLoadJobCompletionWithProgress(Database cx, TraceEvent(SevWarn, "BulkLoadRestoreProgressError").error(e).detail("JobId", jobId); } - if (now() - timeoutStart > timeoutDuration) { - co_return false; // Timed out + if (now() - lastProgressTime > timeoutDuration) { + co_return false; // No progress within the window } co_await delay(pollInterval); @@ -3699,7 +3853,9 @@ struct BulkDumpTaskFunc : BackupTaskFuncBase { .detail("BackupUID", config.getUid()) .detail("BulkDumpJobId", bulkDumpJob.getJobId()) .detail("TimeoutDuration", CLIENT_KNOBS->BULKDUMP_JOB_TIMEOUT) - .detail("Reason", "BulkDump did not complete; backup would not be bulkload-restorable"); + .detail("Reason", + "BulkDump made no progress within the timeout; backup would not be " + "bulkload-restorable"); Params.timeoutOccurred().set(task, true); throw backup_error(); } @@ -3732,6 +3888,7 @@ struct BulkDumpTaskFunc : BackupTaskFuncBase { // Build beginEndKeys from backup ranges std::vector> beginEndKeys; + beginEndKeys.reserve(backupRanges.size()); for (const auto& range : backupRanges) { beginEndKeys.emplace_back(range.begin, range.end); } @@ -4311,7 +4468,8 @@ struct BulkLoadRestoreTaskFunc : RestoreTaskFuncBase { TraceEvent(SevWarn, "BulkLoadRestoreTimeout") .detail("RestoreUID", restore.getUid()) .detail("BulkLoadJobId", bulkLoadJob.getJobId()) - .detail("TimeoutDuration", CLIENT_KNOBS->BULKLOAD_JOB_TIMEOUT); + .detail("TimeoutDuration", CLIENT_KNOBS->BULKLOAD_JOB_TIMEOUT) + .detail("Reason", "BulkLoad restore made no progress within the timeout"); // Restore original BulkLoad mode before throwing if (originalBulkLoadMode != 1) { co_await setBulkLoadMode(cx, originalBulkLoadMode); diff --git a/fdbclient/ManagementAPI.cpp b/fdbclient/ManagementAPI.cpp index 0c6c0f106be..bc9e83c6674 100644 --- a/fdbclient/ManagementAPI.cpp +++ b/fdbclient/ManagementAPI.cpp @@ -2273,6 +2273,7 @@ Future waitForFullReplication(Database cx) { config.fromKeyValues((VectorRef)confResults); std::vector>> replicasFutures; + replicasFutures.reserve(config.regions.size()); for (auto& region : config.regions) { replicasFutures.push_back(tr.get(datacenterReplicasKeyFor(region.dcId))); } @@ -2769,7 +2770,7 @@ Future addBulkLoadJobToHistory(Transaction* tr, BulkLoadJobState jobState) tr->set(newJobKey, bulkLoadJobValue(jobState)); } -AsyncResult> getBulkLoadJobFromHistory(Database cx) { +AsyncResult> getBulkLoadJobFromHistory(Database cx, bool lockAware) { RangeResult jobHistoryResult; Key beginKey = bulkLoadJobHistoryKeys.begin; Key endKey = bulkLoadJobHistoryKeys.end; @@ -2778,6 +2779,15 @@ AsyncResult> getBulkLoadJobFromHistory(Database cx while (true) { Error err; try { + // READ_LOCK_AWARE is the load-bearing one: a caller may read this while the database is + // locked -- a restore holds the lock while deciding whether its bulkload job succeeded -- + // and database_locked is retryable, so without it the loop below spins forever instead of + // failing. The read-only variant suffices here and enforces that this stays a read. + // READ_SYSTEM_KEYS is belt-and-braces for a system-keyspace range. + tr.setOption(FDBTransactionOptions::READ_SYSTEM_KEYS); + if (lockAware) { + tr.setOption(FDBTransactionOptions::READ_LOCK_AWARE); + } jobHistoryResult.clear(); jobHistoryResult = co_await tr.getRange(KeyRangeRef(beginKey, endKey), CLIENT_KNOBS->BULKLOAD_JOB_HISTORY_COUNT_MAX); diff --git a/fdbclient/MonitorLeader.cpp b/fdbclient/MonitorLeader.cpp index d5cfb2c07d5..1747c09d5c4 100644 --- a/fdbclient/MonitorLeader.cpp +++ b/fdbclient/MonitorLeader.cpp @@ -276,6 +276,7 @@ Future> tryResolveHostnamesImpl(ClusterConnectionStr allCoordinatorsSet.insert(coord); } std::vector> fs; + fs.reserve(self->hostnames.size()); for (auto& hostname : self->hostnames) { fs.push_back(map(hostname.resolve(), [&](Optional const& addr) -> Void { if (addr.present()) { @@ -888,6 +889,7 @@ void shrinkProxyList(ClientDBInfo& ni, std::vector& lastGrvProxies) { if (ni.commitProxies.size() > CLIENT_KNOBS->MAX_COMMIT_PROXY_CONNECTIONS) { std::vector commitProxyUIDs; + commitProxyUIDs.reserve(ni.commitProxies.size()); for (auto& commitProxy : ni.commitProxies) { commitProxyUIDs.push_back(commitProxy.id()); } @@ -905,6 +907,7 @@ void shrinkProxyList(ClientDBInfo& ni, } if (ni.grvProxies.size() > CLIENT_KNOBS->MAX_GRV_PROXY_CONNECTIONS) { std::vector grvProxyUIDs; + grvProxyUIDs.reserve(ni.grvProxies.size()); for (auto& grvProxy : ni.grvProxies) { grvProxyUIDs.push_back(grvProxy.id()); } diff --git a/fdbclient/MultiVersionTransaction.cpp b/fdbclient/MultiVersionTransaction.cpp index 95a8fbae282..d974afbb941 100644 --- a/fdbclient/MultiVersionTransaction.cpp +++ b/fdbclient/MultiVersionTransaction.cpp @@ -2263,9 +2263,9 @@ std::vector> MultiVersionApi::copyExternalLibraryPe char tempName[MAX_TMP_NAME_LENGTH]; snprintf(tempName, MAX_TMP_NAME_LENGTH, "%s/%s-XXXXXX", tmpDir.c_str(), filename.c_str()); int tempFd = mkstemp(tempName); - int fd; + int fd = open(path.c_str(), O_RDONLY); - if ((fd = open(path.c_str(), O_RDONLY)) == -1) { + if (fd == -1) { TraceEvent("ExternalClientNotFound").detail("LibraryPath", path); throw file_not_found(); } diff --git a/fdbclient/NativeAPI.actor.cpp b/fdbclient/NativeAPI.cpp similarity index 81% rename from fdbclient/NativeAPI.actor.cpp rename to fdbclient/NativeAPI.cpp index 2360be89e79..83ff5ed2398 100644 --- a/fdbclient/NativeAPI.actor.cpp +++ b/fdbclient/NativeAPI.cpp @@ -1,5 +1,5 @@ /* - * NativeAPI.actor.cpp + * NativeAPI.cpp * * This source file is part of the FoundationDB open source project * @@ -71,6 +71,7 @@ #include "fdbclient/StorageServerInterface.h" #include "fdbclient/SystemData.h" #include "fdbclient/TransactionLineage.h" +#include "fdbclient/VersionVector.h" #include "fdbclient/versions.h" #include "fdbrpc/WellKnownEndpoints.h" #include "fdbrpc/LoadBalance.h" @@ -113,7 +114,6 @@ #else #include #endif -#include "flow/actorcompiler.h" // This must be the last #include. template class RequestStream; template struct NetNotifiedQueue; @@ -174,10 +174,10 @@ bool DatabaseContext::getCachedLocations(const KeyRangeRef& range, auto begin = locationCache.rangeContaining(range.begin); auto end = locationCache.rangeContainingKeyBefore(range.end); - loop { + while (true) { auto r = reverse ? end : begin; if (!r->value()) { - CODE_PROBE(result.size(), "had some but not all cached locations"); + CODE_PROBE(!result.empty(), "had some but not all cached locations"); result.clear(); return false; } @@ -297,9 +297,9 @@ void DatabaseContext::setOption(FDBDatabaseOptions::Option option, Optional(value.get()) : Optional>(), clientLocality.machineId(), clientLocality.dcId()); - if (clientInfo->get().commitProxies.size()) + if (!clientInfo->get().commitProxies.empty()) commitProxies = makeReference(clientInfo->get().commitProxies); - if (clientInfo->get().grvProxies.size()) + if (!clientInfo->get().grvProxies.empty()) grvProxies = makeReference(clientInfo->get().grvProxies); server_interf.clear(); locationCache.insert(allKeys, Reference()); @@ -313,9 +313,9 @@ void DatabaseContext::setOption(FDBDatabaseOptions::Option option, Optional(value.get()) : Optional>()); - if (clientInfo->get().commitProxies.size()) + if (!clientInfo->get().commitProxies.empty()) commitProxies = makeReference(clientInfo->get().commitProxies); - if (clientInfo->get().grvProxies.size()) + if (!clientInfo->get().grvProxies.empty()) grvProxies = makeReference(clientInfo->get().grvProxies); server_interf.clear(); locationCache.insert(allKeys, Reference()); @@ -797,7 +797,7 @@ void setNetworkOption(FDBNetworkOptions::Option option, Optional valu #endif } - ASSERT(supportedVersions.size() > 0); + ASSERT(!supportedVersions.empty()); networkOptions.supportedVersions->set(supportedVersions); break; @@ -936,15 +936,15 @@ void DatabaseContext::updateProxies() { commitProxies.clear(); grvProxies.clear(); bool commitProxyProvisional = false, grvProxyProvisional = false; - if (clientInfo->get().commitProxies.size()) { + if (!clientInfo->get().commitProxies.empty()) { commitProxies = makeReference(clientInfo->get().commitProxies); commitProxyProvisional = clientInfo->get().commitProxies[0].provisional; } - if (clientInfo->get().grvProxies.size()) { + if (!clientInfo->get().grvProxies.empty()) { grvProxies = makeReference(clientInfo->get().grvProxies); grvProxyProvisional = clientInfo->get().grvProxies[0].provisional; } - if (clientInfo->get().commitProxies.size() && clientInfo->get().grvProxies.size()) { + if (!clientInfo->get().commitProxies.empty() && !clientInfo->get().grvProxies.empty()) { ASSERT(commitProxyProvisional == grvProxyProvisional); proxyProvisional = commitProxyProvisional; } @@ -1150,9 +1150,9 @@ AddressExclusion AddressExclusion::parse(StringRef const& key) { } } -Future> getValue(Reference const& trState, - Key const& key, - TransactionRecordLogInfo const& recordLogInfo = TransactionRecordLogInfo::True); +Future> getValue(Reference trState, + Key key, + TransactionRecordLogInfo recordLogInfo = TransactionRecordLogInfo::True); Future getRange(Reference const& trState, KeySelector const& begin, @@ -1238,12 +1238,13 @@ Future getKeyLocation_internal(Database cx, ASSERT(key < allKeys.end); } - if (debugID.present()) + if (debugID.present()) { g_traceBatch.addEvent("TransactionDebug", debugID.get().first(), "NativeAPI.getKeyLocation.Before", spanContext.traceID, spanContext.spanID); + } while (true) { try { @@ -1257,12 +1258,13 @@ Future getKeyLocation_internal(Database cx, useProvisionalProxies, TaskPriority::DefaultPromiseEndpoint); ++cx->transactionKeyServerLocationRequestsCompleted; - if (debugID.present()) + if (debugID.present()) { g_traceBatch.addEvent("TransactionDebug", debugID.get().first(), "NativeAPI.getKeyLocation.After", spanContext.traceID, spanContext.spanID); + } ASSERT(rep.results.size() == 1); auto locationInfo = cx->setCachedLocation(rep.results[0].first, rep.results[0].second); @@ -1395,12 +1397,13 @@ Future> getKeyRangeLocations_internal(Database UseProvisionalProxies useProvisionalProxies, Version version) { Span span("NAPI:getKeyRangeLocations"_loc, spanContext); - if (debugID.present()) + if (debugID.present()) { g_traceBatch.addEvent("TransactionDebug", debugID.get().first(), "NativeAPI.getKeyLocations.Before", spanContext.traceID, spanContext.spanID); + } while (true) { try { @@ -1414,13 +1417,14 @@ Future> getKeyRangeLocations_internal(Database useProvisionalProxies, TaskPriority::DefaultPromiseEndpoint); ++cx->transactionKeyServerLocationRequestsCompleted; - if (debugID.present()) + if (debugID.present()) { g_traceBatch.addEvent("TransactionDebug", debugID.get().first(), "NativeAPI.getKeyLocations.After", spanContext.traceID, spanContext.spanID); - ASSERT(rep.results.size()); + } + ASSERT(!rep.results.empty()); std::vector results; for (int shard = 0; shard < rep.results.size(); ++shard) { @@ -1534,7 +1538,7 @@ Future warmRange_impl(Reference trState, KeyRange keys) trState->readVersion()); totalRanges += CLIENT_KNOBS->WARM_RANGE_SHARD_LIMIT; totalRequests++; - if (locations.size() == 0 || totalRanges >= trState->cx->locationCacheSize || + if (locations.empty() || totalRanges >= trState->cx->locationCacheSize || locations[locations.size() - 1].range.end >= keys.end) break; @@ -1601,8 +1605,32 @@ Reference TransactionState::cloneAndReset(Reference startTransaction(Reference trState) { - wait(success(trState->readVersionFuture)); +Future startTransaction(Reference trStateInput) { + Reference trState(std::move(trStateInput)); + co_await success(trState->readVersionFuture); +} + +TEST_CASE("/fdbclient/NativeAPI/startTransaction/releasesStateOnCompletion") { + for (bool fail : { false, true }) { + auto trState = makeReference(TaskPriority::DefaultEndpoint, SpanContext()); + Promise version; + trState->readVersionFuture = version.getFuture(); + trState->startFuture = ::startTransaction(trState); + ASSERT_EQ(trState->debugGetReferenceCount(), 2); + + if (fail) { + version.sendError(transaction_too_old()); + } else { + version.send(123); + } + + ASSERT(trState->startFuture.isReady()); + ASSERT_EQ(trState->startFuture.isError(), fail); + if (fail) { + ASSERT_EQ(trState->startFuture.getError().code(), error_code_transaction_too_old); + } + ASSERT_EQ(trState->debugGetReferenceCount(), 1); + } return Void(); } @@ -1625,24 +1653,26 @@ Future Transaction::warmRange(KeyRange keys) { return warmRange_impl(trState, keys); } -ACTOR Future> getValue(Reference trState, - Key key, - TransactionRecordLogInfo recordLogInfo) { - wait(trState->startTransaction()); +Future> getValue(Reference trStateInput, + Key keyInput, + TransactionRecordLogInfo recordLogInfo) { + Reference trState(std::move(trStateInput)); + Key key(std::move(keyInput)); + co_await trState->startTransaction(); - state Span span("NAPI:getValue"_loc, trState->spanContext); + Span span("NAPI:getValue"_loc, trState->spanContext); trState->cx->validateVersion(trState->readVersion()); - loop { - state KeyRangeLocationInfo locationInfo = - wait(getKeyLocation(trState, key, &StorageServerInterface::getValue, Reverse::False)); + KeyRangeLocationInfo locationInfo; + while (true) { + locationInfo = co_await getKeyLocation(trState, key, &StorageServerInterface::getValue, Reverse::False); - state Optional getValueID = Optional(); - state uint64_t startTime; - state double startTimeD; - state VersionVector ssLatestCommitVersions; - state Optional readOptions = trState->readOptions; + Optional getValueID; + uint64_t startTime{ 0 }; + double startTimeD{ 0 }; + VersionVector ssLatestCommitVersions; + Optional readOptions = trState->readOptions; trState->cx->getLatestCommitVersions(locationInfo.locations, trState, ssLatestCommitVersions); try { @@ -1671,34 +1701,37 @@ ACTOR Future> getValue(Reference trState, startTimeD = now(); ++trState->cx->transactionPhysicalReads; - state GetValueReply reply; + GetValueReply reply; try { if (CLIENT_BUGGIFY_WITH_PROB(.01)) { throw deterministicRandom()->randomChoice( std::vector{ transaction_too_old(), future_version() }); } - choose { - when(wait(trState->cx->connectionFileChanged())) { - throw transaction_too_old(); - } - when(GetValueReply _reply = wait( - loadBalance(locationInfo.locations->locations(), - &StorageServerInterface::getValue, - GetValueRequest(span.context, - key, - trState->readVersion(), - trState->cx->sampleReadTags() ? trState->options.readTags - : Optional(), - readOptions, - ssLatestCommitVersions), - TaskPriority::DefaultPromiseEndpoint, - AtMostOnce::False, - trState->cx->enableLocalityLoadBalance ? &trState->cx->queueModel : nullptr, - trState->options.enableReplicaConsistencyCheck, - trState->options.requiredReplicas))) { - reply = _reply; - } + auto connectionChanged = trState->cx->connectionFileChanged(); + if (connectionChanged.isReady()) { + co_await connectionChanged; + throw transaction_too_old(); + } + auto res = co_await race( + std::move(connectionChanged), + loadBalance( + locationInfo.locations->locations(), + &StorageServerInterface::getValue, + GetValueRequest(span.context, + key, + trState->readVersion(), + trState->cx->sampleReadTags() ? trState->options.readTags : Optional(), + readOptions, + ssLatestCommitVersions), + TaskPriority::DefaultPromiseEndpoint, + AtMostOnce::False, + trState->cx->enableLocalityLoadBalance ? &trState->cx->queueModel : nullptr, + trState->options.enableReplicaConsistencyCheck, + trState->options.requiredReplicas)); + if (res.index() == 0) { + throw transaction_too_old(); } + reply = std::get<1>(std::move(res)); ++trState->cx->transactionPhysicalReadsCompleted; } catch (Error&) { ++trState->cx->transactionPhysicalReadsCompleted; @@ -1732,7 +1765,7 @@ ACTOR Future> getValue(Reference trState, trState->cx->transactionBytesRead += reply.value.present() ? reply.value.get().size() : 0; ++trState->cx->transactionKeysRead; - return reply.value; + co_return std::move(reply.value); } catch (Error& e) { trState->cx->getValueCompleted->latency = timer_int() - startTime; trState->cx->getValueCompleted->log(); @@ -1745,26 +1778,29 @@ ACTOR Future> getValue(Reference trState, } if (e.code() == error_code_wrong_shard_server || e.code() == error_code_all_alternatives_failed) { trState->cx->invalidateCache(key); - wait(delay(CLIENT_KNOBS->WRONG_SHARD_SERVER_DELAY, trState->taskID)); } else { - if (trState->trLogInfo && recordLogInfo) + if (trState->trLogInfo && recordLogInfo) { trState->trLogInfo->addLog( FdbClientLogEvents::EventGetError( startTimeD, trState->cx->clientLocality.dcId(), static_cast(e.code()), key), trState->spanContext); + } throw e; } } + co_await delay(CLIENT_KNOBS->WRONG_SHARD_SERVER_DELAY, trState->taskID); } } -ACTOR Future getKey(Reference trState, KeySelector k) { - wait(trState->startTransaction()); +Future getKey(Reference trStateInput, KeySelector kInput) { + Reference trState(std::move(trStateInput)); + KeySelector k(std::move(kInput)); + co_await trState->startTransaction(); - state Optional getKeyID; - state Optional readOptions = trState->readOptions; + Optional getKeyID; + Optional readOptions = trState->readOptions; - state Span span("NAPI:getKey"_loc, trState->spanContext); + Span span("NAPI:getKey"_loc, trState->spanContext); if (trState->readOptions.present() && trState->readOptions.get().debugID.present()) { getKeyID = nondeterministicRandom()->randomUniqueID(); readOptions.get().debugID = getKeyID; @@ -1783,25 +1819,26 @@ ACTOR Future getKey(Reference trState, KeySelector k) { // k.getKey()).detail("Offset",k.offset).detail("OrEqual",k.orEqual); } - loop { + KeyRangeLocationInfo locationInfo; + while (true) { if (k.getKey() == allKeys.end) { if (k.offset > 0) { - return allKeys.end; + co_return allKeys.end; } k.orEqual = false; } else if (k.getKey() == allKeys.begin && k.offset <= 0) { - return Key(); + co_return Key(); } Key locationKey(k.getKey(), k.arena()); - state KeyRangeLocationInfo locationInfo = - wait(getKeyLocation(trState, locationKey, &StorageServerInterface::getKey, Reverse{ k.isBackward() })); + locationInfo = + co_await getKeyLocation(trState, locationKey, &StorageServerInterface::getKey, Reverse{ k.isBackward() }); - state VersionVector ssLatestCommitVersions; + VersionVector ssLatestCommitVersions; trState->cx->getLatestCommitVersions(locationInfo.locations, trState, ssLatestCommitVersions); try { - if (getKeyID.present()) + if (getKeyID.present()) { g_traceBatch.addEvent( "GetKeyDebug", getKeyID.get().first(), @@ -1809,6 +1846,7 @@ ACTOR Future getKey(Reference trState, KeySelector k) { trState->spanContext.traceID, trState->spanContext.spanID); //.detail("StartKey", // k.getKey()).detail("Offset",k.offset).detail("OrEqual",k.orEqual); + } ++trState->cx->transactionPhysicalReads; GetKeyRequest req(span.context, @@ -1819,81 +1857,90 @@ ACTOR Future getKey(Reference trState, KeySelector k) { ssLatestCommitVersions); req.arena.dependsOn(k.arena()); - state GetKeyReply reply; + GetKeyReply reply; try { - choose { - when(wait(trState->cx->connectionFileChanged())) { - throw transaction_too_old(); - } - when(GetKeyReply _reply = wait( - loadBalance(locationInfo.locations->locations(), - &StorageServerInterface::getKey, - req, - TaskPriority::DefaultPromiseEndpoint, - AtMostOnce::False, - trState->cx->enableLocalityLoadBalance ? &trState->cx->queueModel : nullptr, - trState->options.enableReplicaConsistencyCheck, - trState->options.requiredReplicas))) { - reply = _reply; - } + auto connectionChanged = trState->cx->connectionFileChanged(); + if (connectionChanged.isReady()) { + co_await connectionChanged; + throw transaction_too_old(); + } + auto res = co_await race( + std::move(connectionChanged), + loadBalance(locationInfo.locations->locations(), + &StorageServerInterface::getKey, + req, + TaskPriority::DefaultPromiseEndpoint, + AtMostOnce::False, + trState->cx->enableLocalityLoadBalance ? &trState->cx->queueModel : nullptr, + trState->options.enableReplicaConsistencyCheck, + trState->options.requiredReplicas)); + if (res.index() == 0) { + throw transaction_too_old(); } + reply = std::get<1>(std::move(res)); ++trState->cx->transactionPhysicalReadsCompleted; } catch (Error&) { ++trState->cx->transactionPhysicalReadsCompleted; throw; } - if (getKeyID.present()) + if (getKeyID.present()) { g_traceBatch.addEvent("GetKeyDebug", getKeyID.get().first(), "NativeAPI.getKey.After", trState->spanContext.traceID, trState->spanContext.spanID); //.detail("NextKey",reply.sel.key).detail("Offset", // reply.sel.offset).detail("OrEqual", k.orEqual); + } k = reply.sel; if (!k.offset && k.orEqual) { - return k.getKey(); + co_return k.getKey(); } + continue; } catch (Error& e) { - if (getKeyID.present()) + if (getKeyID.present()) { g_traceBatch.addEvent("GetKeyDebug", getKeyID.get().first(), "NativeAPI.getKey.Error", trState->spanContext.traceID, trState->spanContext.spanID); + } if (e.code() == error_code_wrong_shard_server || e.code() == error_code_all_alternatives_failed) { trState->cx->invalidateCache(k.getKey(), Reverse{ k.isBackward() }); - - wait(delay(CLIENT_KNOBS->WRONG_SHARD_SERVER_DELAY, trState->taskID)); } else { TraceEvent(SevInfo, "GetKeyError").error(e).detail("AtKey", k.getKey()).detail("Offset", k.offset); throw e; } } + co_await delay(CLIENT_KNOBS->WRONG_SHARD_SERVER_DELAY, trState->taskID); } } +namespace { + +// Each proxy retry must sample the live cache version and tracing context. +class RawReadVersionRequestBuilder { + const VersionVector& versionVector; + const SpanContext& spanContext; + +public: + RawReadVersionRequestBuilder(const VersionVector& versionVector, const SpanContext& spanContext) + : versionVector(versionVector), spanContext(spanContext) {} + + GetReadVersionRequest build() const { + return GetReadVersionRequest(spanContext, 0, TransactionPriority::IMMEDIATE, versionVector.getMaxVersion()); + } +}; + +} // namespace + static Future waitForCommittedVersionImpl(Database cx, Version version, SpanContext spanContext) { Span span("NAPI:waitForCommittedVersion"_loc, spanContext); while (true) { try { - // Reset proxy backoff before constructing a GRV request. - auto proxiesChanged = cx->onProxiesChanged(); - if (proxiesChanged.isReady()) { - co_await proxiesChanged; - continue; - } - auto res = co_await race(std::move(proxiesChanged), - basicLoadBalance(cx->getGrvProxies(UseProvisionalProxies::False), - &GrvProxyInterface::getConsistentReadVersion, - GetReadVersionRequest(span.context, - 0, - TransactionPriority::IMMEDIATE, - cx->ssVersionVectorCache.getMaxVersion()), - cx->taskID)); - if (res.index() == 0) { - continue; - } - GetReadVersionReply v = std::get<1>(std::move(res)); + GetReadVersionReply v = co_await grvProxyLoadBalance( + cx, + RawReadVersionRequestBuilder(cx->ssVersionVectorCache, span.context), + &GrvProxyInterface::getConsistentReadVersion); cx->minAcceptableReadVersion = std::min(cx->minAcceptableReadVersion, v.version); if (v.midShardSize > 0) cx->smoothMidShardSize.setTotal(v.midShardSize); @@ -1926,54 +1973,44 @@ Future waitForCommittedVersion(Database const& cx, Version const& versi return waitForCommittedVersionImpl(cx, version, spanContext); } -ACTOR Future getRawVersion(Reference trState) { - state Span span("NAPI:getRawVersion"_loc, trState->spanContext); - loop { - choose { - when(wait(trState->cx->onProxiesChanged())) {} - when(GetReadVersionReply v = - wait(basicLoadBalance(trState->cx->getGrvProxies(UseProvisionalProxies::False), - &GrvProxyInterface::getConsistentReadVersion, - GetReadVersionRequest(trState->spanContext, - 0, - TransactionPriority::IMMEDIATE, - trState->cx->ssVersionVectorCache.getMaxVersion()), - trState->cx->taskID))) { - if (trState->cx->versionVectorCacheActive(v.ssVersionVectorDelta)) { - if (trState->cx->isCurrentGrvProxy(v.proxyId)) { - trState->cx->ssVersionVectorCache.applyDelta(v.ssVersionVectorDelta); - } else { - trState->cx->ssVersionVectorCache.clear(); - } - } - return v.version; - } +Future getRawVersion(Reference trStateInput) { + Reference trState(std::move(trStateInput)); + Span span("NAPI:getRawVersion"_loc, trState->spanContext); + GetReadVersionReply v = co_await grvProxyLoadBalance( + trState->cx, + RawReadVersionRequestBuilder(trState->cx->ssVersionVectorCache, trState->spanContext), + &GrvProxyInterface::getConsistentReadVersion); + if (trState->cx->versionVectorCacheActive(v.ssVersionVectorDelta)) { + if (trState->cx->isCurrentGrvProxy(v.proxyId)) { + trState->cx->ssVersionVectorCache.applyDelta(v.ssVersionVectorDelta); + } else { + trState->cx->ssVersionVectorCache.clear(); } } + co_return v.version; } -ACTOR Future readVersionBatcher( - DatabaseContext* cx, - FutureStream, Optional>> versionStream, - uint32_t flags); - -ACTOR Future watchValue(Database cx, Reference parameters) { - state Span span("NAPI:watchValue"_loc, parameters->spanContext); - state Version ver = parameters->version; +Future watchValue(Database cxInput, Reference parametersInput) { + Database cx(std::move(cxInput)); + Reference parameters(std::move(parametersInput)); + Span span("NAPI:watchValue"_loc, parameters->spanContext); + Version ver = parameters->version; cx->validateVersion(parameters->version); ASSERT(parameters->version != latestVersion); - loop { - state KeyRangeLocationInfo locationInfo = wait(getKeyLocation(cx, - parameters->key, - &StorageServerInterface::watchValue, - parameters->spanContext, - parameters->debugID, - parameters->useProvisionalProxies, - Reverse::False, - parameters->version)); + KeyRangeLocationInfo locationInfo; + while (true) { + locationInfo = co_await getKeyLocation(cx, + parameters->key, + &StorageServerInterface::watchValue, + parameters->spanContext, + parameters->debugID, + parameters->useProvisionalProxies, + Reverse::False, + parameters->version); + Error err; try { - state Optional watchValueID = Optional(); + Optional watchValueID; if (parameters->debugID.present()) { watchValueID = nondeterministicRandom()->randomUniqueID(); @@ -1988,24 +2025,25 @@ ACTOR Future watchValue(Database cx, Reference p parameters->spanContext.traceID, parameters->spanContext.spanID); //.detail("TaskID", g_network->getCurrentTask()); } - state WatchValueReply resp; - choose { - when(WatchValueReply r = wait( - loadBalance(locationInfo.locations->locations(), - &StorageServerInterface::watchValue, - WatchValueRequest(span.context, - parameters->key, - parameters->value, - ver, - cx->sampleReadTags() ? parameters->tags : Optional(), - watchValueID), - TaskPriority::DefaultPromiseEndpoint))) { - resp = r; - } - when(wait(cx->connectionRecord ? cx->connectionRecord->onChange() : Never())) { - wait(Never()); - } + auto watchResponse = + loadBalance(locationInfo.locations->locations(), + &StorageServerInterface::watchValue, + WatchValueRequest(span.context, + parameters->key, + parameters->value, + ver, + cx->sampleReadTags() ? parameters->tags : Optional(), + watchValueID), + TaskPriority::DefaultPromiseEndpoint); + auto connectionChanged = watchResponse.isReady() || !cx->connectionRecord + ? Future(Never()) + : cx->connectionRecord->onChange(); + // Release the losing RPC before waiting indefinitely on a connection change. + auto res = co_await race(std::exchange(watchResponse, {}), std::exchange(connectionChanged, {})); + if (res.index() == 1) { + co_await Future(Never()); } + WatchValueReply resp = std::get<0>(res); if (watchValueID.present()) { g_traceBatch.addEvent("WatchValueDebug", watchValueID.get().first(), @@ -2017,7 +2055,7 @@ ACTOR Future watchValue(Database cx, Reference p // FIXME: wait for known committed version on the storage server before replying, // cannot do this until the storage server is notified on knownCommittedVersion changes from tlog (faster // than the current update loop) - Version v = wait(waitForCommittedVersion(cx, resp.version, span.context)); + Version v = co_await waitForCommittedVersion(cx, resp.version, span.context); // False if there is a master failure between getting the response // and getting the committed version, Dependent on @@ -2028,7 +2066,7 @@ ACTOR Future watchValue(Database cx, Reference p bool buggifyRetry = g_network->isSimulated() && !g_simulator->speedUpSimulation && buggify(0.1); CODE_PROBE(buggifyRetry, "Watch buggifying version gap retry"); if (v - resp.version < 50'000'000 && !buggifyRetry) { - return resp.version; + co_return resp.version; } ver = v; @@ -2039,25 +2077,26 @@ ACTOR Future watchValue(Database cx, Reference p parameters->spanContext.traceID, parameters->spanContext.spanID); } + continue; } catch (Error& e) { - if (e.code() == error_code_wrong_shard_server || e.code() == error_code_all_alternatives_failed) { - cx->invalidateCache(parameters->key); - wait(delay(CLIENT_KNOBS->WRONG_SHARD_SERVER_DELAY, parameters->taskID)); - } else if (e.code() == error_code_watch_cancelled || e.code() == error_code_process_behind) { - // clang-format off - CODE_PROBE(e.code() == error_code_watch_cancelled, "Too many watches on the storage server, poll for changes instead"); - CODE_PROBE(e.code() == error_code_process_behind, "The storage servers are all behind", probe::decoration::rare); - // clang-format on - wait(delay(CLIENT_KNOBS->WATCH_POLLING_TIME, parameters->taskID)); - } else if (e.code() == error_code_timed_out) { // The storage server occasionally times out watches in case - // it was cancelled - CODE_PROBE(true, "A watch timed out"); - wait(delay(CLIENT_KNOBS->FUTURE_VERSION_RETRY_DELAY, parameters->taskID)); - } else { - state Error err = e; - wait(delay(CLIENT_KNOBS->FUTURE_VERSION_RETRY_DELAY, parameters->taskID)); - throw err; - } + err = e; + } + if (err.code() == error_code_wrong_shard_server || err.code() == error_code_all_alternatives_failed) { + cx->invalidateCache(parameters->key); + co_await delay(CLIENT_KNOBS->WRONG_SHARD_SERVER_DELAY, parameters->taskID); + } else if (err.code() == error_code_watch_cancelled || err.code() == error_code_process_behind) { + // clang-format off + CODE_PROBE(err.code() == error_code_watch_cancelled, "Too many watches on the storage server, poll for changes instead"); + CODE_PROBE(err.code() == error_code_process_behind, "The storage servers are all behind", probe::decoration::rare); + // clang-format on + co_await delay(CLIENT_KNOBS->WATCH_POLLING_TIME, parameters->taskID); + } else if (err.code() == error_code_timed_out) { // The storage server occasionally times out watches in case + // it was cancelled + CODE_PROBE(true, "A watch timed out"); + co_await delay(CLIENT_KNOBS->FUTURE_VERSION_RETRY_DELAY, parameters->taskID); + } else { + co_await delay(CLIENT_KNOBS->FUTURE_VERSION_RETRY_DELAY, parameters->taskID); + throw err; } } } @@ -2218,8 +2257,8 @@ class WatchRefCountUpdater { } cx = std::move(other.cx); - key = std::move(other.key); - version = std::move(other.version); + key = other.key; + version = other.version; cx->increaseWatchRefCount(key, version); @@ -2235,23 +2274,26 @@ class WatchRefCountUpdater { } // namespace -ACTOR Future watchValueMap(Future version, - Key key, - Optional value, - Database cx, - TagSet tags, - SpanContext spanContext, - TaskPriority taskID, - Optional debugID, - UseProvisionalProxies useProvisionalProxies) { - state Version ver = wait(version); - state WatchRefCountUpdater watchRefCountUpdater(cx, key, ver); - - wait(getWatchFuture( - cx, - makeReference(key, value, ver, tags, spanContext, taskID, debugID, useProvisionalProxies))); - - return Void(); +Future watchValueMap(Future versionInput, + Key keyInput, + Optional valueInput, + Database cxInput, + TagSet tagsInput, + SpanContext spanContext, + TaskPriority taskID, + Optional debugID, + UseProvisionalProxies useProvisionalProxies) { + Future version(std::move(versionInput)); + Key key(std::move(keyInput)); + Optional value(std::move(valueInput)); + Database cx(std::move(cxInput)); + TagSet tags(std::move(tagsInput)); + WatchRefCountUpdater watchRefCountUpdater; + Version ver = co_await version; + watchRefCountUpdater = WatchRefCountUpdater(cx, key, ver); + + co_await getWatchFuture( + cx, makeReference(key, value, ver, tags, spanContext, taskID, debugID, useProvisionalProxies)); } template @@ -2286,25 +2328,29 @@ PublicRequestStream StorageServerInterface::* getRang } } -ACTOR template -Future getExactRange(Reference trState, - KeyRange keys, - Key mapper, +template +Future getExactRange(Reference trStateInput, + KeyRange keysInput, + Key mapperInput, GetRangeLimits limits, Reverse reverse) { - state RangeResultFamily output; - state Span span("NAPI:getExactRange"_loc, trState->spanContext); - - loop { - state std::vector locations = - wait(getKeyRangeLocations(trState, - keys, - CLIENT_KNOBS->GET_RANGE_SHARD_LIMIT, - reverse, - getRangeRequestStream())); - ASSERT(locations.size()); - state int shard = 0; - loop { + Reference trState(std::move(trStateInput)); + KeyRange keys(std::move(keysInput)); + Key mapper(std::move(mapperInput)); + RangeResultFamily output; + Span span("NAPI:getExactRange"_loc, trState->spanContext); + std::vector locations; + GetKeyValuesFamilyReply rep; + + while (true) { + locations = co_await getKeyRangeLocations(trState, + keys, + CLIENT_KNOBS->GET_RANGE_SHARD_LIMIT, + reverse, + getRangeRequestStream()); + ASSERT(!locations.empty()); + int shard{ 0 }; + while (true) { const KeyRangeRef& range = locations[shard].range; GetKeyValuesFamilyRequest req; @@ -2346,35 +2392,40 @@ Future getExactRange(Reference trState, .detail("Servers", locations[shard].locations->locations()->description());*/ } ++trState->cx->transactionPhysicalReads; - state GetKeyValuesFamilyReply rep; + rep = GetKeyValuesFamilyReply(); try { - choose { - when(wait(trState->cx->connectionFileChanged())) { - throw transaction_too_old(); - } - when(GetKeyValuesFamilyReply _rep = wait(loadBalance( - locations[shard].locations->locations(), - getRangeRequestStream(), - req, - TaskPriority::DefaultPromiseEndpoint, - AtMostOnce::False, - trState->cx->enableLocalityLoadBalance ? &trState->cx->queueModel : nullptr, - trState->options.enableReplicaConsistencyCheck, - trState->options.requiredReplicas))) { - rep = _rep; - } + auto connectionChanged = trState->cx->connectionFileChanged(); + // An invalidated connection must not dispatch another request. + auto reply = + connectionChanged.isReady() + ? Future(Never()) + : loadBalance(locations[shard].locations->locations(), + getRangeRequestStream(), + req, + TaskPriority::DefaultPromiseEndpoint, + AtMostOnce::False, + trState->cx->enableLocalityLoadBalance ? &trState->cx->queueModel : nullptr, + trState->options.enableReplicaConsistencyCheck, + trState->options.requiredReplicas); + auto res = co_await race(std::move(connectionChanged), std::move(reply)); + connectionChanged = Future(); + reply = Future(); + if (res.index() == 0) { + throw transaction_too_old(); } + rep = std::get<1>(std::move(res)); ++trState->cx->transactionPhysicalReadsCompleted; } catch (Error&) { ++trState->cx->transactionPhysicalReadsCompleted; throw; } - if (trState->readOptions.present() && trState->readOptions.get().debugID.present()) + if (trState->readOptions.present() && trState->readOptions.get().debugID.present()) { g_traceBatch.addEvent("TransactionDebug", trState->readOptions.get().debugID.get().first(), "NativeAPI.getExactRange.After", trState->spanContext.traceID, trState->spanContext.spanID); + } output.arena().dependsOn(rep.arena); output.append(output.arena(), rep.data.begin(), rep.data.size()); @@ -2388,7 +2439,7 @@ Future getExactRange(Reference trState, if (limits.isReached()) { output.more = true; - return output; + co_return output; } bool more = rep.more; @@ -2411,15 +2462,16 @@ Future getExactRange(Reference trState, } CODE_PROBE(true, "GetKeyValuesFamilyReply.more in getExactRange"); // Make next request to the same shard with a beginning key just after the last key returned - if (reverse) + if (reverse) { locations[shard].range = KeyRangeRef(locations[shard].range.begin, output[output.size() - 1].key); - else + } else { locations[shard].range = KeyRangeRef(keyAfter(output[output.size() - 1].key), locations[shard].range.end); + } } - bool redoKeyLocationRequest = false; + bool redoKeyLocationRequest{ false }; if (!more || locations[shard].range.empty()) { CODE_PROBE(true, "getExactrange (!more || locations[shard].first.empty())"); if (shard == locations.size() - 1) { @@ -2429,7 +2481,7 @@ Future getExactRange(Reference trState, if (begin >= end) { output.more = false; - return output; + co_return output; } keys = KeyRangeRef(begin, end); @@ -2444,13 +2496,14 @@ Future getExactRange(Reference trState, // fetch entirely. if (limits.hasSatisfiedMinRows() && output.size() > 0) { output.more = true; - return output; + co_return output; } if (redoKeyLocationRequest) { CODE_PROBE(true, "Multiple requests of key locations"); break; } + continue; } catch (Error& e) { if (e.code() == error_code_wrong_shard_server || e.code() == error_code_all_alternatives_failed) { const KeyRangeRef& range = locations[shard].range; @@ -2461,9 +2514,6 @@ Future getExactRange(Reference trState, keys = KeyRangeRef(range.begin, keys.end); trState->cx->invalidateCache(keys); - - wait(delay(CLIENT_KNOBS->WRONG_SHARD_SERVER_DELAY, trState->taskID)); - break; } else { TraceEvent(SevInfo, "GetExactRangeError") .error(e) @@ -2472,6 +2522,8 @@ Future getExactRange(Reference trState, throw; } } + co_await delay(CLIENT_KNOBS->WRONG_SHARD_SERVER_DELAY, trState->taskID); + break; } } } @@ -2486,29 +2538,32 @@ Future resolveKey(Reference trState, KeySelector const& k return getKey(trState, key); } -ACTOR template -Future getRangeFallback(Reference trState, - KeySelector begin, - KeySelector end, - Key mapper, +template +Future getRangeFallback(Reference trStateInput, + KeySelector beginInput, + KeySelector endInput, + Key mapperInput, GetRangeLimits limits, Reverse reverse) { + Reference trState(std::move(trStateInput)); + KeySelector begin(std::move(beginInput)); + KeySelector end(std::move(endInput)); + Key mapper(std::move(mapperInput)); Future fb = resolveKey(trState, begin); - state Future fe = resolveKey(trState, end); + Future fe = resolveKey(trState, end); - state Key b = wait(fb); - state Key e = wait(fe); + Key b = co_await fb; + Key e = co_await fe; if (b >= e) { - return RangeResultFamily(); + co_return RangeResultFamily(); } // if e is allKeys.end, we have read through the end of the database // if b is allKeys.begin, we have either read through the beginning of the database // or allKeys.begin exists in the database and will be part of the conflict range anyways - RangeResultFamily _r = wait(getExactRange( - trState, KeyRangeRef(b, e), mapper, limits, reverse)); - RangeResultFamily r = _r; + RangeResultFamily r = co_await getExactRange( + trState, KeyRangeRef(b, e), mapper, limits, reverse); if (b == allKeys.begin && ((reverse && !r.more) || !reverse)) r.readToBegin = true; @@ -2535,7 +2590,7 @@ Future getRangeFallback(Reference trState, .detail("DeliveredRows", r.size()); } - return r; + co_return r; } int64_t inline getRangeResultFamilyBytes(RangeResultRef result) { @@ -2615,33 +2670,41 @@ void getRangeFinished(Reference trState, } } -ACTOR template -Future getRange(Reference trState, - KeySelector begin, - KeySelector end, - Key mapper, +template +Future getRange(Reference trStateInput, + KeySelector beginInput, + KeySelector endInput, + Key mapperInput, GetRangeLimits limits, - Promise> conflictRange, + Promise> conflictRangeInput, Snapshot snapshot, Reverse reverse) { - // state using RangeResultRefFamily = typename RangeResultFamily::RefType; - state GetRangeLimits originalLimits(limits); - state KeySelector originalBegin = begin; - state KeySelector originalEnd = end; - state RangeResultFamily output; - state Span span("NAPI:getRange"_loc, trState->spanContext); - state Optional getRangeID = Optional(); + Reference trState(std::move(trStateInput)); + KeySelector begin(std::move(beginInput)); + KeySelector end(std::move(endInput)); + Key mapper(std::move(mapperInput)); + Promise> conflictRange(std::move(conflictRangeInput)); + GetRangeLimits originalLimits(limits); + KeySelector originalBegin = begin; + KeySelector originalEnd = end; + RangeResultFamily output; + Span span("NAPI:getRange"_loc, trState->spanContext); + Optional getRangeID; + KeyRangeLocationInfo beginServer; + KeyRange shard; + GetKeyValuesFamilyRequest req; + GetKeyValuesFamilyReply rep; try { - wait(trState->startTransaction()); + co_await trState->startTransaction(); trState->cx->validateVersion(trState->readVersion()); - state double startTime = now(); + double startTime = now(); if (begin.getKey() == allKeys.begin && begin.offset < 1) { output.readToBegin = true; @@ -2651,20 +2714,20 @@ Future getRange(Reference trState, ASSERT(!limits.isReached()); ASSERT((!limits.hasRowLimit() || limits.rows >= limits.minRows) && limits.minRows >= 0); - loop { + while (true) { if (end.getKey() == allKeys.begin && (end.offset < 1 || end.isFirstGreaterOrEqual())) { getRangeFinished( trState, startTime, originalBegin, originalEnd, snapshot, conflictRange, reverse, output); - return output; + co_return output; } Key locationKey = reverse ? Key(end.getKey(), end.arena()) : Key(begin.getKey(), begin.arena()); Reverse locationBackward{ reverse ? (end - 1).isBackward() : begin.isBackward() }; - state KeyRangeLocationInfo beginServer = wait(getKeyLocation( - trState, locationKey, getRangeRequestStream(), locationBackward)); - state KeyRange shard = beginServer.range; - state bool modifiedSelectors = false; - state GetKeyValuesFamilyRequest req; + beginServer = co_await getKeyLocation( + trState, locationKey, getRangeRequestStream(), locationBackward); + shard = beginServer.range; + bool modifiedSelectors{ false }; + req = GetKeyValuesFamilyRequest(); req.mapper = mapper; req.arena.dependsOn(mapper.arena()); req.options = trState->readOptions; @@ -2675,7 +2738,7 @@ Future getRange(Reference trState, // In case of async tss comparison, also make req arena depend on begin, end, and/or shard's arena depending // on which is used - bool dependOnShard = false; + bool dependOnShard{ false }; if (reverse && (begin - 1).isDefinitelyLess(shard.begin) && (!begin.isFirstGreaterOrEqual() || begin.getKey() != shard.begin)) { // In this case we would be setting modifiedSelectors to true, but @@ -2714,6 +2777,7 @@ Future getRange(Reference trState, trState->spanContext.traceID, trState->spanContext.spanID); } + Error err; try { if (getRangeID.present()) { g_traceBatch.addEvent("TransactionDebug", @@ -2741,23 +2805,21 @@ Future getRange(Reference trState, } ++trState->cx->transactionPhysicalReads; - state GetKeyValuesFamilyReply rep; + rep = GetKeyValuesFamilyReply(); try { if (CLIENT_BUGGIFY_WITH_PROB(.01)) { throw deterministicRandom()->randomChoice( std::vector{ transaction_too_old(), future_version() }); } - // state AnnotateActor annotation(currentLineage); - GetKeyValuesFamilyReply _rep = - wait(loadBalance(beginServer.locations->locations(), - getRangeRequestStream(), - req, - TaskPriority::DefaultPromiseEndpoint, - AtMostOnce::False, - trState->cx->enableLocalityLoadBalance ? &trState->cx->queueModel : nullptr, - trState->options.enableReplicaConsistencyCheck, - trState->options.requiredReplicas)); - rep = _rep; + rep = co_await loadBalance(beginServer.locations->locations(), + getRangeRequestStream(), + req, + TaskPriority::DefaultPromiseEndpoint, + AtMostOnce::False, + trState->cx->enableLocalityLoadBalance ? &trState->cx->queueModel + : nullptr, + trState->options.enableReplicaConsistencyCheck, + trState->options.requiredReplicas); ++trState->cx->transactionPhysicalReadsCompleted; } catch (Error&) { ++trState->cx->transactionPhysicalReadsCompleted; @@ -2821,7 +2883,7 @@ Future getRange(Reference trState, getRangeFinished( trState, startTime, originalBegin, originalEnd, snapshot, conflictRange, reverse, output); - return output; + co_return output; } if (readThrough) { @@ -2837,7 +2899,7 @@ Future getRange(Reference trState, if (!output.more) { ASSERT(!output.readThrough.present()); } - return output; + co_return output; } output.arena().dependsOn(rep.arena); @@ -2855,7 +2917,7 @@ Future getRange(Reference trState, if (!output.more) { ASSERT(!output.readThrough.present()); } - return output; + co_return output; } if (!rep.more) { @@ -2863,12 +2925,13 @@ Future getRange(Reference trState, CODE_PROBE(true, "!GetKeyValuesFamilyReply.more and modifiedSelectors in getRange"); if (!rep.data.size()) { - RangeResultFamily result = wait( - getRangeFallback( - trState, originalBegin, originalEnd, mapper, originalLimits, reverse)); + RangeResultFamily result = co_await getRangeFallback( + trState, originalBegin, originalEnd, mapper, originalLimits, reverse); getRangeFinished( trState, startTime, originalBegin, originalEnd, snapshot, conflictRange, reverse, result); - return result; + co_return result; } if (reverse) @@ -2882,41 +2945,44 @@ Future getRange(Reference trState, else begin = firstGreaterThan(output[output.size() - 1].key); } - + continue; } catch (Error& e) { - if (getRangeID.present()) { - g_traceBatch.addEvent("TransactionDebug", - getRangeID.get().first(), - "NativeAPI.getRange.Error", - trState->spanContext.traceID, - trState->spanContext.spanID); - TraceEvent("TransactionDebugError", getRangeID.get()).error(e); + err = e; + } + if (getRangeID.present()) { + g_traceBatch.addEvent("TransactionDebug", + getRangeID.get().first(), + "NativeAPI.getRange.Error", + trState->spanContext.traceID, + trState->spanContext.spanID); + TraceEvent("TransactionDebugError", getRangeID.get()).error(err); + } + if (err.code() == error_code_wrong_shard_server || err.code() == error_code_all_alternatives_failed) { + trState->cx->invalidateCache(reverse ? end.getKey() : begin.getKey(), + Reverse{ reverse ? (end - 1).isBackward() : begin.isBackward() }); + + if (err.code() == error_code_wrong_shard_server) { + RangeResultFamily result = co_await getRangeFallback( + trState, originalBegin, originalEnd, mapper, originalLimits, reverse); + getRangeFinished( + trState, startTime, originalBegin, originalEnd, snapshot, conflictRange, reverse, result); + co_return result; } - if (e.code() == error_code_wrong_shard_server || e.code() == error_code_all_alternatives_failed) { - trState->cx->invalidateCache(reverse ? end.getKey() : begin.getKey(), - Reverse{ reverse ? (end - 1).isBackward() : begin.isBackward() }); - if (e.code() == error_code_wrong_shard_server) { - RangeResultFamily result = wait( - getRangeFallback( - trState, originalBegin, originalEnd, mapper, originalLimits, reverse)); - getRangeFinished( - trState, startTime, originalBegin, originalEnd, snapshot, conflictRange, reverse, result); - return result; - } - - wait(delay(CLIENT_KNOBS->WRONG_SHARD_SERVER_DELAY, trState->taskID)); - } else { - if (trState->trLogInfo) - trState->trLogInfo->addLog( - FdbClientLogEvents::EventGetRangeError(startTime, - trState->cx->clientLocality.dcId(), - static_cast(e.code()), - begin.getKey(), - end.getKey()), - trState->spanContext); - throw e; + co_await delay(CLIENT_KNOBS->WRONG_SHARD_SERVER_DELAY, trState->taskID); + } else { + if (trState->trLogInfo) { + trState->trLogInfo->addLog( + FdbClientLogEvents::EventGetRangeError(startTime, + trState->cx->clientLocality.dcId(), + static_cast(err.code()), + begin.getKey(), + end.getKey()), + trState->spanContext); } + throw err; } } } catch (Error& e) { @@ -2951,26 +3017,30 @@ struct TSSDuplicateStreamData { // Error tracking here is weird, and latency doesn't really mean the same thing here as it does with normal tss // comparisons, so this is pretty much just counting mismatches -ACTOR template -static Future tssStreamComparison(Request request, - TSSDuplicateStreamData streamData, - ReplyPromiseStream tssReplyStream, - TSSEndpointData tssData) { - state bool ssEndOfStream = false; - state bool tssEndOfStream = false; - state Optional ssReply = Optional(); - state Optional tssReply = Optional(); - - loop { +template +static Future tssStreamComparison(Request requestInput, + TSSDuplicateStreamData streamDataInput, + ReplyPromiseStream tssReplyStreamInput, + TSSEndpointData tssDataInput) { + Request request(std::move(requestInput)); + TSSDuplicateStreamData streamData(std::move(streamDataInput.stream)); + streamData.tssComparisonDone = std::move(streamDataInput.tssComparisonDone); + ReplyPromiseStream tssReplyStream(std::move(tssReplyStreamInput)); + TSSEndpointData tssData(std::move(tssDataInput)); + bool ssEndOfStream{ false }; + bool tssEndOfStream{ false }; + Optional ssReply; + Optional tssReply; + + while (true) { // reset replies ssReply = Optional(); tssReply = Optional(); - state double startTime = now(); + double startTime = now(); // wait for ss response try { - REPLYSTREAM_TYPE(Request) _ssReply = waitNext(streamData.stream.getFuture()); - ssReply = _ssReply; + ssReply = co_await streamData.stream.getFuture(); } catch (Error& e) { if (e.code() == error_code_actor_cancelled) { streamData.setDone(); @@ -2986,17 +3056,19 @@ static Future tssStreamComparison(Request request, CODE_PROBE(e.code() != error_code_end_of_stream, "SS got error in TSS stream comparison"); } - state double sleepTime = std::max(startTime + FLOW_KNOBS->LOAD_BALANCE_TSS_TIMEOUT - now(), 0.0); + double sleepTime = std::max(startTime + FLOW_KNOBS->LOAD_BALANCE_TSS_TIMEOUT - now(), 0.0); // wait for tss response try { - choose { - when(REPLYSTREAM_TYPE(Request) _tssReply = waitNext(tssReplyStream.getFuture())) { - tssReply = _tssReply; - } - when(wait(delay(sleepTime))) { - ++tssData.metrics->tssTimeouts; - CODE_PROBE(true, "Got TSS timeout in stream comparison", probe::decoration::rare); - } + auto reply = tssReplyStream.getFuture(); + auto timeout = reply.isReady() ? Future(Never()) : delay(sleepTime); + auto res = co_await race(std::move(reply), std::move(timeout)); + reply = FutureStream(); + timeout = Future(); + if (res.index() == 0) { + tssReply = std::get<0>(std::move(res)); + } else { + ++tssData.metrics->tssTimeouts; + CODE_PROBE(true, "Got TSS timeout in stream comparison", probe::decoration::rare); } } catch (Error& e) { if (e.code() == error_code_actor_cancelled) { @@ -3063,13 +3135,13 @@ static Future tssStreamComparison(Request request, mismatchEvent.disable(); } streamData.setDone(); - return Void(); + co_return; } } if (!ssReply.present() || !tssReply.present() || ssEndOfStream || tssEndOfStream) { // if both streams don't still have more data, stop comparison streamData.setDone(); - return Void(); + co_return; } } } @@ -3100,23 +3172,32 @@ Optional> maybeDuplicateTSSStr } // Streams all of the KV pairs in a target key range directly to the client in order. -ACTOR Future getRangeStreamImpl(Reference trState, - PromiseStream results, - KeyRange keys, - GetRangeLimits limits, - Snapshot snapshot, - Reverse reverse, - SpanContext spanContext) { - loop { - state std::vector locations = wait(getKeyRangeLocations( - trState, keys, CLIENT_KNOBS->GET_RANGE_SHARD_LIMIT, reverse, &StorageServerInterface::getKeyValuesStream)); - ASSERT(locations.size()); - state int shard = 0; - loop { +Future getRangeStreamImpl(Reference trStateInput, + PromiseStream resultsInput, + KeyRange keysInput, + GetRangeLimits limits, + Snapshot snapshot, + Reverse reverse, + SpanContext spanContext) { + Reference trState(std::move(trStateInput)); + PromiseStream results(std::move(resultsInput)); + KeyRange keys(std::move(keysInput)); + std::vector locations; + // Keep the active streams alive through retry delays and location refreshes. + Optional> tssDuplicateStream; + GetKeyValuesStreamRequest req; + GetKeyValuesStreamReply rep; + ReplyPromiseStream replyStream; + while (true) { + locations = co_await getKeyRangeLocations( + trState, keys, CLIENT_KNOBS->GET_RANGE_SHARD_LIMIT, reverse, &StorageServerInterface::getKeyValuesStream); + ASSERT(!locations.empty()); + int shard{ 0 }; + while (true) { const KeyRange& range = locations[shard].range; + tssDuplicateStream = Optional>(); + req = GetKeyValuesStreamRequest(); - state Optional> tssDuplicateStream; - state GetKeyValuesStreamRequest req; req.version = trState->readVersion(); req.begin = firstGreaterOrEqual(range.begin); req.end = firstGreaterOrEqual(range.end); @@ -3144,20 +3225,20 @@ ACTOR Future getRangeStreamImpl(Reference trState, trState->spanContext.spanID); } ++trState->cx->transactionPhysicalReads; - state GetKeyValuesStreamReply rep; + rep = GetKeyValuesStreamReply(); if (locations[shard].locations->size() == 0) { - wait(trState->cx->connectionFileChanged()); + co_await trState->cx->connectionFileChanged(); results.sendError(transaction_too_old()); - return Void(); + co_return; } - state int useIdx = -1; + int useIdx = -1; - loop { + while (true) { // FIXME: create a load balance function for this code so future users of reply streams do not have // to duplicate this code - int count = 0; + int count{ 0 }; for (int i = 0; i < locations[shard].locations->size(); i++) { if (!IFailureMonitor::failureMonitor() .getState(locations[shard] @@ -3190,36 +3271,35 @@ ACTOR Future getRangeStreamImpl(Reference trState, .detail("Alternatives", locations[shard].locations->description()); } - wait(allAlternativesFailedDelay(quorum(ok, 1))); + co_await allAlternativesFailedDelay(quorum(ok, 1)); } - state ReplyPromiseStream replyStream = - locations[shard] - .locations->get(useIdx, &StorageServerInterface::getKeyValuesStream) - .getReplyStream(req); + replyStream = locations[shard] + .locations->get(useIdx, &StorageServerInterface::getKeyValuesStream) + .getReplyStream(req); tssDuplicateStream = maybeDuplicateTSSStreamFragment( req, trState->cx->enableLocalityLoadBalance ? &trState->cx->queueModel : nullptr, &locations[shard].locations->get(useIdx, &StorageServerInterface::getKeyValuesStream)); - state bool breakAgain = false; - loop { - wait(results.onEmpty()); + bool breakAgain{ false }; + while (true) { + co_await results.onEmpty(); try { - choose { - when(wait(trState->cx->connectionFileChanged())) { - results.sendError(transaction_too_old()); - if (tssDuplicateStream.present() && !tssDuplicateStream.get().done()) { - tssDuplicateStream.get().stream.sendError(transaction_too_old()); - } - return Void(); - } - - when(GetKeyValuesStreamReply _rep = waitNext(replyStream.getFuture())) { - rep = _rep; + auto connectionChanged = trState->cx->connectionFileChanged(); + auto reply = replyStream.getFuture(); + auto res = co_await race(std::move(connectionChanged), std::move(reply)); + connectionChanged = Future(); + reply = FutureStream(); + if (res.index() == 0) { + results.sendError(transaction_too_old()); + if (tssDuplicateStream.present() && !tssDuplicateStream.get().done()) { + tssDuplicateStream.get().stream.sendError(transaction_too_old()); } + co_return; } + rep = std::get<1>(std::move(res)); ++trState->cx->transactionPhysicalReadsCompleted; } catch (Error& e) { ++trState->cx->transactionPhysicalReadsCompleted; @@ -3237,12 +3317,13 @@ ACTOR Future getRangeStreamImpl(Reference trState, } rep = GetKeyValuesStreamReply(); } - if (trState->readOptions.present() && trState->readOptions.get().debugID.present()) + if (trState->readOptions.present() && trState->readOptions.get().debugID.present()) { g_traceBatch.addEvent("TransactionDebug", trState->readOptions.get().debugID.get().first(), "NativeAPI.getExactRange.After", trState->spanContext.traceID, trState->spanContext.spanID); + } RangeResult output(RangeResultRef(rep.data, rep.more), rep.arena); if (tssDuplicateStream.present() && !tssDuplicateStream.get().done()) { @@ -3256,7 +3337,7 @@ ACTOR Future getRangeStreamImpl(Reference trState, tssDuplicateStream.get().stream.send(replyCopy); } - int64_t bytes = 0; + int64_t bytes{ 0 }; for (const KeyValueRef& kv : output) { bytes += kv.key.size() + kv.value.size(); } @@ -3265,13 +3346,13 @@ ACTOR Future getRangeStreamImpl(Reference trState, trState->cx->transactionKeysRead += output.size(); // If the reply says there is more but we know that we finished the shard, then fix rep.more - if (reverse && output.more && rep.data.size() > 0 && + if (reverse && output.more && !rep.data.empty() && output[output.size() - 1].key == locations[shard].range.begin) { output.more = false; } if (output.more) { - if (!rep.data.size()) { + if (rep.data.empty()) { TraceEvent(SevError, "GetRangeStreamError") .detail("Reason", "More data indicated but no rows present") .detail("LimitBytes", limits.bytes) @@ -3284,12 +3365,13 @@ ACTOR Future getRangeStreamImpl(Reference trState, } CODE_PROBE(true, "GetKeyValuesStreamReply.more in getRangeStream"); // Make next request to the same shard with a beginning key just after the last key returned - if (reverse) + if (reverse) { locations[shard].range = KeyRangeRef(locations[shard].range.begin, output[output.size() - 1].key); - else + } else { locations[shard].range = KeyRangeRef(keyAfter(output[output.size() - 1].key), locations[shard].range.end); + } } if (locations[shard].range.empty()) { @@ -3318,7 +3400,7 @@ ACTOR Future getRangeStreamImpl(Reference trState, if (tssDuplicateStream.present() && !tssDuplicateStream.get().done()) { tssDuplicateStream.get().stream.sendError(end_of_stream()); } - return Void(); + co_return; } keys = KeyRangeRef(begin, end); breakAgain = true; @@ -3333,7 +3415,7 @@ ACTOR Future getRangeStreamImpl(Reference trState, break; } - ASSERT(output.size()); + ASSERT(!output.empty()); if (keys.begin == allKeys.begin && !reverse) { output.readToBegin = true; } @@ -3345,6 +3427,7 @@ ACTOR Future getRangeStreamImpl(Reference trState, if (breakAgain) { break; } + continue; } catch (Error& e) { // send errors to tss duplicate stream, including actor_cancelled if (tssDuplicateStream.present() && !tssDuplicateStream.get().done()) { @@ -3363,44 +3446,48 @@ ACTOR Future getRangeStreamImpl(Reference trState, keys = KeyRangeRef(range.begin, keys.end); trState->cx->invalidateCache(keys); - - wait(delay(CLIENT_KNOBS->WRONG_SHARD_SERVER_DELAY, trState->taskID)); - break; } else { results.sendError(e); - return Void(); + co_return; } } + co_await delay(CLIENT_KNOBS->WRONG_SHARD_SERVER_DELAY, trState->taskID); + break; } } } -ACTOR Future>> getRangeSplitPoints(Reference trState, - KeyRange keys, - int64_t chunkSize, - int limit); +Future>> getRangeSplitPoints(Reference trState, + KeyRange keys, + int64_t chunkSize, + int limit); // Streams the requested key range directly from storage servers without fragment-level parallelism. -ACTOR Future getRangeStream(Reference trState, - PromiseStream _results, - KeySelector begin, - KeySelector end, - GetRangeLimits limits, - Promise> conflictRange, - Snapshot snapshot, - Reverse reverse) { +Future getRangeStream(Reference trStateInput, + PromiseStream _resultsInput, + KeySelector beginInput, + KeySelector endInput, + GetRangeLimits limits, + Promise> conflictRangeInput, + Snapshot snapshot, + Reverse reverse) { + Reference trState(std::move(trStateInput)); + PromiseStream _results(std::move(_resultsInput)); + KeySelector begin(std::move(beginInput)); + KeySelector end(std::move(endInput)); + Promise> conflictRange(std::move(conflictRangeInput)); // FIXME: better handling to disable row limits ASSERT(!limits.hasRowLimit()); - state Span span("NAPI:getRangeStream"_loc, trState->spanContext); + Span span("NAPI:getRangeStream"_loc, trState->spanContext); - wait(trState->startTransaction()); + co_await trState->startTransaction(); trState->cx->validateVersion(trState->readVersion()); Future fb = resolveKey(trState, begin); - state Future fe = resolveKey(trState, end); + Future fe = resolveKey(trState, end); - state Key b = wait(fb); - state Key e = wait(fe); + Key b = co_await fb; + Key e = co_await fe; if (!snapshot) { // FIXME: this conflict range is too large, and should be updated continuously as results are returned @@ -3410,11 +3497,11 @@ ACTOR Future getRangeStream(Reference trState, if (b >= e) { _results.sendError(end_of_stream()); - return Void(); + co_return; } - wait(getRangeStreamImpl(trState, _results, KeyRange(KeyRangeRef(b, e)), limits, snapshot, reverse, span.context)); - return Void(); + co_await getRangeStreamImpl( + trState, _results, KeyRange(KeyRangeRef(b, e)), limits, snapshot, reverse, span.context); } Future getRange(Reference const& trState, @@ -3588,42 +3675,35 @@ Future restartWatch(Database cx, } // FIXME: This seems pretty horrible. Now a Database can't die until all of its watches do... -ACTOR Future watch(Reference watch, - Database cx, - TagSet tags, - SpanContext spanContext, - TaskPriority taskID, - Optional debugID, - UseProvisionalProxies useProvisionalProxies) { +Future watch(Reference watchInput, + Database cxInput, + TagSet tagsInput, + SpanContext spanContext, + TaskPriority taskID, + Optional debugID, + UseProvisionalProxies useProvisionalProxies) { + Reference watch(std::move(watchInput)); + Database cx(std::move(cxInput)); + TagSet tags(std::move(tagsInput)); try { - choose { - // RYOW write to value that is being watched (if applicable) - // Errors - when(wait(watch->onChangeTrigger.getFuture())) {} - - // NativeAPI finished commit and updated watchFuture - when(wait(watch->onSetWatchTrigger.getFuture())) { - - loop { - choose { - // NativeAPI watchValue future finishes or errors - when(wait(watch->watchFuture)) { - break; - } - - when(wait(cx->connectionFileChanged())) { - CODE_PROBE(true, "Recreated a watch after switch"); - watch->watchFuture = restartWatch(cx, - watch->key, - watch->value, - tags, - spanContext, - taskID, - debugID, - useProvisionalProxies); - } - } + // RYOW write to value that is being watched (if applicable), or errors. + auto changed = watch->onChangeTrigger.getFuture(); + // NativeAPI finished commit and updated watchFuture. + auto watchSet = changed.isReady() ? Future(Never()) : watch->onSetWatchTrigger.getFuture(); + auto res = co_await race(std::exchange(changed, {}), std::exchange(watchSet, {})); + if (res.index() == 1) { + while (true) { + // NativeAPI watchValue future finishes or errors. + auto pendingWatch = watch->watchFuture; + auto connectionChanged = pendingWatch.isReady() ? Future(Never()) : cx->connectionFileChanged(); + // Release the selection's copy before replacing watchFuture on a connection change. + auto watchResult = co_await race(std::exchange(pendingWatch, {}), std::exchange(connectionChanged, {})); + if (watchResult.index() == 0) { + break; } + CODE_PROBE(true, "Recreated a watch after switch"); + watch->watchFuture = restartWatch( + cx, watch->key, watch->value, tags, spanContext, taskID, debugID, useProvisionalProxies); } } } catch (Error& e) { @@ -3632,7 +3712,6 @@ ACTOR Future watch(Reference watch, } cx->decreaseWatchCounter(); - return Void(); } Future Transaction::getRawReadVersion() { @@ -3674,7 +3753,7 @@ Future>> getAddressesForKeyActor(Reference src; std::vector ignore; // 'ignore' is so named because it is the vector into which we decode the 'dest' servers in @@ -3689,7 +3768,7 @@ Future>> getAddressesForKeyActor(Reference> addresses; - for (auto i : ssi) { + for (const auto& i : ssi) { std::string ipString = trState->options.includePort ? i.address().toString() : i.address().ip.toString(); char* c_string = new (addresses.arena()) char[ipString.length() + 1]; strcpy(c_string, ipString.c_str()); @@ -3704,17 +3783,20 @@ Future>> Transaction::getAddressesForKey(const return getAddressesForKeyActor(trState, key); } -ACTOR Future getKeyAndConflictRange(Reference trState, - KeySelector k, - Promise> conflictRange) { +Future getKeyAndConflictRange(Reference trStateInput, + KeySelector kInput, + Promise> conflictRangeInput) { + Reference trState(std::move(trStateInput)); + KeySelector k(std::move(kInput)); + Promise> conflictRange(std::move(conflictRangeInput)); try { - Key rep = wait(getKey(trState, k)); + Key rep = co_await getKey(trState, k); if (k.offset <= 0) conflictRange.send(std::make_pair(rep, k.orEqual ? keyAfter(k.getKey()) : Key(k.getKey(), k.arena()))); else conflictRange.send( std::make_pair(k.orEqual ? keyAfter(k.getKey()) : Key(k.getKey(), k.arena()), keyAfter(rep))); - return rep; + co_return rep; } catch (Error& e) { conflictRange.send(std::make_pair(Key(), Key())); throw; @@ -4165,7 +4247,7 @@ bool compareBegin(KeyRangeRef lhs, KeyRangeRef rhs) { // If there is any intersection between the two given sets of ranges, returns a range that // falls within the intersection Optional intersects(VectorRef lhs, VectorRef rhs) { - if (lhs.size() && rhs.size()) { + if (!lhs.empty() && !rhs.empty()) { std::sort(lhs.begin(), lhs.end(), compareBegin); std::sort(rhs.begin(), rhs.end(), compareBegin); @@ -4225,7 +4307,7 @@ Future checkWrites(Uncancellable, checkedRanges++; if (m.cleared) { RangeResult shouldBeEmpty = co_await tr.getRange(it->range(), 1); - if (shouldBeEmpty.size()) { + if (!shouldBeEmpty.empty()) { TraceEvent(SevError, "CheckWritesFailed") .detail("Class", "Clear") .detail("KeyBegin", it->range().begin) @@ -4366,7 +4448,7 @@ void Transaction::setupWatches() { try { Future watchVersion = getCommittedVersion() > 0 ? getCommittedVersion() : getReadVersion(); - for (auto& watch : watches) + for (auto& watch : watches) { watch->setWatch( watchValueMap(watchVersion, watch->key, @@ -4377,6 +4459,7 @@ void Transaction::setupWatches() { trState->taskID, trState->readOptions.present() ? trState->readOptions.get().debugID : Optional(), trState->useProvisionalProxies)); + } watches.clear(); } catch (Error&) { @@ -4385,13 +4468,12 @@ void Transaction::setupWatches() { } } -ACTOR Future> estimateCommitCosts(Reference trState, - CommitTransactionRef const* transaction) { - state ClientTrCommitCostEstimation trCommitCosts; - state KeyRangeRef keyRange; - state int i = 0; +Future> estimateCommitCosts(Reference trStateInput, + CommitTransactionRef const* transaction) { + Reference trState(std::move(trStateInput)); + ClientTrCommitCostEstimation trCommitCosts; - for (; i < transaction->mutations.size(); ++i) { + for (int i = 0; i < transaction->mutations.size(); ++i) { auto const& mutation = transaction->mutations[i]; if (mutation.type == MutationRef::Type::SetValue || mutation.isAtomicOp()) { @@ -4399,16 +4481,16 @@ ACTOR Future> estimateCommitCosts(Referen trCommitCosts.writeCosts += getWriteOperationCost(mutation.expectedSize()); } else if (mutation.type == MutationRef::Type::ClearRange) { trCommitCosts.opsCount++; - keyRange = KeyRangeRef(mutation.param1, mutation.param2); + KeyRangeRef keyRange(mutation.param1, mutation.param2); if (trState->options.expensiveClearCostEstimation) { - StorageMetrics m = wait(trState->cx->getStorageMetrics(keyRange, CLIENT_KNOBS->TOO_MANY, trState)); + StorageMetrics m = co_await trState->cx->getStorageMetrics(keyRange, CLIENT_KNOBS->TOO_MANY, trState); trCommitCosts.clearIdxCosts.emplace_back(i, getWriteOperationCost(m.bytes)); trCommitCosts.writeCosts += getWriteOperationCost(m.bytes); ++trCommitCosts.expensiveCostEstCount; ++trState->cx->transactionsExpensiveClearCostEstCount; } else { - std::vector locations = wait(getKeyRangeLocations( - trState, keyRange, CLIENT_KNOBS->TOO_MANY, Reverse::False, &StorageServerInterface::getShardState)); + std::vector locations = co_await getKeyRangeLocations( + trState, keyRange, CLIENT_KNOBS->TOO_MANY, Reverse::False, &StorageServerInterface::getShardState); if (locations.empty()) { continue; } @@ -4417,7 +4499,7 @@ ACTOR Future> estimateCommitCosts(Referen if (locations.size() == 1) { bytes = CLIENT_KNOBS->INCOMPLETE_SHARD_PLUS; } else { // small clear on the boundary will hit two shards but be much smaller than the shard size - bytes = CLIENT_KNOBS->INCOMPLETE_SHARD_PLUS * 2 + + bytes = static_cast(CLIENT_KNOBS->INCOMPLETE_SHARD_PLUS) * 2 + (locations.size() - 2) * (int64_t)trState->cx->smoothMidShardSize.smoothTotal(); } @@ -4429,7 +4511,7 @@ ACTOR Future> estimateCommitCosts(Referen // sample on written bytes if (!trState->cx->sampleOnCost(trCommitCosts.writeCosts)) - return Optional(); + co_return Optional(); // sample clear op: the expectation of #sampledOp is every COMMIT_SAMPLE_COST sample once // we also scale the cost of mutations whose cost is less than COMMIT_SAMPLE_COST as scaledCost = @@ -4453,34 +4535,42 @@ ACTOR Future> estimateCommitCosts(Referen } trCommitCosts.clearIdxCosts.swap(newClearIdxCosts); - return trCommitCosts; + co_return trCommitCosts; } -ACTOR static Future tryCommit(Reference trState, CommitTransactionRequest req) { - state TraceInterval interval("TransactionCommit"); - state double startTime = now(); - state Span span("NAPI:tryCommit"_loc, trState->spanContext); - state Optional debugID = trState->readOptions.present() ? trState->readOptions.get().debugID : Optional(); +static Future tryCommit(Reference trStateInput, CommitTransactionRequest reqInput) { + Reference trState(std::move(trStateInput)); + CommitTransactionRequest req(std::move(reqInput)); + TraceInterval interval("TransactionCommit"); + double startTime = now(); + Span span("NAPI:tryCommit"_loc, trState->spanContext); + Optional debugID = trState->readOptions.present() ? trState->readOptions.get().debugID : Optional(); if (debugID.present()) { TraceEvent(interval.begin()).detail("Parent", debugID.get()); } // If the read version hasn't already been fetched, then we had no reads and don't need (expensive) full causal // consistency. - state Future startFuture = trState->startTransaction(GetReadVersionRequest::FLAG_CAUSAL_READ_RISKY); - + Future startFuture = trState->startTransaction(GetReadVersionRequest::FLAG_CAUSAL_READ_RISKY); + + Future> commitCostFuture; + // Keep the commit RPC and its output storage alive while resolving an ambiguous result. + int alternativeChosen{ 0 }; + // Only valid if alternativeChosen >= 0. + Reference proxiesUsed; + Future reply; + Error err; try { if (CLIENT_BUGGIFY) { throw deterministicRandom()->randomChoice(std::vector{ not_committed(), transaction_too_old() }); } if (req.tagSet.present() && trState->options.priority < TransactionPriority::IMMEDIATE) { - state Future> commitCostFuture = - estimateCommitCosts(trState, &req.transaction); - wait(startFuture); - wait(store(req.commitCostEstimation, commitCostFuture)); + commitCostFuture = estimateCommitCosts(trState, &req.transaction); + co_await startFuture; + req.commitCostEstimation = co_await commitCostFuture; } else { - wait(startFuture); + co_await startFuture; } req.transaction.read_snapshot = trState->readVersion(); @@ -4490,7 +4580,7 @@ ACTOR static Future tryCommit(Reference trState, CommitT } startTime = now(); - state Optional commitID = Optional(); + Optional commitID = Optional(); if (debugID.present()) { commitID = nondeterministicRandom()->randomUniqueID(); @@ -4507,12 +4597,9 @@ ACTOR static Future tryCommit(Reference trState, CommitT } req.debugID = commitID; - state Future reply; // Only gets filled in in the happy path where we don't have to commit on the first proxy or use provisional // proxies - state int alternativeChosen = -1; - // Only valid if alternativeChosen >= 0 - state Reference proxiesUsed; + alternativeChosen = -1; if (trState->options.commitOnFirstProxy) { if (trState->cx->clientInfo->get().firstCommitProxy.present()) { @@ -4520,8 +4607,9 @@ ACTOR static Future tryCommit(Reference trState, CommitT trState->cx->clientInfo->get().firstCommitProxy.get().commit.tryGetReply(req))); } else { const std::vector& proxies = trState->cx->clientInfo->get().commitProxies; - reply = proxies.size() ? throwErrorOr(brokenPromiseToMaybeDelivered(proxies[0].commit.tryGetReply(req))) - : Never(); + reply = !proxies.empty() + ? throwErrorOr(brokenPromiseToMaybeDelivered(proxies[0].commit.tryGetReply(req))) + : Never(); } } else { proxiesUsed = trState->cx->getCommitProxies(trState->useProvisionalProxies); @@ -4532,167 +4620,176 @@ ACTOR static Future tryCommit(Reference trState, CommitT AtMostOnce::True, &alternativeChosen); } - state double grvTime = now(); - choose { - when(wait(trState->cx->onProxiesChanged())) { - reply.cancel(); - throw request_maybe_delivered(); - } - when(CommitID ci = wait(reply)) { - Version v = ci.version; - if (v != invalidVersion) { - if (CLIENT_BUGGIFY) { - throw commit_unknown_result(); - } - trState->cx->updateCachedReadVersion(grvTime, v); - if (debugID.present()) - TraceEvent(interval.end()).detail("CommittedVersion", v); - trState->committedVersion = v; - if (v > trState->cx->metadataVersionCache[trState->cx->mvCacheInsertLocation].first) { - trState->cx->mvCacheInsertLocation = - (trState->cx->mvCacheInsertLocation + 1) % trState->cx->metadataVersionCache.size(); - trState->cx->metadataVersionCache[trState->cx->mvCacheInsertLocation] = - std::make_pair(v, ci.metadataVersion); - } + double grvTime = now(); - Standalone ret = makeString(10); - placeVersionstamp(mutateString(ret), v, ci.txnBatchId); - trState->versionstampPromise.send(ret); + auto res = co_await race(trState->cx->onProxiesChanged(), reply); + if (res.index() == 0) { + reply.cancel(); + throw request_maybe_delivered(); + } else if (res.index() == 1) { + CommitID ci = std::get<1>(std::move(res)); - trState->numErrors = 0; - ++trState->cx->transactionsCommitCompleted; - trState->cx->transactionCommittedMutations += req.transaction.mutations.size(); - trState->cx->transactionCommittedMutationBytes += req.transaction.mutations.expectedSize(); + Version v = ci.version; + if (v != invalidVersion) { + if (CLIENT_BUGGIFY) { + throw commit_unknown_result(); + } + trState->cx->updateCachedReadVersion(grvTime, v); + if (debugID.present()) + TraceEvent(interval.end()).detail("CommittedVersion", v); + trState->committedVersion = v; + if (v > trState->cx->metadataVersionCache[trState->cx->mvCacheInsertLocation].first) { + trState->cx->mvCacheInsertLocation = + (trState->cx->mvCacheInsertLocation + 1) % trState->cx->metadataVersionCache.size(); + trState->cx->metadataVersionCache[trState->cx->mvCacheInsertLocation] = + std::make_pair(v, ci.metadataVersion); + } - if (commitID.present()) - g_traceBatch.addEvent("CommitDebug", - commitID.get().first(), - "NativeAPI.commit.After", - trState->spanContext.traceID, - trState->spanContext.spanID); + Standalone ret = makeString(10); + placeVersionstamp(mutateString(ret), v, ci.txnBatchId); + trState->versionstampPromise.send(ret); - double latency = now() - startTime; - trState->cx->commitLatencies.addSample(latency); - trState->cx->latencies.addSample(now() - trState->startTime); - if (trState->trLogInfo) - trState->trLogInfo->addLog( - FdbClientLogEvents::EventCommit_V2(startTime, - trState->cx->clientLocality.dcId(), - latency, - req.transaction.mutations.size(), - req.transaction.mutations.expectedSize(), - ci.version, - req), - trState->spanContext); - if (trState->automaticIdempotency && alternativeChosen >= 0) { - // Automatic idempotency means we're responsible for best effort idempotency id clean up - proxiesUsed->getInterface(alternativeChosen) - .expireIdempotencyId.send( - ExpireIdempotencyIdRequest{ ci.version, uint8_t(ci.txnBatchId >> 8) }); - } - return Void(); - } else { - // clear the RYW transaction which contains previous conflicting keys - trState->conflictingKeys.reset(); - if (ci.conflictingKRIndices.present()) { - trState->conflictingKeys = - std::make_shared>(conflictingKeysFalse, specialKeys.end); - state Standalone> conflictingKRIndices = ci.conflictingKRIndices.get(); - // drop duplicate indices and merge overlapped ranges - // Note: addReadConflictRange in native transaction object does not merge overlapped ranges - state std::unordered_set mergedIds(conflictingKRIndices.begin(), - conflictingKRIndices.end()); - for (auto const& rCRIndex : mergedIds) { - const KeyRangeRef kr = req.transaction.read_conflict_ranges[rCRIndex]; - const KeyRange krWithPrefix = KeyRangeRef(kr.begin.withPrefix(conflictingKeysRange.begin), - kr.end.withPrefix(conflictingKeysRange.begin)); - trState->conflictingKeys->insert(krWithPrefix, conflictingKeysTrue); - } - } + trState->numErrors = 0; + ++trState->cx->transactionsCommitCompleted; + trState->cx->transactionCommittedMutations += req.transaction.mutations.size(); + trState->cx->transactionCommittedMutationBytes += req.transaction.mutations.expectedSize(); - if (debugID.present()) - TraceEvent(interval.end()).detail("Conflict", 1); + if (commitID.present()) { + g_traceBatch.addEvent("CommitDebug", + commitID.get().first(), + "NativeAPI.commit.After", + trState->spanContext.traceID, + trState->spanContext.spanID); + } - if (commitID.present()) - g_traceBatch.addEvent("CommitDebug", - commitID.get().first(), - "NativeAPI.commit.After", - trState->spanContext.traceID, - trState->spanContext.spanID); + double latency = now() - startTime; + trState->cx->commitLatencies.addSample(latency); + trState->cx->latencies.addSample(now() - trState->startTime); + if (trState->trLogInfo) { + trState->trLogInfo->addLog( + FdbClientLogEvents::EventCommit_V2(startTime, + trState->cx->clientLocality.dcId(), + latency, + req.transaction.mutations.size(), + req.transaction.mutations.expectedSize(), + ci.version, + req), + trState->spanContext); + } + if (trState->automaticIdempotency && alternativeChosen >= 0) { + // Automatic idempotency means we're responsible for best effort idempotency id clean up + proxiesUsed->getInterface(alternativeChosen) + .expireIdempotencyId.send( + ExpireIdempotencyIdRequest{ ci.version, uint8_t(ci.txnBatchId >> 8) }); + } + co_return; + } else { + // clear the RYW transaction which contains previous conflicting keys + trState->conflictingKeys.reset(); + if (ci.conflictingKRIndices.present()) { + trState->conflictingKeys = + std::make_shared>(conflictingKeysFalse, specialKeys.end); + Standalone> conflictingKRIndices = ci.conflictingKRIndices.get(); + // drop duplicate indices and merge overlapped ranges + // Note: addReadConflictRange in native transaction object does not merge overlapped ranges + std::unordered_set mergedIds(conflictingKRIndices.begin(), conflictingKRIndices.end()); + for (auto const& rCRIndex : mergedIds) { + const KeyRangeRef kr = req.transaction.read_conflict_ranges[rCRIndex]; + const KeyRange krWithPrefix = KeyRangeRef(kr.begin.withPrefix(conflictingKeysRange.begin), + kr.end.withPrefix(conflictingKeysRange.begin)); + trState->conflictingKeys->insert(krWithPrefix, conflictingKeysTrue); + } + } + + if (debugID.present()) + TraceEvent(interval.end()).detail("Conflict", 1); - throw not_committed(); + if (commitID.present()) { + g_traceBatch.addEvent("CommitDebug", + commitID.get().first(), + "NativeAPI.commit.After", + trState->spanContext.traceID, + trState->spanContext.spanID); } + + throw not_committed(); } + } else { + UNREACHABLE(); } } catch (Error& e) { - if (e.code() == error_code_request_maybe_delivered || e.code() == error_code_commit_unknown_result || - e.code() == error_code_never_reply) { - // We don't know if the commit happened, and it might even still be in flight. - - if (!trState->options.causalWriteRisky || req.idempotencyId.valid()) { - // Make sure it's not still in flight, either by ensuring the master we submitted to is dead, or the - // version we submitted with is dead, or by committing a conflicting transaction successfully - // if ( cx->getCommitProxies()->masterGeneration <= originalMasterGeneration ) - - // To ensure the original request is not in flight, we need a key range which intersects its read - // conflict ranges We pick a key range which also intersects its write conflict ranges, since that - // avoids potentially creating conflicts where there otherwise would be none We make the range as small - // as possible (a single key range) to minimize conflicts The intersection will never be empty, because - // if it were (since !causalWriteRisky) makeSelfConflicting would have been applied automatically to req - KeyRangeRef selfConflictingRange = - intersects(req.transaction.write_conflict_ranges, req.transaction.read_conflict_ranges).get(); - - CODE_PROBE(true, "Waiting for dummy transaction to report commit_unknown_result"); - - wait(commitDummyTransaction(trState, singleKeyRange(selfConflictingRange.begin))); - if (req.idempotencyId.valid()) { - Optional commitResult = wait(determineCommitStatus( - trState, - req.transaction.read_snapshot, - req.transaction.read_snapshot + CLIENT_KNOBS->MAX_WRITE_TRANSACTION_LIFE_VERSIONS, - req.idempotencyId)); - if (commitResult.present()) { - trState->committedVersion = commitResult.get().commitVersion; - Standalone ret = makeString(10); - placeVersionstamp( - mutateString(ret), commitResult.get().commitVersion, commitResult.get().batchIndex); - trState->versionstampPromise.send(ret); - CODE_PROBE(true, "AutomaticIdempotencyCommitted"); - return Void(); - } else { - CODE_PROBE(true, "AutomaticIdempotencyNotCommitted"); - throw transaction_too_old(); - } + err = e; + } + + if (err.code() == error_code_request_maybe_delivered || err.code() == error_code_commit_unknown_result || + err.code() == error_code_never_reply) { + // We don't know if the commit happened, and it might even still be in flight. + + if (!trState->options.causalWriteRisky || req.idempotencyId.valid()) { + // Make sure it's not still in flight, either by ensuring the master we submitted to is dead, or the + // version we submitted with is dead, or by committing a conflicting transaction successfully + // if ( cx->getCommitProxies()->masterGeneration <= originalMasterGeneration ) + + // To ensure the original request is not in flight, we need a key range which intersects its read + // conflict ranges We pick a key range which also intersects its write conflict ranges, since that + // avoids potentially creating conflicts where there otherwise would be none We make the range as + // small as possible (a single key range) to minimize conflicts The intersection will never be + // empty, because if it were (since !causalWriteRisky) makeSelfConflicting would have been applied + // automatically to req + KeyRangeRef selfConflictingRange = + intersects(req.transaction.write_conflict_ranges, req.transaction.read_conflict_ranges).get(); + + CODE_PROBE(true, "Waiting for dummy transaction to report commit_unknown_result"); + + co_await commitDummyTransaction(trState, singleKeyRange(selfConflictingRange.begin)); + if (req.idempotencyId.valid()) { + Optional commitResult = co_await determineCommitStatus( + trState, + req.transaction.read_snapshot, + req.transaction.read_snapshot + CLIENT_KNOBS->MAX_WRITE_TRANSACTION_LIFE_VERSIONS, + req.idempotencyId); + if (commitResult.present()) { + trState->committedVersion = commitResult.get().commitVersion; + Standalone ret = makeString(10); + placeVersionstamp( + mutateString(ret), commitResult.get().commitVersion, commitResult.get().batchIndex); + trState->versionstampPromise.send(ret); + CODE_PROBE(true, "AutomaticIdempotencyCommitted"); + co_return; + } else { + CODE_PROBE(true, "AutomaticIdempotencyNotCommitted"); + throw transaction_too_old(); } } + } - // The user needs to be informed that we aren't sure whether the commit happened. Standard retry loops - // retry it anyway (relying on transaction idempotence) but a client might do something else. - throw commit_unknown_result(); - } else { - if (e.code() != error_code_transaction_too_old && e.code() != error_code_not_committed && - e.code() != error_code_database_locked && e.code() != error_code_commit_proxy_memory_limit_exceeded && - e.code() != error_code_grv_proxy_memory_limit_exceeded && - e.code() != error_code_batch_transaction_throttled && e.code() != error_code_tag_throttled && - e.code() != error_code_process_behind && e.code() != error_code_future_version && - e.code() != error_code_transaction_throttled_hot_shard && - e.code() != error_code_transaction_rejected_range_locked) { - TraceEvent(SevError, "TryCommitError").error(e); - } - if (trState->trLogInfo) - trState->trLogInfo->addLog( - FdbClientLogEvents::EventCommitError( - startTime, trState->cx->clientLocality.dcId(), static_cast(e.code()), req), - trState->spanContext); - throw; + // The user needs to be informed that we aren't sure whether the commit happened. Standard retry loops + // retry it anyway (relying on transaction idempotence) but a client might do something else. + throw commit_unknown_result(); + } else { + if (err.code() != error_code_transaction_too_old && err.code() != error_code_not_committed && + err.code() != error_code_database_locked && err.code() != error_code_commit_proxy_memory_limit_exceeded && + err.code() != error_code_grv_proxy_memory_limit_exceeded && + err.code() != error_code_batch_transaction_throttled && err.code() != error_code_tag_throttled && + err.code() != error_code_process_behind && err.code() != error_code_future_version && + err.code() != error_code_transaction_throttled_hot_shard && + err.code() != error_code_transaction_rejected_range_locked) { + TraceEvent(SevError, "TryCommitError").error(err); + } + if (trState->trLogInfo) { + trState->trLogInfo->addLog( + FdbClientLogEvents::EventCommitError( + startTime, trState->cx->clientLocality.dcId(), static_cast(err.code()), req), + trState->spanContext); } + throw err; } } Future Transaction::commitMutations() { try { // if this is a read-only transaction return immediately - if (!tr.transaction.write_conflict_ranges.size() && !tr.transaction.mutations.size()) { + if (tr.transaction.write_conflict_ranges.empty() && tr.transaction.mutations.empty()) { trState->numErrors = 0; trState->committedVersion = invalidVersion; @@ -4732,10 +4829,11 @@ Future Transaction::commitMutations() { } bool isCheckingWrites = trState->options.checkWritesEnabled && deterministicRandom()->random01() < 0.01; - for (const auto& extraConflictRange : extraConflictRanges) + for (const auto& extraConflictRange : extraConflictRanges) { if (extraConflictRange.isReady() && extraConflictRange.get().first < extraConflictRange.get().second) tr.transaction.read_conflict_ranges.emplace_back( tr.arena, extraConflictRange.get().first, extraConflictRange.get().second); + } if (tr.idempotencyId.valid()) { // We need to be able confirm that this transaction is no longer in @@ -4762,11 +4860,12 @@ Future Transaction::commitMutations() { if (trState->options.debugDump) { UID u = nondeterministicRandom()->randomUniqueID(); TraceEvent("TransactionDump", u).log(); - for (auto i = tr.transaction.mutations.begin(); i != tr.transaction.mutations.end(); ++i) + for (auto i = tr.transaction.mutations.begin(); i != tr.transaction.mutations.end(); ++i) { TraceEvent("TransactionMutation", u) .detail("T", i->type) .detail("P1", i->param1) .detail("P2", i->param2); + } } if (trState->options.lockAware) { @@ -4805,9 +4904,9 @@ Future Transaction::commitMutations() { } } -ACTOR Future commitAndWatch(Transaction* self) { +Future commitAndWatch(Transaction* self) { try { - wait(self->commitMutations()); + co_await self->commitMutations(); self->getDatabase()->transactionTracingSample = (self->getCommittedVersion() % 60000000) < (60000000 * FLOW_KNOBS->TRACING_SAMPLE_RATE); @@ -4819,8 +4918,6 @@ ACTOR Future commitAndWatch(Transaction* self) { if (!self->apiVersionAtLeast(700)) { self->reset(); } - - return Void(); } catch (Error& e) { if (e.code() != error_code_actor_cancelled) { if (!self->watches.empty()) { @@ -4926,7 +5023,7 @@ void Transaction::setOption(FDBTransactionOptions::Option option, Optional 100 || value.get().size() == 0) { + if (value.get().size() > 100 || value.get().empty()) { throw invalid_option_value(); } @@ -5157,79 +5254,90 @@ void Transaction::setOption(FDBTransactionOptions::Option option, Optional getConsistentReadVersion(SpanContext parentSpan, - DatabaseContext* cx, - uint32_t transactionCount, - TransactionPriority priority, - uint32_t flags, - TransactionTagMap tags, - Optional debugID, - Optional maxGrvQueueDelayMS) { - state Span span("NAPI:getConsistentReadVersion"_loc, parentSpan); +Future getConsistentReadVersion(SpanContext parentSpan, + DatabaseContext* cx, + uint32_t transactionCount, + TransactionPriority priority, + uint32_t flags, + TransactionTagMap tagsInput, + Optional debugID, + Optional maxGrvQueueDelayMS) { + TransactionTagMap tags(std::move(tagsInput)); + Span span("NAPI:getConsistentReadVersion"_loc, parentSpan); + GetReadVersionRequest req; + Future onProxiesChanged; ++cx->transactionReadVersionBatches; - if (debugID.present()) + if (debugID.present()) { g_traceBatch.addEvent("TransactionDebug", debugID.get().first(), "NativeAPI.getConsistentReadVersion.Before", parentSpan.traceID, parentSpan.spanID); - loop { + } + while (true) { try { - state GetReadVersionRequest req(span.context, - transactionCount, - priority, - cx->ssVersionVectorCache.getMaxVersion(), - flags, - tags, - debugID, - maxGrvQueueDelayMS); - state Future onProxiesChanged = cx->onProxiesChanged(); - - choose { - when(wait(onProxiesChanged)) { - onProxiesChanged = cx->onProxiesChanged(); - } - when(GetReadVersionReply v = - wait(basicLoadBalance(cx->getGrvProxies(UseProvisionalProxies( - flags & GetReadVersionRequest::FLAG_USE_PROVISIONAL_PROXIES)), - &GrvProxyInterface::getConsistentReadVersion, - req, - cx->taskID))) { - if (tags.size() != 0) { - auto& priorityThrottledTags = cx->throttledTags[priority]; - for (auto& tag : tags) { - auto itr = v.tagThrottleInfo.find(tag.first); - if (itr == v.tagThrottleInfo.end()) { - CODE_PROBE(true, "Removing client throttle"); - priorityThrottledTags.erase(tag.first); - } else { - CODE_PROBE(true, "Setting client throttle"); - auto result = priorityThrottledTags.try_emplace(tag.first, itr->second); - if (!result.second) { - result.first->second.update(itr->second); - } + req = GetReadVersionRequest(span.context, + transactionCount, + priority, + cx->ssVersionVectorCache.getMaxVersion(), + flags, + tags, + debugID, + maxGrvQueueDelayMS); + onProxiesChanged = cx->onProxiesChanged(); + + Future reply = + onProxiesChanged.isReady() + ? Never() + : basicLoadBalance(cx->getGrvProxies(UseProvisionalProxies( + flags & GetReadVersionRequest::FLAG_USE_PROVISIONAL_PROXIES)), + &GrvProxyInterface::getConsistentReadVersion, + req, + cx->taskID); + auto res = co_await race(onProxiesChanged, std::move(reply)); + reply = Future(); + if (res.index() == 0) { + onProxiesChanged = cx->onProxiesChanged(); + } else if (res.index() == 1) { + GetReadVersionReply v = std::get<1>(std::move(res)); + + if (!tags.empty()) { + auto& priorityThrottledTags = cx->throttledTags[priority]; + for (auto& tag : tags) { + auto itr = v.tagThrottleInfo.find(tag.first); + if (itr == v.tagThrottleInfo.end()) { + CODE_PROBE(true, "Removing client throttle"); + priorityThrottledTags.erase(tag.first); + } else { + CODE_PROBE(true, "Setting client throttle"); + auto result = priorityThrottledTags.try_emplace(tag.first, itr->second); + if (!result.second) { + result.first->second.update(itr->second); } } } + } - if (debugID.present()) - g_traceBatch.addEvent("TransactionDebug", - debugID.get().first(), - "NativeAPI.getConsistentReadVersion.After", - parentSpan.traceID, - parentSpan.spanID); - ASSERT(v.version > 0); - cx->minAcceptableReadVersion = std::min(cx->minAcceptableReadVersion, v.version); - if (cx->versionVectorCacheActive(v.ssVersionVectorDelta)) { - if (cx->isCurrentGrvProxy(v.proxyId)) { - cx->ssVersionVectorCache.applyDelta(v.ssVersionVectorDelta); - } else { - continue; // stale GRV reply, retry - } + if (debugID.present()) { + g_traceBatch.addEvent("TransactionDebug", + debugID.get().first(), + "NativeAPI.getConsistentReadVersion.After", + parentSpan.traceID, + parentSpan.spanID); + } + ASSERT(v.version > 0); + cx->minAcceptableReadVersion = std::min(cx->minAcceptableReadVersion, v.version); + if (cx->versionVectorCacheActive(v.ssVersionVectorDelta)) { + if (cx->isCurrentGrvProxy(v.proxyId)) { + cx->ssVersionVectorCache.applyDelta(v.ssVersionVectorDelta); + } else { + continue; // stale GRV reply, retry } - return v; } + co_return v; + } else { + UNREACHABLE(); } } catch (Error& e) { if (e.code() != error_code_broken_promise && e.code() != error_code_batch_transaction_throttled && @@ -5241,39 +5349,43 @@ ACTOR Future getConsistentReadVersion(SpanContext parentSpa } } -ACTOR Future readVersionBatcher(DatabaseContext* cx, - FutureStream versionStream, - TransactionPriority priority, - uint32_t flags, - Optional maxGrvQueueDelayMS) { - state std::vector> requests; - state PromiseStream> addActor; - state Future collection = actorCollection(addActor.getFuture()); - state Future timeout; - state Optional debugID; - state bool send_batch; - state Reference batchSizeDist = Histogram::getHistogram( +Future readVersionBatcher(DatabaseContext* cx, + FutureStream versionStreamInput, + TransactionPriority priority, + uint32_t flags, + Optional maxGrvQueueDelayMS) { + FutureStream versionStream(std::move(versionStreamInput)); + std::vector> requests; + PromiseStream> addActor; + Future collection = actorCollection(addActor.getFuture()); + Future timeout; + Optional debugID; + Reference batchSizeDist = Histogram::getHistogram( "GrvBatcher"_sr, "ClientGrvBatchSize"_sr, Histogram::Unit::countLinear, 0, CLIENT_KNOBS->MAX_BATCH_SIZE * 2); - state Reference batchIntervalDist = - Histogram::getHistogram("GrvBatcher"_sr, - "ClientGrvBatchInterval"_sr, - Histogram::Unit::milliseconds, - 0, - CLIENT_KNOBS->GRV_BATCH_TIMEOUT * 1000000 * 2); - state Reference grvReplyLatencyDist = + Reference batchIntervalDist = Histogram::getHistogram("GrvBatcher"_sr, + "ClientGrvBatchInterval"_sr, + Histogram::Unit::milliseconds, + 0, + CLIENT_KNOBS->GRV_BATCH_TIMEOUT * 1000000 * 2); + Reference grvReplyLatencyDist = Histogram::getHistogram("GrvBatcher"_sr, "ClientGrvReplyLatency"_sr, Histogram::Unit::milliseconds); - state double lastRequestTime = now(); + double lastRequestTime = now(); - state TransactionTagMap tags; + TransactionTagMap tags; // dynamic batching - state PromiseStream replyTimes; - state double batchTime = 0; - state Span span("NAPI:readVersionBatcher"_loc); - loop { - send_batch = false; - choose { - when(DatabaseContext::VersionRequest req = waitNext(versionStream)) { + PromiseStream replyTimes; + double batchTime = 0; + Span span("NAPI:readVersionBatcher"_loc); + while (true) { + bool send_batch = false; + { + Future batchTimeout = timeout.isValid() ? timeout : Never(); + auto replyLatency = replyTimes.getFuture(); + auto res = co_await race(versionStream, std::move(batchTimeout), std::move(replyLatency), collection); + if (res.index() == 0) { + DatabaseContext::VersionRequest req = std::get<0>(std::move(res)); + if (req.debugID.present()) { if (!debugID.present()) { debugID = nondeterministicRandom()->randomUniqueID(); @@ -5296,18 +5408,17 @@ ACTOR Future readVersionBatcher(DatabaseContext* cx, } else if (!timeout.isValid()) { timeout = delay(batchTime, TaskPriority::GetConsistentReadVersion); } - } - when(wait(timeout.isValid() ? timeout : Never())) { + } else if (res.index() == 1) { send_batch = true; ++cx->transactionGrvTimedOutBatches; - } - // dynamic batching monitors reply latencies - when(double reply_latency = waitNext(replyTimes.getFuture())) { + } else if (res.index() == 2) { + // Dynamic batching monitors reply latencies. + double reply_latency = std::get<2>(std::move(res)); + double target_latency = reply_latency * 0.5; batchTime = std::min(0.1 * target_latency + 0.9 * batchTime, CLIENT_KNOBS->GRV_BATCH_TIMEOUT); grvReplyLatencyDist->sampleSeconds(reply_latency); - } - when(wait(collection)) {} // for errors + } // collection is included for errors } if (send_batch) { int count = requests.size(); @@ -5325,7 +5436,7 @@ ACTOR Future readVersionBatcher(DatabaseContext* cx, Future batch = incrementalBroadcastWithError( getConsistentReadVersion( - span.context, cx, count, priority, flags, std::move(tags), std::move(debugID), maxGrvQueueDelayMS), + span.context, cx, count, priority, flags, std::move(tags), debugID, maxGrvQueueDelayMS), std::move(requests), CLIENT_KNOBS->BROADCAST_BATCH_SIZE); @@ -5339,13 +5450,16 @@ ACTOR Future readVersionBatcher(DatabaseContext* cx, } } -ACTOR Future extractReadVersion(Reference trState, - Location location, - SpanContext spanContext, - Future f, - Promise> metadataVersion) { - state Span span(spanContext, location, trState->spanContext); - GetReadVersionReply rep = wait(f); +Future extractReadVersion(Reference trStateInput, + Location location, + SpanContext spanContext, + Future fInput, + Promise> metadataVersionInput) { + Reference trState(std::move(trStateInput)); + Future f(std::move(fInput)); + Promise> metadataVersion(std::move(metadataVersionInput)); + Span span(spanContext, location, trState->spanContext); + GetReadVersionReply rep = co_await f; if (CLIENT_BUGGIFY) { throw grv_proxy_memory_limit_exceeded(); } @@ -5360,13 +5474,14 @@ ACTOR Future extractReadVersion(Reference trState, trState->cx->lastRkDefaultThrottleTime = replyTime; } trState->cx->GRVLatencies.addSample(latency); - if (trState->trLogInfo) + if (trState->trLogInfo) { trState->trLogInfo->addLog(FdbClientLogEvents::EventGetVersion_V3(trState->startTime, trState->cx->clientLocality.dcId(), latency, trState->options.priority, rep.version), trState->spanContext); + } if (rep.locked && !trState->options.lockAware) throw database_locked(); @@ -5423,7 +5538,26 @@ ACTOR Future extractReadVersion(Reference trState, trState->cx->ssVersionVectorCache.clear(); } } - return rep.version; + co_return rep.version; +} + +TEST_CASE("/fdbclient/NativeAPI/extractReadVersion/releasesStateAndMetadataOnError") { + auto trState = makeReference(TaskPriority::DefaultEndpoint, SpanContext()); + Promise reply; + Promise> metadataVersion; + auto metadata = metadataVersion.getFuture(); + trState->readVersionFuture = extractReadVersion( + trState, "NativeAPI:readVersionLifetimeTest"_loc, SpanContext(), reply.getFuture(), std::move(metadataVersion)); + ASSERT_EQ(trState->debugGetReferenceCount(), 2); + + reply.sendError(transaction_too_old()); + + ASSERT(trState->readVersionFuture.isError()); + ASSERT_EQ(trState->readVersionFuture.getError().code(), error_code_transaction_too_old); + ASSERT(metadata.isError()); + ASSERT_EQ(metadata.getError().code(), error_code_broken_promise); + ASSERT_EQ(trState->debugGetReferenceCount(), 1); + return Void(); } bool rkThrottlingCooledDown(DatabaseContext* cx, TransactionPriority priority) { @@ -5443,18 +5577,18 @@ bool rkThrottlingCooledDown(DatabaseContext* cx, TransactionPriority priority) { return false; } -ACTOR static Future backgroundGrvUpdater(DatabaseContext* cx) { - state Transaction tr; - state double grvDelay = 0.001; - state Backoff backoff; +static Future backgroundGrvUpdater(DatabaseContext* cx) { + Transaction tr; + double grvDelay = 0.001; + Backoff backoff; try { - loop { + while (true) { if (CLIENT_KNOBS->FORCE_GRV_CACHE_OFF) - return Void(); - wait(refreshTransaction(cx, &tr)); - state double curTime = now(); - state double lastTime = cx->getLastGrvTime(); - state double lastProxyTime = cx->lastProxyRequestTime; + co_return; + co_await refreshTransaction(cx, &tr); + double curTime = now(); + double lastTime = cx->getLastGrvTime(); + double lastProxyTime = cx->lastProxyRequestTime; TraceEvent(SevDebug, "BackgroundGrvUpdaterBefore") .detail("CurTime", curTime) .detail("LastTime", lastTime) @@ -5465,9 +5599,10 @@ ACTOR static Future backgroundGrvUpdater(DatabaseContext* cx) { .detail("Bound", CLIENT_KNOBS->MAX_VERSION_CACHE_LAG - grvDelay); if (curTime - lastTime >= (CLIENT_KNOBS->MAX_VERSION_CACHE_LAG - grvDelay) || curTime - lastProxyTime > CLIENT_KNOBS->MAX_PROXY_CONTACT_LAG) { + Error err; try { tr.setOption(FDBTransactionOptions::SKIP_GRV_CACHE); - wait(success(tr.getReadVersion())); + co_await tr.getReadVersion(); cx->lastProxyRequestTime = curTime; grvDelay = (grvDelay + (now() - curTime)) / 2.0; TraceEvent(SevDebug, "BackgroundGrvUpdaterSuccess") @@ -5475,16 +5610,19 @@ ACTOR static Future backgroundGrvUpdater(DatabaseContext* cx) { .detail("CachedReadVersion", cx->getCachedReadVersion()) .detail("CachedTime", cx->getLastGrvTime()); backoff = Backoff(); + continue; } catch (Error& e) { - TraceEvent(SevInfo, "BackgroundGrvUpdaterTxnError").errorUnsuppressed(e); - wait(tr.onError(e)); - wait(backoff.onError()); + err = e; } + + TraceEvent(SevInfo, "BackgroundGrvUpdaterTxnError").errorUnsuppressed(err); + co_await tr.onError(err); + co_await backoff.onError(); } else { - wait( - delay(std::max(0.001, - std::min(CLIENT_KNOBS->MAX_PROXY_CONTACT_LAG - (curTime - lastProxyTime), - (CLIENT_KNOBS->MAX_VERSION_CACHE_LAG - grvDelay) - (curTime - lastTime))))); + co_await delay( + std::max(0.001, + std::min(CLIENT_KNOBS->MAX_PROXY_CONTACT_LAG - (curTime - lastProxyTime), + (CLIENT_KNOBS->MAX_VERSION_CACHE_LAG - grvDelay) - (curTime - lastTime)))); } } } catch (Error& e) { @@ -5645,20 +5783,21 @@ Future> getCoordinatorProtocolFromConnectPacket(Networ // Returns the protocol version reported by the given coordinator // If an expected version is given, the future won't return until the protocol version is different than expected -ACTOR Future getClusterProtocolImpl( - Reference> const> coordinator, +Future getClusterProtocolImpl( + Reference> const> coordinatorInput, Optional expectedVersion) { - state bool needToConnect = true; - state Future protocolVersion = Never(); + Reference> const> coordinator(std::move(coordinatorInput)); + bool needToConnect = true; + Future protocolVersion = Never(); - loop { + while (true) { if (!coordinator->get().present()) { - wait(coordinator->onChange()); + co_await coordinator->onChange(); } else { - state NetworkAddress coordinatorAddress; + NetworkAddress coordinatorAddress; if (coordinator->get().get().hostname.present()) { - state Hostname h = coordinator->get().get().hostname.get(); - wait(store(coordinatorAddress, h.resolveWithRetry())); + Hostname h = coordinator->get().get().hostname.get(); + coordinatorAddress = co_await h.resolveWithRetry(); } else { coordinatorAddress = coordinator->get().get().getLeader.getEndpoint().getPrimaryAddress(); } @@ -5669,29 +5808,35 @@ ACTOR Future getClusterProtocolImpl( protocolVersion = getCoordinatorProtocol(coordinatorAddress); needToConnect = false; } - choose { - when(wait(coordinator->onChange())) { - needToConnect = true; - } - when(ProtocolVersion pv = wait(protocolVersion)) { - if (!expectedVersion.present() || expectedVersion.get() != pv) { - return pv; - } + auto coordinatorChanged = coordinator->onChange(); + // Older coordinators report their protocol version only in the connect packet. + Future> connectPacketVersion = + coordinatorChanged.isReady() || protocolVersion.isReady() + ? Never() + : getCoordinatorProtocolFromConnectPacket(coordinatorAddress, expectedVersion); + auto res = co_await race(std::move(coordinatorChanged), protocolVersion, std::move(connectPacketVersion)); + connectPacketVersion = Future>(); + if (res.index() == 0) { + needToConnect = true; + } else if (res.index() == 1) { + ProtocolVersion pv = std::get<1>(res); - protocolVersion = Never(); + if (!expectedVersion.present() || expectedVersion.get() != pv) { + co_return pv; } - // Older versions of FDB don't have an endpoint to return the protocol version, so we get this info from - // the connect packet - when(Optional pv = - wait(getCoordinatorProtocolFromConnectPacket(coordinatorAddress, expectedVersion))) { - if (pv.present()) { - return pv.get(); - } else { - needToConnect = true; - } + protocolVersion = Never(); + } else if (res.index() == 2) { + Optional pv = std::get<2>(res); + + if (pv.present()) { + co_return pv.get(); + } else { + needToConnect = true; } + } else { + UNREACHABLE(); } } } @@ -6197,30 +6342,34 @@ class RangeSplitPointsBuilder { } }; -ACTOR Future>> getRangeSplitPoints(Reference trState, - KeyRange keys, - int64_t chunkSize, - int limit) { - state Span span("NAPI:GetRangeSplitPoints"_loc, trState->spanContext); - state Key beginKey = keys.begin; - state RangeSplitPointsBuilder results(keys.begin, limit); +Future>> getRangeSplitPoints(Reference trStateInput, + KeyRange keysInput, + int64_t chunkSize, + int limit) { + Reference trState(std::move(trStateInput)); + KeyRange keys(std::move(keysInput)); + Span span("NAPI:GetRangeSplitPoints"_loc, trState->spanContext); + Key beginKey = keys.begin; + RangeSplitPointsBuilder results(keys.begin, limit); if (limit == 0) { - return results.finish(keys.end); + co_return results.finish(keys.end); } - loop { - state std::vector locations = wait(getKeyRangeLocations( + // Keep location interfaces and pending requests alive across retries. + std::vector locations; + std::vector> fReplies; + while (true) { + locations = co_await getKeyRangeLocations( trState, KeyRangeRef(beginKey, keys.end), getRangeSplitPointsLocationLimit( results.getRemaining(), CLIENT_KNOBS->TOO_MANY, CLIENT_KNOBS->STORAGE_METRICS_SHARD_LIMIT), Reverse::False, - &StorageServerInterface::getRangeSplitPoints)); + &StorageServerInterface::getRangeSplitPoints); try { - state int nLocs = locations.size(); + int nLocs = locations.size(); if (limit >= 0) { - state int i = 0; - for (; i < nLocs; i++) { + for (int i = 0; i < nLocs; i++) { if (i > 0 || beginKey != keys.begin) { results.appendShardBoundary(locations[i].range.begin); } @@ -6230,14 +6379,14 @@ ACTOR Future>> getRangeSplitPoints(Referencelocations(), - &StorageServerInterface::getRangeSplitPoints, - req, - TaskPriority::DataDistribution)); + SplitRangeReply reply = co_await loadBalance(locations[i].locations->locations(), + &StorageServerInterface::getRangeSplitPoints, + req, + TaskPriority::DataDistribution); results.appendSplitPoints(reply.splitPoints); } } else { - state std::vector> fReplies(nLocs); + fReplies = std::vector>(nLocs); for (int i = 0; i < nLocs; i++) { KeyRef partBegin = (i == 0) ? beginKey : locations[i].range.begin; KeyRef partEnd = std::min(keys.end, locations[i].range.end); @@ -6247,7 +6396,7 @@ ACTOR Future>> getRangeSplitPoints(Reference 0 || beginKey != keys.begin) { results.appendShardBoundary(locations[i].range.begin); @@ -6256,20 +6405,21 @@ ACTOR Future>> getRangeSplitPoints(Referencecx->invalidateCache(keys); beginKey = keys.begin; results = RangeSplitPointsBuilder(keys.begin, limit); - wait(delay(CLIENT_KNOBS->WRONG_SHARD_SERVER_DELAY, TaskPriority::DataDistribution)); } else { TraceEvent(SevError, "GetRangeSplitPoints").error(e); throw; } } + co_await delay(CLIENT_KNOBS->WRONG_SHARD_SERVER_DELAY, TaskPriority::DataDistribution); } } @@ -6410,55 +6560,57 @@ Future>> readStorageWiggleValues( } } -ACTOR Future splitStorageMetricsStream(PromiseStream resultStream, - Database cx, - KeyRange keys, - StorageMetrics limit, - StorageMetrics estimated, - Optional minSplitBytes) { - state Span span("NAPI:SplitStorageMetricsStream"_loc); - state Key beginKey = keys.begin; - state Key globalLastKey = beginKey; +Future splitStorageMetricsStream(PromiseStream resultStreamInput, + Database cxInput, + KeyRange keysInput, + StorageMetrics limit, + StorageMetrics estimated, + Optional minSplitBytes) { + PromiseStream resultStream(std::move(resultStreamInput)); + Database cx(std::move(cxInput)); + KeyRange keys(std::move(keysInput)); + Span span("NAPI:SplitStorageMetricsStream"_loc); + Key beginKey = keys.begin; + Key globalLastKey = beginKey; resultStream.send(beginKey); // track used across loops - state StorageMetrics globalUsed; - loop { - state std::vector locations = - wait(getKeyRangeLocations(cx, - KeyRangeRef(beginKey, keys.end), - CLIENT_KNOBS->STORAGE_METRICS_SHARD_LIMIT, - Reverse::False, - &StorageServerInterface::splitMetrics, - span.context, - Optional(), - UseProvisionalProxies::False, - latestVersion)); + StorageMetrics globalUsed; + std::vector locations; + while (true) { + locations = co_await getKeyRangeLocations(cx, + KeyRangeRef(beginKey, keys.end), + CLIENT_KNOBS->STORAGE_METRICS_SHARD_LIMIT, + Reverse::False, + &StorageServerInterface::splitMetrics, + span.context, + Optional(), + UseProvisionalProxies::False, + latestVersion); try { //TraceEvent("SplitStorageMetrics").detail("Locations", locations.size()); - state StorageMetrics localUsed = globalUsed; - state Key localLastKey = globalLastKey; - state Standalone> results; - state int i = 0; - for (; i < locations.size(); i++) { + StorageMetrics localUsed = globalUsed; + Key localLastKey = globalLastKey; + Standalone> results; + for (int i = 0; i < locations.size(); i++) { SplitMetricsRequest req(locations[i].range, limit, localUsed, estimated, i == locations.size() - 1 && keys.end <= locations.back().range.end, minSplitBytes); - SplitMetricsReply res = wait(loadBalance(locations[i].locations->locations(), - &StorageServerInterface::splitMetrics, - req, - TaskPriority::DataDistribution)); - if (res.splits.size() && + SplitMetricsReply res = co_await loadBalance(locations[i].locations->locations(), + &StorageServerInterface::splitMetrics, + req, + TaskPriority::DataDistribution); + if (!res.splits.empty() && res.splits[0] <= localLastKey) { // split points are out of order, possibly because // of moving data, throw error to retry ASSERT_WE_THINK(false); // FIXME: This seems impossible and doesn't seem to be covered by testing throw all_alternatives_failed(); } - if (res.splits.size()) { + if (!res.splits.empty()) { results.append(results.arena(), res.splits.begin(), res.splits.size()); results.arena().dependsOn(res.splits.arena()); localLastKey = res.splits.back(); @@ -6490,6 +6642,7 @@ ACTOR Future splitStorageMetricsStream(PromiseStream resultStream, } else { beginKey = locations.back().range.end; } + continue; } catch (Error& e) { if (e.code() == error_code_operation_cancelled) { throw e; @@ -6500,10 +6653,9 @@ ACTOR Future splitStorageMetricsStream(PromiseStream resultStream, throw; } cx->invalidateCache(keys); - wait(delay(CLIENT_KNOBS->WRONG_SHARD_SERVER_DELAY, TaskPriority::DataDistribution)); } + co_await delay(CLIENT_KNOBS->WRONG_SHARD_SERVER_DELAY, TaskPriority::DataDistribution); } - return Void(); } Future DatabaseContext::splitStorageMetricsStream(const PromiseStream& resultStream, @@ -6535,21 +6687,21 @@ static Future>>> splitStorageMetricsWithLo &StorageServerInterface::splitMetrics, req, TaskPriority::DataDistribution); - if (res.splits.size() && + if (!res.splits.empty() && res.splits[0] <= results.back()) { // split points are out of order, possibly // because of moving data, throw error to retry ASSERT_WE_THINK(false); // FIXME: This seems impossible and doesn't seem to be covered by testing throw all_alternatives_failed(); } - if (res.splits.size()) { + if (!res.splits.empty()) { results.append(results.arena(), res.splits.begin(), res.splits.size()); results.arena().dependsOn(res.splits.arena()); } used = res.used; - if (res.more && res.splits.size()) { + if (res.more && !res.splits.empty()) { // Next request will return split points after this one beginKey = KeyRef(beginKey.arena(), res.splits.back()); } else { @@ -6585,41 +6737,43 @@ Future>>> splitStorageMetricsWithLocations return splitStorageMetricsWithLocationsImpl(locations, keys, limit, estimated, minSplitBytes); } -ACTOR Future>> splitStorageMetrics(Database cx, - KeyRange keys, - StorageMetrics limit, - StorageMetrics estimated, - Optional minSplitBytes) { - state Span span("NAPI:SplitStorageMetrics"_loc); - loop { - state std::vector locations = - wait(getKeyRangeLocations(cx, - keys, - CLIENT_KNOBS->STORAGE_METRICS_SHARD_LIMIT, - Reverse::False, - &StorageServerInterface::splitMetrics, - span.context, - Optional(), - UseProvisionalProxies::False, - latestVersion)); +Future>> splitStorageMetrics(Database cxInput, + KeyRange keysInput, + StorageMetrics limit, + StorageMetrics estimated, + Optional minSplitBytes) { + Database cx(std::move(cxInput)); + KeyRange keys(std::move(keysInput)); + Span span("NAPI:SplitStorageMetrics"_loc); + std::vector locations; + while (true) { + locations = co_await getKeyRangeLocations(cx, + keys, + CLIENT_KNOBS->STORAGE_METRICS_SHARD_LIMIT, + Reverse::False, + &StorageServerInterface::splitMetrics, + span.context, + Optional(), + UseProvisionalProxies::False, + latestVersion); // SOMEDAY: Right now, if there are too many shards we delay and check again later. There may be a better // solution to this. if (locations.size() == CLIENT_KNOBS->STORAGE_METRICS_SHARD_LIMIT) { - wait(delay(CLIENT_KNOBS->STORAGE_METRICS_TOO_MANY_SHARDS_DELAY, TaskPriority::DataDistribution)); + co_await delay(CLIENT_KNOBS->STORAGE_METRICS_TOO_MANY_SHARDS_DELAY, TaskPriority::DataDistribution); cx->invalidateCache(keys); continue; } Optional>> results = - wait(splitStorageMetricsWithLocations(locations, keys, limit, estimated, minSplitBytes)); + co_await splitStorageMetricsWithLocations(locations, keys, limit, estimated, minSplitBytes); if (results.present()) { - return results.get(); + co_return results.get(); } cx->invalidateCache(keys); - wait(delay(CLIENT_KNOBS->WRONG_SHARD_SERVER_DELAY, TaskPriority::DataDistribution)); + co_await delay(CLIENT_KNOBS->WRONG_SHARD_SERVER_DELAY, TaskPriority::DataDistribution); } } @@ -6687,22 +6841,24 @@ Future snapCreate(Database cx, Standalone snapCmd, UID snapUID) } } -ACTOR template -static Future createCheckpointImpl(T tr, - std::vector ranges, +template +static Future createCheckpointImpl(T trInput, + std::vector rangesInput, CheckpointFormat format, Optional actionId) { + T tr(std::move(trInput)); + std::vector ranges(std::move(rangesInput)); ASSERT(!ranges.empty()); ASSERT(actionId.present()); TraceEvent(SevDebug, "CreateCheckpointTransactionBegin").detail("Ranges", describe(ranges)); - state RangeResult UIDtoTagMap = wait(tr->getRange(serverTagKeys, CLIENT_KNOBS->TOO_MANY)); + RangeResult UIDtoTagMap = co_await tr->getRange(serverTagKeys, CLIENT_KNOBS->TOO_MANY); ASSERT(!UIDtoTagMap.more && UIDtoTagMap.size() < CLIENT_KNOBS->TOO_MANY); - state std::unordered_map> rangeMap; - state std::unordered_map> srcMap; + std::unordered_map> rangeMap; + std::unordered_map> srcMap; for (const auto& range : ranges) { - RangeResult keyServers = wait(krmGetRanges(tr, keyServersPrefix, range)); + RangeResult keyServers = co_await krmGetRanges(tr, keyServersPrefix, range); ASSERT(!keyServers.more); for (int i = 0; i < keyServers.size() - 1; ++i) { const KeyRangeRef currentRange(keyServers[i].key, keyServers[i + 1].key); @@ -6732,8 +6888,6 @@ static Future createCheckpointImpl(T tr, } else { throw not_implemented(); } - - return Void(); } Future createCheckpoint(Reference tr, @@ -6806,19 +6960,20 @@ static Future getCheckpointMetaDataInternal(KeyRange range, throw error.get(); } -ACTOR static Future>> getCheckpointMetaDataForRange( - Database cx, - KeyRange range, +static Future>> getCheckpointMetaDataForRange( + Database cxInput, + KeyRange rangeInput, Version version, CheckpointFormat format, Optional actionId, double timeout) { - state Span span("NAPI:GetCheckpointMetaDataForRange"_loc); - state int index = 0; - state std::vector> futures; - state std::vector locations; + Database cx(std::move(cxInput)); + KeyRange range(std::move(rangeInput)); + Span span("NAPI:GetCheckpointMetaDataForRange"_loc); + std::vector> futures; + std::vector locations; - loop { + while (true) { locations.clear(); TraceEvent(SevDebug, "GetCheckpointMetaDataForRangeBegin") .detail("Range", range.toString()) @@ -6827,18 +6982,17 @@ ACTOR static Future>> getChe futures.clear(); try { - wait(store(locations, - getKeyRangeLocations(cx, - range, - CLIENT_KNOBS->TOO_MANY, - Reverse::False, - &StorageServerInterface::checkpoint, - span.context, - Optional(), - UseProvisionalProxies::False, - latestVersion))); - - for (index = 0; index < locations.size(); ++index) { + locations = co_await getKeyRangeLocations(cx, + range, + CLIENT_KNOBS->TOO_MANY, + Reverse::False, + &StorageServerInterface::checkpoint, + span.context, + Optional(), + UseProvisionalProxies::False, + latestVersion); + + for (int index = 0; index < locations.size(); ++index) { futures.push_back(getCheckpointMetaDataInternal( locations[index].range, version, format, actionId, locations[index].locations, timeout)); TraceEvent(SevDebug, "GetCheckpointShardBegin") @@ -6847,37 +7001,40 @@ ACTOR static Future>> getChe .detail("StorageServers", locations[index].locations->description()); } - choose { - when(wait(cx->connectionFileChanged())) { - cx->invalidateCache(range); - } - when(wait(waitForAll(futures))) { - break; - } - when(wait(delay(timeout))) { - TraceEvent(SevWarn, "GetCheckpointTimeout").detail("Range", range).detail("Version", version); - } + auto connectionChanged = cx->connectionFileChanged(); + Future allReplies = connectionChanged.isReady() ? Never() : waitForAll(futures); + Future timedOut = connectionChanged.isReady() || allReplies.isReady() ? Never() : delay(timeout); + auto res = co_await race(std::move(connectionChanged), std::move(allReplies), std::move(timedOut)); + connectionChanged = Future(); + allReplies = Future(); + timedOut = Future(); + if (res.index() == 0) { + cx->invalidateCache(range); + } else if (res.index() == 1) { + break; + } else { + TraceEvent(SevWarn, "GetCheckpointTimeout").detail("Range", range).detail("Version", version); } + continue; } catch (Error& e) { TraceEvent(SevWarn, "GetCheckpointError").errorUnsuppressed(e).detail("Range", range); - if (e.code() == error_code_wrong_shard_server || e.code() == error_code_all_alternatives_failed || - e.code() == error_code_connection_failed || e.code() == error_code_broken_promise) { - cx->invalidateCache(range); - wait(delay(CLIENT_KNOBS->WRONG_SHARD_SERVER_DELAY)); - } else { + if (e.code() != error_code_wrong_shard_server && e.code() != error_code_all_alternatives_failed && + e.code() != error_code_connection_failed && e.code() != error_code_broken_promise) { throw; } + cx->invalidateCache(range); } + co_await delay(CLIENT_KNOBS->WRONG_SHARD_SERVER_DELAY); } std::vector> res; - for (index = 0; index < futures.size(); ++index) { + for (int index = 0; index < futures.size(); ++index) { TraceEvent(SevDebug, "GetCheckpointShardEnd") .detail("Range", locations[index].range) .detail("Checkpoint", futures[index].get().toString()); res.emplace_back(locations[index].range, futures[index].get()); } - return res; + co_return res; } Future>> getCheckpointMetaData(Database cx, @@ -6887,6 +7044,7 @@ Future>> getCheckpointMetaDa Optional actionId, double timeout) { std::vector>>> futures; + futures.reserve(ranges.size()); // TODO(heliu): Avoid send requests to the same shard. for (const auto& range : ranges) { @@ -6976,20 +7134,22 @@ Future checkSafeExclusions(Database cx, std::vector excl } // returns true if we can connect to the given worker interface -ACTOR Future verifyInterfaceActor(Reference connectLock, ClientWorkerInterface workerInterf) { - wait(connectLock->take()); - state FlowLock::Releaser releaser(*connectLock); - state ClientLeaderRegInterface leaderInterf(workerInterf.address()); - choose { - when(Optional rep = - wait(brokenPromiseToNever(leaderInterf.getLeader.getReply(GetLeaderRequest())))) { - return true; - } - when(wait(delay(CLIENT_KNOBS->CLI_CONNECT_TIMEOUT))) { - // NOTE : change timeout time here if necessary - return false; - } - } +static Future verifyInterfaceImpl(Reference connectLockInput, ClientWorkerInterface workerInterfInput) { + Reference connectLock(std::move(connectLockInput)); + ClientWorkerInterface workerInterf(std::move(workerInterfInput)); + co_await connectLock->take(); + FlowLock::Releaser releaser(*connectLock); + ClientLeaderRegInterface leaderInterf(workerInterf.address()); + auto leaderReply = brokenPromiseToNever(leaderInterf.getLeader.getReply(GetLeaderRequest())); + Future timedOut = leaderReply.isReady() ? Never() : delay(CLIENT_KNOBS->CLI_CONNECT_TIMEOUT); + auto res = co_await race(std::move(leaderReply), std::move(timedOut)); + leaderReply = Future>(); + timedOut = Future(); + co_return res.index() == 0; +} + +Future verifyInterfaceActor(Reference const& connectLock, ClientWorkerInterface const& workerInterf) { + return verifyInterfaceImpl(connectLock, workerInterf); } static Future rebootWorkerActor(DatabaseContext* cx, ValueRef addr, bool check, int duration) { @@ -7023,7 +7183,7 @@ static Future rebootWorkerActor(DatabaseContext* cx, ValueRef addr, boo std::vector> verifyInterfs; for (const auto& requestedAddress : addressesVec) { // step 1: check that the requested address is in the worker list provided by CC - if (!workerInterfaces.count(Key(requestedAddress))) + if (!workerInterfaces.contains(Key(requestedAddress))) co_return 0; // step 2: try to establish connections to the requested worker verifyInterfs.push_back(verifyInterfaceActor(connectLock, workerInterfaces[Key(requestedAddress)])); diff --git a/fdbclient/NativeCdc.cpp b/fdbclient/NativeCdc.cpp index 6cb2f858b96..b0ce1f7282c 100644 --- a/fdbclient/NativeCdc.cpp +++ b/fdbclient/NativeCdc.cpp @@ -68,6 +68,8 @@ class NativeCdcIdentifierAllocator { ++tagStreamCounts[tag.id]; } + bool hasStreams(Tag tag) const { return tagStreamCounts.contains(tag.id); } + std::pair allocate(int tagCount) const { if (sawStream && maxStreamId == std::numeric_limits::max()) { throw operation_failed(); @@ -120,8 +122,29 @@ Future getNativeCdcCurrentTag(Transaction* tr, CDCStreamId streamId) { co_return decodeCDCTagHistoryKey(history.front().key).tag; } -// TODO: Persist current per-tag ownership so registration does not reconstruct it by scanning all active streams. Future> getNativeCdcProxyAssignmentForTag(Transaction* tr, Tag targetTag) { + const Key ownerKey = cdcTagOwnerKeyFor(targetTag); + Optional indexedStream = co_await tr->get(ownerKey); + if (indexedStream.present()) { + const CDCStreamId streamId = decodeCDCTagOwnerValue(indexedStream.get()); + Future> activeStream = tr->get(cdcStreamKeyFor(streamId)); + RangeResult history = co_await tr->getRange(cdcTagHistoryRangeFor(streamId), 1, Snapshot::False, Reverse::True); + // Keep the await separate so GCC 13 does not evaluate history.front() before the short-circuit guards. + const Optional activeStreamValue = co_await activeStream; + // The index is derived: removal or retagging can invalidate its representative, and the per-stream + // assignment remains authoritative across proxy replacement, including by older metadata writers. + if (activeStreamValue.present() && !history.empty() && + decodeCDCTagHistoryKey(history.front().key).tag == targetTag) { + Optional proxyId = co_await getNativeCdcProxyAssignment(tr, streamId); + if (proxyId.present()) { + CODE_PROBE(true, "Native CDC resolves a shared tag owner from its persisted index"); + co_return proxyId; + } + } + CODE_PROBE(true, "Native CDC rebuilds a stale tag owner index"); + tr->clear(ownerKey); + } + std::set activeStreamIds; Key begin = cdcStreamKeys.begin; while (begin < cdcStreamKeys.end) { @@ -156,6 +179,8 @@ Future> getNativeCdcProxyAssignmentForTag(Transaction* tr, Tag tar if (tag == targetTag) { Optional proxyId = co_await getNativeCdcProxyAssignment(tr, streamId); if (proxyId.present()) { + tr->set(ownerKey, cdcTagOwnerValue(streamId)); + CODE_PROBE(true, "Native CDC reconstructs a missing tag owner index from active streams"); co_return proxyId; } } @@ -388,6 +413,9 @@ Future registerNativeCdcStream(Database cx, Key name, KeyRange keys probe::decoration::rare); const UID selectedProxy = sharedTagProxy.present() ? sharedTagProxy.get() : proxyId; tr.set(cdcProxyKeyFor(streamId, selectedProxy), Value()); + if (!sharedTagProxy.present()) { + tr.set(cdcTagOwnerKeyFor(tag), cdcTagOwnerValue(streamId)); + } signalNativeCdcProxyAssignmentChange(&tr); co_await tr.commit(); } @@ -413,9 +441,15 @@ Future registerNativeCdcStream(Database cx, Key name, KeyRange keys tr.set(cdcTagHistoryKeyFor(streamId, registrationVersion, tag), Value()); tr.atomicOp( cdcMinVersionKeyFor(streamId), cdcVersionstampedMinVersionValue(), MutationRef::SetVersionstampedValue); - Optional sharedTagProxy = co_await getNativeCdcProxyAssignmentForTag(&tr, tag); + Optional sharedTagProxy; + if (allocator.hasStreams(tag)) { + sharedTagProxy = co_await getNativeCdcProxyAssignmentForTag(&tr, tag); + } const UID selectedProxy = sharedTagProxy.present() ? sharedTagProxy.get() : proxyId; tr.set(cdcProxyKeyFor(streamId, selectedProxy), Value()); + if (!sharedTagProxy.present()) { + tr.set(cdcTagOwnerKeyFor(tag), cdcTagOwnerValue(streamId)); + } signalNativeCdcProxyAssignmentChange(&tr); co_await tr.commit(); co_return streamId; @@ -474,6 +508,11 @@ Future removeNativeCdcStream(Database cx, Key name, CDCStreamId streamId, tr.clear(nameKey); tr.clear(cdcStreamKeyFor(streamId)); for (const Tag& tag : removedTags) { + const Key ownerKey = cdcTagOwnerKeyFor(tag); + Optional indexedStream = co_await tr.get(ownerKey); + if (indexedStream.present() && decodeCDCTagOwnerValue(indexedStream.get()) == streamId) { + tr.clear(ownerKey); + } tr.set(cdcRetiredTagPopKeyFor(tag), Value()); tr.atomicOp(cdcRetiredTagPopVersionKeyFor(tag), cdcVersionstampedMinVersionValue(), diff --git a/fdbclient/ProxyLoadBalance.h b/fdbclient/ProxyLoadBalance.h index 4d591b0dfd8..e0e9033c377 100644 --- a/fdbclient/ProxyLoadBalance.h +++ b/fdbclient/ProxyLoadBalance.h @@ -28,7 +28,7 @@ #include "fdbclient/DatabaseContext.h" #include "fdbclient/GrvProxyInterface.h" #include "fdbclient/NativeAPI.h" -#include "fdbrpc/LoadBalance.actor.h" +#include "fdbrpc/LoadBalance.h" // Stores constructor arguments so a request can be rebuilt on each retry. template @@ -78,19 +78,39 @@ AsyncResult commitProxyLoadBalance(Database cx, return commitProxyLoadBalance(cx, reqBuilder, channel, UseProvisionalProxies::False, cx->taskID, atMostOnce); } +enum class ProxyChangePriority { ReplyFirst, ChangeFirst }; + // Retries a GRV-proxy request whenever the proxy set changes before a reply arrives. -template +template AsyncResult grvProxyLoadBalance(Database cx, Builder reqBuilder, RequestStream GrvProxyInterface::* channel, AtMostOnce atMostOnce = AtMostOnce::False, ExplicitVoid = {}) { while (true) { - Future replyFuture = basicLoadBalance( - cx->getGrvProxies(UseProvisionalProxies::False), channel, reqBuilder.build(), cx->taskID, atMostOnce); - auto res = co_await race(replyFuture, cx->onProxiesChanged()); - if (res.index() == 0) { - co_return std::get<0>(std::move(res)); + if constexpr (priority == ProxyChangePriority::ChangeFirst) { + // Reset proxy backoff before constructing a request, and do not dispatch against an already-changed set. + auto proxiesChanged = cx->onProxiesChanged(); + if (proxiesChanged.isReady()) { + co_await proxiesChanged; + continue; + } + auto res = co_await race(std::move(proxiesChanged), + basicLoadBalance(cx->getGrvProxies(UseProvisionalProxies::False), + channel, + reqBuilder.build(), + cx->taskID, + atMostOnce)); + if (res.index() == 1) { + co_return std::get<1>(std::move(res)); + } + } else { + Future replyFuture = basicLoadBalance( + cx->getGrvProxies(UseProvisionalProxies::False), channel, reqBuilder.build(), cx->taskID, atMostOnce); + auto res = co_await race(replyFuture, cx->onProxiesChanged()); + if (res.index() == 0) { + co_return std::get<0>(std::move(res)); + } } } } diff --git a/fdbclient/RangeLock.cpp b/fdbclient/RangeLock.cpp index 7a76e4bd86b..85db5ad8491 100644 --- a/fdbclient/RangeLock.cpp +++ b/fdbclient/RangeLock.cpp @@ -87,6 +87,29 @@ Future removeRangeLockOwner(Database cx, RangeLockOwnerName ownerUniqueID) } RangeLockOwner owner = decodeRangeLockOwner(res.get()); ASSERT(owner.isValid()); + // Every page must share the deletion's read version and conflict ranges. + // Restart the scan on retry so a concurrent acquisition cannot leave an orphan. + Key beginKey = normalKeys.begin; + while (beginKey < normalKeys.end) { + RangeResult result = co_await krmGetRanges(&tr, rangeLockPrefix, KeyRangeRef(beginKey, normalKeys.end)); + ASSERT(result.size() > 1 && result.back().key > beginKey); + for (int i = 0; i < result.size() - 1; ++i) { + if (result[i].value.empty()) { + continue; + } + RangeLockStateSet locks = decodeRangeLockStateSet(result[i].value); + ASSERT(locks.isValid()); + for (const auto& [name, lock] : locks.getLocks()) { + if (lock.getOwnerUniqueId() == ownerUniqueID) { + TraceEvent(SevDebug, "RemoveRangeLockOwnerRejected") + .detail("Owner", ownerUniqueID) + .detail("Range", KeyRangeRef(result[i].key, result[i + 1].key)); + throw range_lock_reject(); + } + } + } + beginKey = result.back().key; + } tr.clear(rangeLockOwnerKeyFor(ownerUniqueID)); co_await tr.commit(); co_return; diff --git a/fdbclient/Schemas.cpp b/fdbclient/Schemas.cpp index 4e6ef262e7f..c90f571f04e 100644 --- a/fdbclient/Schemas.cpp +++ b/fdbclient/Schemas.cpp @@ -310,6 +310,102 @@ const KeyRef JSONSchemas::statusSchema = R"statusSchema( "p99":0.0, "p99.9":0.0 }, + "commit_batch_transactions":{ + "count":0, + "min":0.0, + "max":0.0, + "median":0.0, + "mean":0.0, + "p25":0.0, + "p90":0.0, + "p95":0.0, + "p99":0.0, + "p99.9":0.0 + }, + "commit_batch_bytes":{ + "count":0, + "min":0.0, + "max":0.0, + "median":0.0, + "mean":0.0, + "p25":0.0, + "p90":0.0, + "p95":0.0, + "p99":0.0, + "p99.9":0.0 + }, + "commit_batching_waiting":{ + "count":0, + "min":0.0, + "max":0.0, + "median":0.0, + "mean":0.0, + "p25":0.0, + "p90":0.0, + "p95":0.0, + "p99":0.0, + "p99.9":0.0 + }, + "commit_preresolution_latency":{ + "count":0, + "min":0.0, + "max":0.0, + "median":0.0, + "mean":0.0, + "p25":0.0, + "p90":0.0, + "p95":0.0, + "p99":0.0, + "p99.9":0.0 + }, + "commit_resolution_latency":{ + "count":0, + "min":0.0, + "max":0.0, + "median":0.0, + "mean":0.0, + "p25":0.0, + "p90":0.0, + "p95":0.0, + "p99":0.0, + "p99.9":0.0 + }, + "commit_postresolution_latency":{ + "count":0, + "min":0.0, + "max":0.0, + "median":0.0, + "mean":0.0, + "p25":0.0, + "p90":0.0, + "p95":0.0, + "p99":0.0, + "p99.9":0.0 + }, + "commit_tlog_logging_latency":{ + "count":0, + "min":0.0, + "max":0.0, + "median":0.0, + "mean":0.0, + "p25":0.0, + "p90":0.0, + "p95":0.0, + "p99":0.0, + "p99.9":0.0 + }, + "commit_reply_latency":{ + "count":0, + "min":0.0, + "max":0.0, + "median":0.0, + "mean":0.0, + "p25":0.0, + "p90":0.0, + "p95":0.0, + "p99":0.0, + "p99.9":0.0 + }, "grv_latency_bands":{ "$map": 1 }, @@ -536,6 +632,7 @@ const KeyRef JSONSchemas::statusSchema = R"statusSchema( }, "active_tss_count":0, "degraded_processes":0, + "degraded_multi_region":true, "database_available":true, "database_lock_state": { "locked": true, @@ -1454,6 +1551,7 @@ file is writable and has not been overwritten externally." }, "maintenance_zone":"0ccb4e0fdbdb5583010f6b77d9d10ece", "maintenance_seconds_remaining":1.0, + "degraded_multi_region":true, "data":{ "least_operating_space_bytes_log_server":0, "average_partition_size_bytes":0, diff --git a/fdbclient/SpecialKeySpace.cpp b/fdbclient/SpecialKeySpace.cpp index 77f8ae4993c..3a60697c8e0 100644 --- a/fdbclient/SpecialKeySpace.cpp +++ b/fdbclient/SpecialKeySpace.cpp @@ -1072,10 +1072,7 @@ Future checkExclusion(Database db, } } - } - // NOTE: ActorCompiler only accepts Error& or ... (for std::exception), it is not possible to capture - // std::exception& - catch (...) { + } catch (...) { *msg = ManagementAPIError::toJsonString( false, markFailed ? "exclude failed" : "exclude", errorString + "General exception raised.\n"); co_return false; @@ -1257,6 +1254,7 @@ Future ExclusionInProgressActor(ReadYourWritesTransaction* ryw, Key std::vector excludedLocalities = fExcludedLocalities.get(); // Decode the excluded localities to check if any server is excluded by locality. std::vector> decodedExcludedLocalities; + decodedExcludedLocalities.reserve(excludedLocalities.size()); for (auto& excludedLocality : excludedLocalities) { decodedExcludedLocalities.push_back(decodeLocality(excludedLocality)); } diff --git a/fdbclient/StorageServerInterface.cpp b/fdbclient/StorageServerInterface.cpp index 3c731c3a46d..e7019c5bbb3 100644 --- a/fdbclient/StorageServerInterface.cpp +++ b/fdbclient/StorageServerInterface.cpp @@ -22,6 +22,7 @@ // FIXME: actually it should be renamed "ReplyComparison" because TSS vs SS // is just one use case. This code is agnostic to the specific use cases. // Fundamentally it is just about comparing replies. Where they came from is incidental. +#include "fdbclient/ProxyLoadBalanceMetrics.h" #include "fdbclient/StorageServerInterface.h" #include "crc32/crc32c.h" // for crc32c_append, to checksum values in tss trace events @@ -518,6 +519,78 @@ void TSSMetrics::recordLatency(const OverlappingChangeFeedsRequest& req, double // ------------------- +namespace { +class LoadBalanceTestInterface { +public: + PublicRequestStream waitMetrics; + + UID id() const { return waitMetrics.getEndpoint().token; } + std::string toString() const { return id().shortString(); } +}; + +Future replyToWaitMetricsRequest(FutureStream requests) { + ReplyPromise reply; + { + WaitMetricsRequest request = co_await requests; + reply = std::move(request.reply); + } + reply.send(StorageMetrics()); + co_return; +} +} // namespace + +TEST_CASE("/fdbclient/LoadBalance/releasesCompletedRequest") { + StorageServerInterface storageServer; + FutureStream requests = storageServer.waitMetrics.getFuture(); + IFailureMonitor::failureMonitor().setStatus(storageServer.waitMetrics.getEndpoint().getPrimaryAddress(), + FailureStatus(false)); + auto server = makeReference>(storageServer); + auto alternatives = makeReference>>( + std::vector>>{ server }); + + WaitMetricsRequest request( + 0, KeyRangeRef("load-balance-begin"_sr, "load-balance-end"_sr), StorageMetrics(), StorageMetrics()); + ReplyPromise reply = request.reply; + Future result = loadBalance(alternatives, &StorageServerInterface::waitMetrics, std::move(request)); + + ASSERT(!result.isReady()); + ASSERT(alternatives->debugGetReferenceCount() > 1); + ASSERT(reply.getPromiseReferenceCount() > 1); + co_await replyToWaitMetricsRequest(requests); + + ASSERT(result.isReady()); + ASSERT(!result.isError()); + ASSERT(alternatives->debugGetReferenceCount() == 1); + ASSERT(reply.getPromiseReferenceCount() == 1); + co_return; +} + +TEST_CASE("/fdbclient/BasicLoadBalance/releasesCompletedRequest") { + LoadBalanceTestInterface server; + FutureStream requests = server.waitMetrics.getFuture(); + IFailureMonitor::failureMonitor().setStatus(server.waitMetrics.getEndpoint().getPrimaryAddress(), + FailureStatus(false)); + auto alternatives = makeReference>( + std::vector{ server }); + + WaitMetricsRequest request( + 0, KeyRangeRef("load-balance-begin"_sr, "load-balance-end"_sr), StorageMetrics(), StorageMetrics()); + ReplyPromise reply = request.reply; + Future result = + basicLoadBalance(alternatives, &LoadBalanceTestInterface::waitMetrics, std::move(request)); + + ASSERT(!result.isReady()); + ASSERT(alternatives->debugGetReferenceCount() > 1); + ASSERT(reply.getPromiseReferenceCount() > 1); + co_await replyToWaitMetricsRequest(requests); + + ASSERT(result.isReady()); + ASSERT(!result.isError()); + ASSERT(alternatives->debugGetReferenceCount() == 1); + ASSERT(reply.getPromiseReferenceCount() == 1); + co_return; +} + TEST_CASE("/StorageServerInterface/TSSCompare/TestComparison") { printf("testing tss comparisons\n"); diff --git a/fdbclient/SystemData.cpp b/fdbclient/SystemData.cpp index 40d88181e21..0ea881e10d4 100644 --- a/fdbclient/SystemData.cpp +++ b/fdbclient/SystemData.cpp @@ -787,6 +787,7 @@ const KeyRangeRef cdcStreamNameKeys("\xff/cdc/name/"_sr, "\xff/cdc/name0"_sr); const KeyRef cdcMaxStreamIdKey = "\xff/cdc/maxStreamId"_sr; const KeyRangeRef cdcStreamKeys("\xff/cdc/keys/"_sr, "\xff/cdc/keys0"_sr); const KeyRangeRef cdcTagHistoryKeys("\xff/cdc/tagHistory/"_sr, "\xff/cdc/tagHistory0"_sr); +const KeyRangeRef cdcTagOwnerKeys("\xff\x02/cdc/tagOwner/"_sr, "\xff\x02/cdc/tagOwner0"_sr); const KeyRangeRef cdcMinVersionKeys("\xff\x02/cdc/minVersion/"_sr, "\xff\x02/cdc/minVersion0"_sr); const KeyRangeRef cdcRetiredTagPopKeys("\xff/cdc/retiredTagPop/"_sr, "\xff/cdc/retiredTagPop0"_sr); const KeyRangeRef cdcRetiredTagPopVersionKeys("\xff\x02/cdc/retiredTagPopVersion/"_sr, @@ -884,6 +885,28 @@ CDCTagHistoryEntry decodeCDCTagHistoryKey(KeyRef const& key) { return CDCTagHistoryEntry(streamId, bigEndian64(encodedVersion), tag); } +Key cdcTagOwnerKeyFor(Tag tag) { + BinaryWriter wr(Unversioned()); + wr.serializeBytes(cdcTagOwnerKeys.begin); + wr << tag; + return wr.toValue(); +} + +Tag decodeCDCTagOwnerKey(KeyRef const& key) { + Tag tag; + BinaryReader reader(key.removePrefix(cdcTagOwnerKeys.begin), Unversioned()); + reader >> tag; + return tag; +} + +Value cdcTagOwnerValue(CDCStreamId streamId) { + return cdcStreamNameValue(streamId); +} + +CDCStreamId decodeCDCTagOwnerValue(ValueRef const& value) { + return decodeCDCStreamNameValue(value); +} + Key cdcMinVersionKeyFor(CDCStreamId streamId) { BinaryWriter wr(Unversioned()); wr.serializeBytes(cdcMinVersionKeys.begin); @@ -1931,6 +1954,11 @@ TEST_CASE("/SystemData/NativeCDC") { ASSERT_EQ(decodeCDCMaxStreamIdValue(cdcMaxStreamIdValue(streamId)), streamId); ASSERT_EQ(decodeCDCStreamKey(cdcStreamKeyFor(streamId)), streamId); ASSERT_EQ(decodeCDCStreamKeysValue(cdcStreamKeysValue(keys)), keys); + const Key tagOwnerKey = cdcTagOwnerKeyFor(tag); + ASSERT_EQ(decodeCDCTagOwnerKey(tagOwnerKey), tag); + ASSERT(cdcTagOwnerKeys.contains(tagOwnerKey)); + ASSERT(nonMetadataSystemKeys.contains(tagOwnerKey)); + ASSERT_EQ(decodeCDCTagOwnerValue(cdcTagOwnerValue(streamId)), streamId); ASSERT_EQ(decodeCDCMinVersionKey(cdcMinVersionKeyFor(streamId)), streamId); ASSERT_EQ(decodeCDCMinVersionValue(cdcMinVersionValue(minVersion)), minVersion); ASSERT(nonMetadataSystemKeys.contains(cdcMinVersionKeyFor(streamId))); diff --git a/fdbclient/Tracing.cpp b/fdbclient/Tracing.cpp index 229e1a0a73d..47b5ca3f353 100644 --- a/fdbclient/Tracing.cpp +++ b/fdbclient/Tracing.cpp @@ -363,6 +363,7 @@ ITracer::~ITracer() = default; Span& Span::operator=(Span&& o) { if (begin > 0.0 && context.isSampled()) { + ensureAddressAttribute(); end = g_network->now(); g_tracer->trace(*this); } @@ -391,6 +392,7 @@ Span& Span::operator=(Span&& o) { Span::~Span() { if (begin > 0.0 && context.isSampled()) { + ensureAddressAttribute(); end = g_network->now(); g_tracer->trace(*this); } @@ -482,6 +484,48 @@ TEST_CASE("/flow/Tracing/AddAttributes") { return Void(); }; +TEST_CASE("/flow/Tracing/MoveConstructionPreservesAttributes") { + Span source("span_move_attrs"_loc, + SpanContext(deterministicRandom()->randomUniqueID(), + deterministicRandom()->randomUInt64(), + TraceFlags::sampled)); + source.addAttribute("operation"_sr, "getValue"_sr); + ASSERT_EQ(source.attributes.size(), 2); + + Span moved(std::move(source)); + // NOLINTNEXTLINE(bugprone-use-after-move): The move constructor explicitly resets the source attributes. + ASSERT_EQ(source.attributes.size(), 0); + ASSERT_EQ(moved.attributes.size(), 2); + ASSERT_EQ(moved.attributes[0].key, "address"_sr); + ASSERT_EQ(moved.attributes[1], KeyValueRef("operation"_sr, "getValue"_sr)); + return Void(); +}; + +TEST_CASE("/flow/Tracing/UnsampledSpanSkipsAddress") { + Span span("unsampled_span"_loc, SpanContext(UID(1, 2), 3, TraceFlags::unsampled)); + ASSERT(!span.context.isSampled()); + ASSERT(span.attributes.empty()); + + span.setParent(SpanContext(UID(4, 5), 6, TraceFlags::sampled)); + ASSERT(span.context.isSampled()); + ASSERT_EQ(span.attributes.size(), 1); + ASSERT_EQ(span.attributes[0].key, "address"_sr); + + span.setParent(SpanContext(UID(7, 8), 9, TraceFlags::unsampled)); + span.setParent(SpanContext(UID(10, 11), 12, TraceFlags::sampled)); + ASSERT_EQ(span.attributes.size(), 1); + + Span linked("linked_span"_loc); + ASSERT(linked.attributes.empty()); + linked.addAttribute("operation"_sr, "getValue"_sr); + linked.addLink(SpanContext(UID(13, 14), 15, TraceFlags::sampled)); + ASSERT(linked.context.isSampled()); + ASSERT_EQ(linked.attributes.size(), 2); + ASSERT_EQ(linked.attributes[0], KeyValueRef("operation"_sr, "getValue"_sr)); + ASSERT_EQ(linked.attributes[1].key, "address"_sr); + return Void(); +}; + TEST_CASE("/flow/Tracing/AddLinks") { Span span1("span_with_links"_loc); ASSERT(!span1.context.isSampled()); diff --git a/fdbclient/VersionVector.cpp b/fdbclient/VersionVector.cpp index b15a3c76fd9..6884741196b 100644 --- a/fdbclient/VersionVector.cpp +++ b/fdbclient/VersionVector.cpp @@ -120,6 +120,7 @@ void populateVersionVector(VersionVector& vv, } // Populate ids. + ids.reserve(tagCount > 0 ? tagCount : 0); for (int i = 0; i < tagCount; i++) { // Some of the ids could be duplicates, that's fine. ids.push_back(deterministicRandom()->randomInt(0, maxTagId)); diff --git a/fdbclient/azurestorage.cmake b/fdbclient/azurestorage.cmake deleted file mode 100644 index b9678249488..00000000000 --- a/fdbclient/azurestorage.cmake +++ /dev/null @@ -1,17 +0,0 @@ -cmake_minimum_required(VERSION 3.13) - -project(azurestorage-download) - -include(ExternalProject) -ExternalProject_Add(azurestorage - GIT_REPOSITORY https://github.com/Azure/azure-storage-cpplite.git - GIT_TAG 11e1f98b021446ef340f4886796899a6eb1ad9a5 # v0.3.0 - SOURCE_DIR "${CMAKE_CURRENT_BINARY_DIR}/azurestorage-src" - BINARY_DIR "${CMAKE_CURRENT_BINARY_DIR}/azurestorage-build" - CMAKE_ARGS "-DCMAKE_BUILD_TYPE=Release" - CONFIGURE_COMMAND "" - BUILD_COMMAND "" - INSTALL_COMMAND "" - TEST_COMMAND "" - BUILD_BYPRODUCTS "${CMAKE_CURRENT_BINARY_DIR}/libazure-storage-lite.a" -) diff --git a/fdbclient/include/fdbclient/GrvProxyInterface.h b/fdbclient/include/fdbclient/GrvProxyInterface.h index 4f74142ae19..3c57708ce75 100644 --- a/fdbclient/include/fdbclient/GrvProxyInterface.h +++ b/fdbclient/include/fdbclient/GrvProxyInterface.h @@ -25,7 +25,7 @@ #include "fdbclient/VersionVector.h" #include "flow/FileIdentifier.h" #include "fdbrpc/fdbrpc.h" -#include "fdbrpc/LoadBalance.actor.h" +#include "fdbrpc/LoadBalance.h" #include "fdbrpc/Stats.h" #include "fdbrpc/TimedRequest.h" #include "fdbclient/FDBTypes.h" diff --git a/fdbclient/include/fdbclient/KeyBackedTypes.h b/fdbclient/include/fdbclient/KeyBackedTypes.h index d928b3722f0..ce9203b8043 100644 --- a/fdbclient/include/fdbclient/KeyBackedTypes.h +++ b/fdbclient/include/fdbclient/KeyBackedTypes.h @@ -40,7 +40,7 @@ #include "flow/ThreadHelper.h" #define KEYBACKEDTYPES_DEBUG 0 -#if KEYBACKEDTYPES_DEBUG || !defined(NO_INTELLISENSE) +#if KEYBACKEDTYPES_DEBUG #define kbt_debug fmt::print #else #define kbt_debug(...) @@ -176,6 +176,7 @@ struct TupleCodec> { static std::vector unpack(Standalone const& val) { Tuple t = Tuple::unpack(val); std::vector v; + v.reserve(t.size()); for (int i = 0; i < t.size(); i++) { v.push_back(TupleCodec::unpack(t.getString(i))); diff --git a/fdbclient/include/fdbclient/ManagementAPI.h b/fdbclient/include/fdbclient/ManagementAPI.h index 589f9eb4049..2f0934ac847 100644 --- a/fdbclient/include/fdbclient/ManagementAPI.h +++ b/fdbclient/include/fdbclient/ManagementAPI.h @@ -204,7 +204,7 @@ Future getBulkLoadTask(Transaction* tr, Future addBulkLoadJobToHistory(Transaction* tr, BulkLoadJobState jobState); // Get all past bulkLoad jobs from history map -AsyncResult> getBulkLoadJobFromHistory(Database cx); +AsyncResult> getBulkLoadJobFromHistory(Database cx, bool lockAware = false); // Erase all bulkLoad job history metadata if jobId is not provided. Otherwise, erase the job with the given jobId. Future clearBulkLoadJobHistory(Database cx, Optional jobId = Optional()); diff --git a/fdbclient/include/fdbclient/NativeAPI.h b/fdbclient/include/fdbclient/NativeAPI.h index dccdd436455..2ed2fcdb31b 100644 --- a/fdbclient/include/fdbclient/NativeAPI.h +++ b/fdbclient/include/fdbclient/NativeAPI.h @@ -583,11 +583,11 @@ inline uint64_t getWriteOperationCost(uint64_t bytes) { // Measured in bytes, rounded up to the nearest page size. inline uint64_t getReadOperationCost(uint64_t bytes) { - if (bytes == 0) { - return CLIENT_KNOBS->TAG_THROTTLING_PAGE_SIZE; - } else { - return ((bytes - 1) / CLIENT_KNOBS->TAG_THROTTLING_PAGE_SIZE + 1) * CLIENT_KNOBS->TAG_THROTTLING_PAGE_SIZE; + const uint64_t pageSize = CLIENT_KNOBS->TAG_THROTTLING_PAGE_SIZE; + if (bytes <= pageSize) { + return pageSize; } + return ((bytes - 1) / pageSize + 1) * pageSize; } // Create a transaction to set the value of system key \xff/conf/perpetual_storage_wiggle. If enable == true, the value diff --git a/fdbclient/include/fdbclient/RangeLock.h b/fdbclient/include/fdbclient/RangeLock.h index cd1721814ae..02186ee3c7b 100644 --- a/fdbclient/include/fdbclient/RangeLock.h +++ b/fdbclient/include/fdbclient/RangeLock.h @@ -154,6 +154,7 @@ struct RangeLockStateSet { std::vector getAllLockStats() const { std::vector res; + res.reserve(locks.size()); for (const auto& [name, lock] : locks) { res.push_back(lock); } @@ -245,7 +246,8 @@ struct RangeLockStateSet { // A range can only be locked by a registered owner. Future registerRangeLockOwner(Database cx, RangeLockOwnerName ownerUniqueID, std::string description); -// Remove an owner from the database metadata. +// Remove an owner only if it holds no locks; otherwise throw range_lock_reject. +// An already absent owner is a no-op. Future removeRangeLockOwner(Database cx, RangeLockOwnerName ownerUniqueID); // Get all registered rangeLock owners. diff --git a/fdbclient/include/fdbclient/StackLineage.h b/fdbclient/include/fdbclient/StackLineage.h index 41235e35598..cb0bc89749e 100644 --- a/fdbclient/include/fdbclient/StackLineage.h +++ b/fdbclient/include/fdbclient/StackLineage.h @@ -33,6 +33,7 @@ struct StackLineageCollector : IALPCollector { auto vec = lineage->stack(&StackLineage::actorName); std::vector res; + res.reserve(vec.size()); for (const auto& str : vec) { res.push_back(std::string_view(reinterpret_cast(str.begin()), str.size())); } diff --git a/fdbclient/include/fdbclient/StorageServerInterface.h b/fdbclient/include/fdbclient/StorageServerInterface.h index 309edf9e62e..4f290630d00 100644 --- a/fdbclient/include/fdbclient/StorageServerInterface.h +++ b/fdbclient/include/fdbclient/StorageServerInterface.h @@ -22,6 +22,8 @@ #define FDBCLIENT_STORAGESERVERINTERFACE_H #pragma once +#include + #include "fdbclient/Audit.h" #include "fdbclient/BulkDumping.h" #include "fdbclient/FDBTypes.h" @@ -31,7 +33,7 @@ #include "fdbrpc/Locality.h" #include "fdbrpc/QueueModel.h" #include "fdbrpc/fdbrpc.h" -#include "fdbrpc/LoadBalance.actor.h" +#include "fdbrpc/LoadBalance.h" #include "fdbrpc/Stats.h" #include "fdbrpc/TimedRequest.h" #include "fdbclient/TSSComparison.h" @@ -204,7 +206,7 @@ struct GetValueReply : public LoadBalancedReply { bool cached; GetValueReply() : cached(false) {} - GetValueReply(Optional value, bool cached) : value(value), cached(cached) {} + GetValueReply(Optional value, bool cached) : value(std::move(value)), cached(cached) {} template void serialize(Ar& ar) { @@ -233,8 +235,8 @@ struct GetValueRequest : TimedRequest { Optional tags, Optional options, VersionVector latestCommitVersions) - : spanContext(spanContext), key(key), version(ver), tags(tags), options(options), - ssLatestCommitVersions(latestCommitVersions) {} + : spanContext(spanContext), key(key), version(ver), tags(std::move(tags)), options(options), + ssLatestCommitVersions(std::move(latestCommitVersions)) {} template void serialize(Ar& ar) { @@ -496,8 +498,8 @@ struct GetKeyRequest : TimedRequest { Optional tags, Optional options, VersionVector latestCommitVersions) - : spanContext(spanContext), sel(sel), version(version), tags(tags), options(options), - ssLatestCommitVersions(latestCommitVersions) {} + : spanContext(spanContext), sel(sel), version(version), tags(std::move(tags)), options(options), + ssLatestCommitVersions(std::move(latestCommitVersions)) {} template void serialize(Ar& ar) { @@ -1269,6 +1271,6 @@ inline int mvccStorageBytes(int mutationBytes) { (mutationBytes + MutationRef::OVERHEAD_BYTES) * 2; } -#include "fdbclient/StorageServerLoadBalance.actor.h" +#include "fdbclient/StorageServerLoadBalance.h" #endif diff --git a/fdbclient/include/fdbclient/StorageServerLoadBalance.actor.h b/fdbclient/include/fdbclient/StorageServerLoadBalance.h similarity index 87% rename from fdbclient/include/fdbclient/StorageServerLoadBalance.actor.h rename to fdbclient/include/fdbclient/StorageServerLoadBalance.h index 65a68751a81..aa1aab94723 100644 --- a/fdbclient/include/fdbclient/StorageServerLoadBalance.actor.h +++ b/fdbclient/include/fdbclient/StorageServerLoadBalance.h @@ -1,5 +1,5 @@ /* - * StorageServerLoadBalance.actor.h + * StorageServerLoadBalance.h * * This source file is part of the FoundationDB open source project * @@ -20,21 +20,14 @@ #pragma once -// This header is included at the end of StorageServerInterface.h, after the storage-server request and reply types are -// defined. Keeping these hooks here lets fdbrpc's generic load balancer stay unaware of storage-specific comparisons. -#if defined(NO_INTELLISENSE) && !defined(FDBCLIENT_STORAGESERVERLOADBALANCE_ACTOR_G_H) -#define FDBCLIENT_STORAGESERVERLOADBALANCE_ACTOR_G_H -#include "fdbclient/StorageServerLoadBalance.actor.g.h" -#elif !defined(FDBCLIENT_STORAGESERVERLOADBALANCE_ACTOR_H) -#define FDBCLIENT_STORAGESERVERLOADBALANCE_ACTOR_H - #include "fdbclient/SimulationCapabilities.h" -#include "flow/actorcompiler.h" // This must be the last #include. +#include "fdbclient/StorageServerInterface.h" +#include "flow/CoroUtils.h" enum ComparisonType { TSS_COMPARISON, REPLICA_COMPARISON }; // FIXME: use a less obscure name than `P` here -ACTOR template +template Future tssComparison(Req req, Future> fSource, Future> fTss, @@ -42,35 +35,30 @@ Future tssComparison(Req req, uint64_t srcEndpointId, Reference>> ssTeam, RequestStream StorageServerInterface::* channel) { - state double startTime = now(); - state Future>> fTssWithTimeout = timeout(fTss, FLOW_KNOBS->LOAD_BALANCE_TSS_TIMEOUT); - state int finished = 0; - state double srcEndTime; - state double tssEndTime; + double startTime = now(); + Future>> fTssWithTimeout = timeout(fTss, FLOW_KNOBS->LOAD_BALANCE_TSS_TIMEOUT); + int finished = 0; + double srcEndTime{ 0 }; + double tssEndTime{ 0 }; // we want to record ss/tss errors to metrics - state int srcErrorCode = error_code_success; - state int tssErrorCode = error_code_success; - state ErrorOr src; - state Optional> tss; - - loop { - choose { - when(wait(store(src, fSource))) { - srcEndTime = now(); - fSource = Never(); - finished++; - if (finished == 2) { - break; - } - } - when(wait(store(tss, fTssWithTimeout))) { - tssEndTime = now(); - fTssWithTimeout = Never(); - finished++; - if (finished == 2) { - break; - } - } + int srcErrorCode = error_code_success; + int tssErrorCode = error_code_success; + ErrorOr src; + Optional> tss; + + while (true) { + auto res = co_await race(fSource, fTssWithTimeout); + if (res.index() == 0) { + src = std::get<0>(std::move(res)); + srcEndTime = now(); + fSource = Never(); + } else { + tss = std::get<1>(std::move(res)); + tssEndTime = now(); + fTssWithTimeout = Never(); + } + if (++finished == 2) { + break; } } ++tssData.metrics->requests; @@ -99,10 +87,10 @@ Future tssComparison(Req req, if (!TSS_doCompare(src.get(), tss.get().get())) { CODE_PROBE(true, "TSS Mismatch"); - state TraceEvent mismatchEvent( - (fdbSimulationHasCapability(FDBSimulationCapability::WarnOnStorageMismatch)) ? SevWarnAlways - : SevError, - LB_mismatchTraceName(req, TSS_COMPARISON)); + TraceEvent mismatchEvent((fdbSimulationHasCapability(FDBSimulationCapability::WarnOnStorageMismatch)) + ? SevWarnAlways + : SevError, + LB_mismatchTraceName(req, TSS_COMPARISON)); mismatchEvent.setMaxEventLength(FLOW_KNOBS->TSS_LARGE_TRACE_SIZE); mismatchEvent.detail("TSSID", tssData.tssId); @@ -111,7 +99,7 @@ Future tssComparison(Req req, // if there is more than 1 SS in the team, attempt to verify that the other SS servers have the same // data - state std::vector>> restOfTeamFutures; + std::vector>> restOfTeamFutures; restOfTeamFutures.reserve(ssTeam->size() - 1); for (int i = 0; i < ssTeam->size(); i++) { RequestStream const* si = &ssTeam->get(i, channel); @@ -122,7 +110,7 @@ Future tssComparison(Req req, } } - wait(waitForAllReady(restOfTeamFutures)); + co_await waitForAllReady(restOfTeamFutures); int numError = 0; int numMatchSS = 0; @@ -192,24 +180,22 @@ Future tssComparison(Req req, .detail("SSError", srcErrorCode) .detail("TSSError", tssErrorCode); } - - return Void(); } -ACTOR template +template Future replicaComparison(Req req, Future> fSource, uint64_t srcEndpointId, Reference>> ssTeam, RequestStream StorageServerInterface::* channel, int requiredReplicas) { - state ErrorOr src; + ErrorOr src; if (ssTeam->size() <= 1 || requiredReplicas == 0) { - return Void(); + co_return; } - wait(store(src, fSource)); + co_await store(src, fSource); if (src.isError()) { ASSERT_WE_THINK(false); // TODO: Change this into an ASSERT after getting enough test coverage. @@ -217,7 +203,7 @@ Future replicaComparison(Req req, throw src.getError(); } } else { - state Optional srcLB = getLoadBalancedReply(&src.get()); + Optional srcLB = getLoadBalancedReply(&src.get()); if (srcLB.present() && srcLB.get().error.present()) { ASSERT_WE_THINK(false); // TODO: Change this into an ASSERT after getting enough test coverage. @@ -254,7 +240,7 @@ Future replicaComparison(Req req, .detail("AvailableReplicas", candidates.size()); } } - state std::vector>>> restOfTeamFutures; + std::vector>>> restOfTeamFutures; restOfTeamFutures.reserve(numReplicaToRead); // Randomly select numReplicaToRead SSes to read from deterministicRandom()->randomShuffle(candidates); @@ -274,7 +260,7 @@ Future replicaComparison(Req req, : timeout(errorOr(si->getReply(req)), FLOW_KNOBS->LOAD_BALANCE_FETCH_REPLICA_TIMEOUT))); } - wait(waitForAllReady(restOfTeamFutures)); + co_await waitForAllReady(restOfTeamFutures); int numError = 0; int numMismatch = 0; @@ -358,7 +344,6 @@ Future replicaComparison(Req req, } } } - return Void(); } template @@ -432,6 +417,3 @@ struct LoadBalanceRequestHooks + #include "flow/network.h" #include "flow/IRandom.h" #include "flow/Arena.h" @@ -135,10 +137,7 @@ class Span { begin(g_network->now()) { this->kind = SpanKind::SERVER; this->status = SpanStatus::OK; - this->attributes.push_back( - // this->arena, KeyValueRef("address"_sr, StringRef(this->arena, "localhost:4000"))); - this->arena, - KeyValueRef("address"_sr, StringRef(this->arena, FlowTransport::transport().getLocalAddressAsString()))); + ensureAddressAttribute(); } // Construct Span with a location, parent, and optional links. @@ -170,6 +169,7 @@ class Span { end = o.end; links = o.links; events = o.events; + attributes = std::exchange(o.attributes, decltype(o.attributes)()); status = o.status; o.context = SpanContext(); o.parentContext = SpanContext(); @@ -207,6 +207,7 @@ class Span { context.traceID = deterministicRandom()->randomUniqueID(); context.spanID = deterministicRandom()->randomUInt64(); } + ensureAddressAttribute(); } return *this; } @@ -239,9 +240,25 @@ class Span { context.traceID = parent.traceID; context.spanID = deterministicRandom()->randomUInt64(); context.m_Flags = parent.m_Flags; + ensureAddressAttribute(); return *this; } + void ensureAddressAttribute() { + // Unsampled spans are common on the client and storage read paths. Avoid copying the cached address into a new + // arena until the span is actually eligible to be emitted. + if (!context.isSampled()) { + return; + } + for (const auto& attribute : attributes) { + if (attribute.key == "address"_sr) { + return; + } + } + attributes.push_back( + arena, KeyValueRef("address"_sr, StringRef(arena, FlowTransport::transport().getLocalAddressAsString()))); + } + Arena arena; SpanContext context; Location location; diff --git a/fdbrpc/ActorFuzz.actor.cpp b/fdbrpc/ActorFuzz.actor.cpp deleted file mode 100644 index b9bb9e8af1c..00000000000 --- a/fdbrpc/ActorFuzz.actor.cpp +++ /dev/null @@ -1,922 +0,0 @@ - -/* - * ActorFuzz.actor.cpp - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -// THIS FILE WAS GENERATED BY actorFuzz.py; DO NOT MODIFY IT DIRECTLY - -#include "ActorFuzz.h" -#include "flow/actorcompiler.h" // has to be last include - -#ifndef WIN32 - -ACTOR Future actorFuzz0(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - try { - outputStream.send(390229); - state std::vector list637154; - list637154.push_back(1); - list637154.push_back(2); - list637154.push_back(3); - for (auto i : list637154) { - (void)i; - outputStream.send(596271); - return 574865; - } - outputStream.send(616994); - } catch (...) { - outputStream.send(282473); - state int i538813; - for (i538813 = 0; i538813 < 5; i538813++) { - outputStream.send(202451); - state std::vector list106964; - list106964.push_back(1); - list106964.push_back(2); - list106964.push_back(3); - for (auto i : list106964) { - (void)i; - outputStream.send(306539); - if ((++ifstate & 1) == 1) { - outputStream.send(980726); - try { - outputStream.send(103523); - if ((++ifstate & 1) == 1) { - outputStream.send(750915); - int input = waitNext(inputStream); - outputStream.send(input + 714763); - outputStream.send(838596); - } else { - outputStream.send(883911); - try { - outputStream.send(625121); - int input = waitNext(inputStream); - outputStream.send(input + 826693); - outputStream.send(593359); - } catch (...) { - outputStream.send(376812); - return 247718; - } - outputStream.send(855094); - } - outputStream.send(547309); - } catch (...) { - outputStream.send(822404); - break; - } - outputStream.send(496422); - } else { - outputStream.send(972353); - break; - } - outputStream.send(732279); - } - outputStream.send(710432); - } - outputStream.send(492398); - } - return 240968; -} - -ACTOR Future actorFuzz1(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - state int i661806; - for (i661806 = 0; i661806 < 5; i661806++) { - outputStream.send(477566); - try { - outputStream.send(815578); - continue; - } catch (...) { - outputStream.send(787898); - state std::vector list781874; - list781874.push_back(1); - list781874.push_back(2); - list781874.push_back(3); - for (auto i : list781874) { - (void)i; - outputStream.send(625656); - try { - outputStream.send(114830); - state int i996799; - for (i996799 = 0; i996799 < 5; i996799++) { - outputStream.send(188397); - try { - outputStream.send(649779); - wait(error); // throw operation_failed() - outputStream.send(705317); - break; - } catch (...) { - outputStream.send(761942); - return 469895; - } - } - outputStream.send(613669); - } catch (...) { - outputStream.send(703172); - return 811053; - } - outputStream.send(329141); - } - outputStream.send(647010); - } - outputStream.send(287946); - break; - } - return 917160; -} - -ACTOR Future actorFuzz2(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - int input = waitNext(inputStream); - outputStream.send(input + 475677); - return 930237; -} - -ACTOR Future actorFuzz3(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - throw_operation_failed(); - return 499525; -} - -ACTOR Future actorFuzz4(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - try { - outputStream.send(180600); - state std::vector list613889; - list613889.push_back(1); - list613889.push_back(2); - list613889.push_back(3); - for (auto i : list613889) { - (void)i; - outputStream.send(177605); - continue; - } - outputStream.send(954508); - } catch (...) { - outputStream.send(461484); - return 117481; - } - return 810052; -} - -ACTOR Future actorFuzz5(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - wait(error); // throw operation_failed() - outputStream.send(382339); - if ((++ifstate & 1) == 1) { - outputStream.send(655938); - state int i139204; - for (i139204 = 0; i139204 < 5; i139204++) { - outputStream.send(481427); - wait(error); // throw operation_failed() - outputStream.send(577939); - } - outputStream.send(252859); - } - return 273288; -} - -ACTOR Future actorFuzz6(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - state int i956829; - for (i956829 = 0; i956829 < 5; i956829++) { - outputStream.send(320321); - state int i350925; - for (i350925 = 0; i350925 < 5; i350925++) { - outputStream.send(266526); - try { - outputStream.send(762336); - break; - } catch (...) { - outputStream.send(391672); - continue; - } - } - outputStream.send(463730); - } - return 945289; -} - -ACTOR Future actorFuzz7(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - try { - outputStream.send(406152); - try { - outputStream.send(478841); - try { - outputStream.send(609181); - state int i510243; - for (i510243 = 0; i510243 < 5; i510243++) { - outputStream.send(634881); - state std::vector list596949; - list596949.push_back(1); - list596949.push_back(2); - list596949.push_back(3); - for (auto i : list596949) { - (void)i; - outputStream.send(253861); - int input = waitNext(inputStream); - outputStream.send(input + 591023); - outputStream.send(240597); - } - outputStream.send(415949); - int input = waitNext(inputStream); - outputStream.send(input + 165335); - outputStream.send(478331); - } - outputStream.send(331905); - } catch (...) { - outputStream.send(686252); - return 997694; - } - outputStream.send(946924); - state std::vector list833282; - list833282.push_back(1); - list833282.push_back(2); - list833282.push_back(3); - for (auto i : list833282) { - (void)i; - outputStream.send(663973); - if ((++ifstate & 1) == 1) { - outputStream.send(797073); - wait(error); // throw operation_failed() - outputStream.send(953652); - } - outputStream.send(807309); - } - outputStream.send(996672); - } catch (...) { - outputStream.send(971923); - state int i793430; - for (i793430 = 0; i793430 < 5; i793430++) { - outputStream.send(295772); - try { - outputStream.send(923567); - state std::vector list814034; - list814034.push_back(1); - list814034.push_back(2); - list814034.push_back(3); - for (auto i : list814034) { - (void)i; - outputStream.send(559259); - continue; - } - outputStream.send(325678); - } catch (...) { - outputStream.send(691889); - continue; - } - outputStream.send(679187); - } - outputStream.send(534407); - } - outputStream.send(814172); - } catch (...) { - outputStream.send(117532); - state std::vector list243466; - list243466.push_back(1); - list243466.push_back(2); - list243466.push_back(3); - for (auto i : list243466) { - (void)i; - outputStream.send(593203); - try { - outputStream.send(289002); - return 321054; - } catch (...) { - outputStream.send(540106); - return 919162; - } - } - outputStream.send(679173); - } - return 949658; -} - -ACTOR Future actorFuzz8(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - int input = waitNext(inputStream); - outputStream.send(input + 284937); - return 696473; -} - -ACTOR Future actorFuzz9(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - int input = waitNext(inputStream); - outputStream.send(input + 140463); - return 397424; -} - -ACTOR Future actorFuzz10(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - state int i832228; - for (i832228 = 0; i832228 < 5; i832228++) { - outputStream.send(543113); - wait(error); // throw operation_failed() - outputStream.send(780932); - } - return 402988; -} - -ACTOR Future actorFuzz11(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - wait(error); // throw operation_failed() - return 672734; -} - -ACTOR Future actorFuzz12(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - state std::vector list466120; - list466120.push_back(1); - list466120.push_back(2); - list466120.push_back(3); - for (auto i : list466120) { - (void)i; - outputStream.send(970588); - return 981887; - } - return 869298; -} - -ACTOR Future actorFuzz13(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - if ((++ifstate & 1) == 0) { - outputStream.send(571414); - return 591307; - } - return 861219; -} - -ACTOR Future actorFuzz14(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - state std::vector list370902; - list370902.push_back(1); - list370902.push_back(2); - list370902.push_back(3); - for (auto i : list370902) { - (void)i; - outputStream.send(527098); - continue; - } - return 628047; -} - -ACTOR Future actorFuzz15(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - state int i450301; - for (i450301 = 0; i450301 < 5; i450301++) { - outputStream.send(582389); - state std::vector list863601; - list863601.push_back(1); - list863601.push_back(2); - list863601.push_back(3); - for (auto i : list863601) { - (void)i; - outputStream.send(240216); - break; - } - outputStream.send(732317); - } - return 884781; -} - -ACTOR Future actorFuzz16(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - if ((++ifstate & 1) == 1) { - outputStream.send(943071); - state std::vector list811122; - list811122.push_back(1); - list811122.push_back(2); - list811122.push_back(3); - for (auto i : list811122) { - (void)i; - outputStream.send(492690); - if ((++ifstate & 1) == 1) { - outputStream.send(388192); - wait(error); // throw operation_failed() - outputStream.send(545437); - } else { - outputStream.send(908751); - state int i120459; - for (i120459 = 0; i120459 < 5; i120459++) { - outputStream.send(198776); - return 537939; - } - outputStream.send(649270); - } - outputStream.send(397872); - } - outputStream.send(493007); - } else { - outputStream.send(437137); - state std::vector list226908; - list226908.push_back(1); - list226908.push_back(2); - list226908.push_back(3); - for (auto i : list226908) { - (void)i; - outputStream.send(321651); - if ((++ifstate & 1) == 1) { - outputStream.send(396995); - state int i753710; - for (i753710 = 0; i753710 < 5; i753710++) { - outputStream.send(235407); - break; - } - outputStream.send(792039); - } - outputStream.send(659099); - } - outputStream.send(403928); - } - return 197156; -} - -ACTOR Future actorFuzz17(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - state std::vector list522792; - list522792.push_back(1); - list522792.push_back(2); - list522792.push_back(3); - for (auto i : list522792) { - (void)i; - outputStream.send(249436); - try { - outputStream.send(416782); - continue; - } catch (...) { - outputStream.send(237787); - loop { - outputStream.send(438476); - break; - } - outputStream.send(939594); - } - outputStream.send(670490); - if ((++ifstate & 1) == 0) { - outputStream.send(264281); - try { - outputStream.send(830283); - continue; - } catch (...) { - outputStream.send(157517); - continue; - } - } - outputStream.send(990392); - } - return 299183; -} - -ACTOR Future actorFuzz18(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - try { - outputStream.send(337649); - state int i910140; - for (i910140 = 0; i910140 < 5; i910140++) { - outputStream.send(395297); - break; - } - outputStream.send(807261); - } catch (...) { - outputStream.send(628394); - try { - outputStream.send(658059); - throw operation_failed(); - } catch (...) { - outputStream.send(787535); - if ((++ifstate & 1) == 1) { - outputStream.send(945855); - int input = waitNext(inputStream); - outputStream.send(input + 401313); - outputStream.send(483948); - } else { - outputStream.send(705433); - try { - outputStream.send(110258); - state std::vector list917536; - list917536.push_back(1); - list917536.push_back(2); - list917536.push_back(3); - for (auto i : list917536) { - (void)i; - outputStream.send(539878); - throw operation_failed(); - } - outputStream.send(265595); - } catch (...) { - outputStream.send(919259); - try { - outputStream.send(770240); - throw operation_failed(); - } catch (...) { - outputStream.send(383788); - throw operation_failed(); - } - } - outputStream.send(954545); - } - outputStream.send(365388); - } - outputStream.send(764202); - } - return 517901; -} - -ACTOR Future actorFuzz19(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - state std::vector list476707; - list476707.push_back(1); - list476707.push_back(2); - list476707.push_back(3); - for (auto i : list476707) { - (void)i; - outputStream.send(492598); - int input = waitNext(inputStream); - outputStream.send(input + 138186); - outputStream.send(742053); - } - return 592919; -} - -ACTOR Future actorFuzz20(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - if ((++ifstate & 1) == 0) { - outputStream.send(751400); - int input = waitNext(inputStream); - outputStream.send(input + 106231); - outputStream.send(139622); - } else { - outputStream.send(760082); - throw operation_failed(); - } - return 705285; -} - -ACTOR Future actorFuzz21(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - return 806394; -} - -ACTOR Future actorFuzz22(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - try { - outputStream.send(722878); - try { - outputStream.send(369302); - return 416748; - } catch (...) { - outputStream.send(568306); - state std::vector list504461; - list504461.push_back(1); - list504461.push_back(2); - list504461.push_back(3); - for (auto i : list504461) { - (void)i; - outputStream.send(827088); - if ((++ifstate & 1) == 0) { - outputStream.send(909504); - return 528584; - } - outputStream.send(275831); - } - outputStream.send(739194); - } - outputStream.send(456449); - } catch (...) { - outputStream.send(208944); - try { - outputStream.send(205829); - int input = waitNext(inputStream); - outputStream.send(input + 539161); - outputStream.send(820020); - } catch (...) { - outputStream.send(666594); - if ((++ifstate & 1) == 1) { - outputStream.send(153749); - return 657441; - } - outputStream.send(312545); - } - outputStream.send(803123); - } - outputStream.send(646039); - state std::vector list700360; - list700360.push_back(1); - list700360.push_back(2); - list700360.push_back(3); - for (auto i : list700360) { - (void)i; - outputStream.send(434654); - try { - outputStream.send(292762); - break; - } catch (...) { - outputStream.send(540935); - try { - outputStream.send(202527); - state int i246439; - for (i246439 = 0; i246439 < 5; i246439++) { - outputStream.send(141484); - continue; - } - outputStream.send(265555); - } catch (...) { - outputStream.send(506444); - int input = waitNext(inputStream); - outputStream.send(input + 279285); - outputStream.send(926817); - } - outputStream.send(957345); - } - outputStream.send(893732); - } - return 888702; -} - -ACTOR Future actorFuzz23(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - state std::vector list316142; - list316142.push_back(1); - list316142.push_back(2); - list316142.push_back(3); - for (auto i : list316142) { - (void)i; - outputStream.send(562792); - return 231437; - } - return 226698; -} - -ACTOR Future actorFuzz24(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - int input = waitNext(inputStream); - outputStream.send(input + 846672); - return 835175; -} - -ACTOR Future actorFuzz25(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - try { - outputStream.send(843261); - if ((++ifstate & 1) == 0) { - outputStream.send(166067); - if ((++ifstate & 1) == 1) { - outputStream.send(135307); - throw operation_failed(); - } else { - outputStream.send(150379); - int input = waitNext(inputStream); - outputStream.send(input + 234945); - outputStream.send(806946); - } - outputStream.send(908760); - } - outputStream.send(327560); - } catch (...) { - outputStream.send(573810); - if ((++ifstate & 1) == 0) { - outputStream.send(313835); - throw_operation_failed(); - outputStream.send(749685); - } - outputStream.send(706935); - } - return 592398; -} - -ACTOR Future actorFuzz26(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - if ((++ifstate & 1) == 1) { - outputStream.send(520263); - try { - outputStream.send(306397); - int input = waitNext(inputStream); - outputStream.send(input + 943232); - outputStream.send(366272); - if ((++ifstate & 1) == 0) { - outputStream.send(700651); - state std::vector list649823; - list649823.push_back(1); - list649823.push_back(2); - list649823.push_back(3); - for (auto i : list649823) { - (void)i; - outputStream.send(146918); - return 191890; - } - outputStream.send(987155); - } - outputStream.send(499733); - } catch (...) { - outputStream.send(936386); - try { - outputStream.send(259652); - int input = waitNext(inputStream); - outputStream.send(input + 247889); - outputStream.send(402174); - state int i876439; - for (i876439 = 0; i876439 < 5; i876439++) { - outputStream.send(909715); - state std::vector list905706; - list905706.push_back(1); - list905706.push_back(2); - list905706.push_back(3); - for (auto i : list905706) { - (void)i; - outputStream.send(558855); - return 784546; - } - outputStream.send(260752); - } - outputStream.send(438765); - } catch (...) { - outputStream.send(873214); - int input = waitNext(inputStream); - outputStream.send(input + 980301); - outputStream.send(265293); - } - outputStream.send(133652); - } - outputStream.send(414082); - } else { - outputStream.send(398083); - if ((++ifstate & 1) == 1) { - outputStream.send(396069); - state std::vector list327297; - list327297.push_back(1); - list327297.push_back(2); - list327297.push_back(3); - for (auto i : list327297) { - (void)i; - outputStream.send(571919); - if ((++ifstate & 1) == 0) { - outputStream.send(620625); - return 270285; - } - outputStream.send(892626); - } - outputStream.send(564398); - } - outputStream.send(614487); - } - return 568400; -} - -ACTOR Future actorFuzz27(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - int input = waitNext(inputStream); - outputStream.send(input + 312322); - return 196907; -} - -ACTOR Future actorFuzz28(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - state int i223839; - for (i223839 = 0; i223839 < 5; i223839++) { - outputStream.send(715827); - state std::vector list555985; - list555985.push_back(1); - list555985.push_back(2); - list555985.push_back(3); - for (auto i : list555985) { - (void)i; - outputStream.send(529509); - break; - } - outputStream.send(449273); - } - return 743922; -} - -ACTOR Future actorFuzz29(FutureStream inputStream, PromiseStream outputStream, Future error) { - state int ifstate = 0; - state int i797447; - for (i797447 = 0; i797447 < 5; i797447++) { - outputStream.send(821092); - int input = waitNext(inputStream); - outputStream.send(input + 900028); - outputStream.send(617942); - } - return 560881; -} - -std::pair actorFuzzTests() { - int testsOK = 0; - testsOK += testFuzzActor(&actorFuzz0, "actorFuzz0", { 390229, 596271, 574865 }); - testsOK += - testFuzzActor(&actorFuzz1, - "actorFuzz1", - { 477566, 815578, 477566, 815578, 477566, 815578, 477566, 815578, 477566, 815578, 917160 }); - testsOK += testFuzzActor(&actorFuzz2, "actorFuzz2", { 476677, 930237 }); - testsOK += testFuzzActor(&actorFuzz3, "actorFuzz3", { 1000 }); - testsOK += testFuzzActor(&actorFuzz4, "actorFuzz4", { 180600, 177605, 177605, 177605, 954508, 810052 }); - testsOK += testFuzzActor(&actorFuzz5, "actorFuzz5", { 1000 }); - testsOK += testFuzzActor(&actorFuzz6, "actorFuzz6", { 320321, 266526, 762336, 463730, 320321, 266526, 762336, - 463730, 320321, 266526, 762336, 463730, 320321, 266526, - 762336, 463730, 320321, 266526, 762336, 463730, 945289 }); - testsOK += testFuzzActor( - &actorFuzz7, - "actorFuzz7", - { 406152, 478841, 609181, 634881, 253861, 592023, 240597, 253861, 593023, 240597, 253861, 594023, 240597, - 415949, 169335, 478331, 634881, 253861, 596023, 240597, 253861, 597023, 240597, 253861, 598023, 240597, - 415949, 173335, 478331, 634881, 253861, 600023, 240597, 253861, 601023, 240597, 253861, 602023, 240597, - 415949, 177335, 478331, 634881, 253861, 604023, 240597, 253861, 605023, 240597, 253861, 606023, 240597, - 415949, 181335, 478331, 634881, 253861, 608023, 240597, 253861, 609023, 240597, 253861, 610023, 240597, - 415949, 185335, 478331, 331905, 946924, 663973, 797073, 971923, 295772, 923567, 559259, 559259, 559259, - 325678, 679187, 295772, 923567, 559259, 559259, 559259, 325678, 679187, 295772, 923567, 559259, 559259, - 559259, 325678, 679187, 295772, 923567, 559259, 559259, 559259, 325678, 679187, 295772, 923567, 559259, - 559259, 559259, 325678, 679187, 534407, 814172, 949658 }); - testsOK += testFuzzActor(&actorFuzz8, "actorFuzz8", { 285937, 696473 }); - testsOK += testFuzzActor(&actorFuzz9, "actorFuzz9", { 141463, 397424 }); - testsOK += testFuzzActor(&actorFuzz10, "actorFuzz10", { 543113, 1000 }); - testsOK += testFuzzActor(&actorFuzz11, "actorFuzz11", { 1000 }); - testsOK += testFuzzActor(&actorFuzz12, "actorFuzz12", { 970588, 981887 }); - testsOK += testFuzzActor(&actorFuzz13, "actorFuzz13", { 861219 }); - testsOK += testFuzzActor(&actorFuzz14, "actorFuzz14", { 527098, 527098, 527098, 628047 }); - testsOK += testFuzzActor(&actorFuzz15, - "actorFuzz15", - { 582389, - 240216, - 732317, - 582389, - 240216, - 732317, - 582389, - 240216, - 732317, - 582389, - 240216, - 732317, - 582389, - 240216, - 732317, - 884781 }); - testsOK += testFuzzActor(&actorFuzz16, "actorFuzz16", { 943071, 492690, 908751, 198776, 537939 }); - testsOK += testFuzzActor(&actorFuzz17, "actorFuzz17", { 249436, 416782, 249436, 416782, 249436, 416782, 299183 }); - testsOK += testFuzzActor(&actorFuzz18, "actorFuzz18", { 337649, 395297, 807261, 517901 }); - testsOK += testFuzzActor(&actorFuzz19, - "actorFuzz19", - { 492598, 139186, 742053, 492598, 140186, 742053, 492598, 141186, 742053, 592919 }); - testsOK += testFuzzActor(&actorFuzz20, "actorFuzz20", { 760082, 1000 }); - testsOK += testFuzzActor(&actorFuzz21, "actorFuzz21", { 806394 }); - testsOK += testFuzzActor(&actorFuzz22, "actorFuzz22", { 722878, 369302, 416748 }); - testsOK += testFuzzActor(&actorFuzz23, "actorFuzz23", { 562792, 231437 }); - testsOK += testFuzzActor(&actorFuzz24, "actorFuzz24", { 847672, 835175 }); - testsOK += testFuzzActor(&actorFuzz25, "actorFuzz25", { 843261, 327560, 592398 }); - testsOK += testFuzzActor(&actorFuzz26, "actorFuzz26", { 520263, 306397, 944232, 366272, 700651, 146918, 191890 }); - testsOK += testFuzzActor(&actorFuzz27, "actorFuzz27", { 313322, 196907 }); - testsOK += testFuzzActor(&actorFuzz28, - "actorFuzz28", - { 715827, - 529509, - 449273, - 715827, - 529509, - 449273, - 715827, - 529509, - 449273, - 715827, - 529509, - 449273, - 715827, - 529509, - 449273, - 743922 }); - testsOK += testFuzzActor(&actorFuzz29, - "actorFuzz29", - { 821092, - 901028, - 617942, - 821092, - 902028, - 617942, - 821092, - 903028, - 617942, - 821092, - 904028, - 617942, - 821092, - 905028, - 617942, - 560881 }); - return std::make_pair(testsOK, 30); -} -#endif // WIN32 diff --git a/fdbrpc/ActorFuzz.h b/fdbrpc/ActorFuzz.h deleted file mode 100644 index 3d5216cfda8..00000000000 --- a/fdbrpc/ActorFuzz.h +++ /dev/null @@ -1,40 +0,0 @@ -/* - * ActorFuzz.h - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -#include "flow/flow.h" -#include - -inline void throw_operation_failed() { - throw operation_failed(); -} -#ifdef OPEN_FOR_IDE -bool testFuzzActor(Future (*actor)(FutureStream, PromiseStream, Future), - const char* desc, - std::vector const& expectedOutput); -#else -// This is in dsltest.actor.cpp: -bool testFuzzActor(Future (*actor)(FutureStream const&, PromiseStream const&, Future const&), - const char* desc, - std::vector const& expectedOutput); -#endif - -// This is defined by ActorFuzz.actor.cpp (generated by actorFuzz.py) -// Returns (tests passed, tests total) -std::pair actorFuzzTests(); diff --git a/fdbrpc/ActorFuzzUnitTest.cpp b/fdbrpc/ActorFuzzUnitTest.cpp deleted file mode 100644 index 9cd469cbb6c..00000000000 --- a/fdbrpc/ActorFuzzUnitTest.cpp +++ /dev/null @@ -1,31 +0,0 @@ -/* - * ActorFuzzUnitTest.cpp - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -#include "ActorFuzz.h" -#include "flow/UnitTest.h" - -// Only used to link unit tests -void forceLinkActorFuzzUnitTests() {} - -TEST_CASE("/actorFuzz") { - std::pair result = actorFuzzTests(); - ASSERT(result.first == result.second); - return Void(); -} diff --git a/fdbrpc/AsyncFileKAIO.h b/fdbrpc/AsyncFileKAIO.h index 3a8cc2c1794..67004e5c952 100644 --- a/fdbrpc/AsyncFileKAIO.h +++ b/fdbrpc/AsyncFileKAIO.h @@ -28,6 +28,7 @@ #include #include #include +#include #include "linux_kaio.h" #include "flow/Knobs.h" #include "fdbrpc/Stats.h" @@ -258,7 +259,7 @@ class AsyncFileKAIO final : public IAsyncFile, public ReferenceCounted result = io->result.getFuture(); + Future result = io->writeResult.getFuture(); #if KAIO_LOGGING // result = map(result, [=](int r) mutable { KAIOLogBlockEvent(io, OpLogEntry::READY, r); return r; }); @@ -267,9 +268,8 @@ class AsyncFileKAIO final : public IAsyncFile, public ReferenceCountedgetActorLineageSet(); // auto index = actorLineageSet.insert(*currentLineage); // ASSERT(index != ActorLineageSet::npos); - Future res = success(result); // actorLineageSet.erase(index); - return res; + return result; } // TODO(alexmiller): Remove when we upgrade the dev docker image to >14.10 #ifndef FALLOC_FL_ZERO_RANGE @@ -504,6 +504,7 @@ class AsyncFileKAIO final : public IAsyncFile, public ReferenceCounted { Promise result; + Promise writeResult; Reference owner; int64_t prio; IOBlock* prev; @@ -517,7 +518,10 @@ class AsyncFileKAIO final : public IAsyncFile, public ReferenceCountedprio < b->prio; } }; - IOBlock(int op, int fd) : prev(nullptr), next(nullptr), startTime(0) { + IOBlock(int op, int fd) + : result(op == IO_CMD_PWRITE ? Promise(nullptr) : Promise()), + writeResult(op == IO_CMD_PWRITE ? Promise() : Promise(nullptr)), prev(nullptr), next(nullptr), + startTime(0) { memset((linux_iocb*)this, 0, sizeof(linux_iocb)); aio_lio_opcode = op; aio_fildes = fd; @@ -528,14 +532,23 @@ class AsyncFileKAIO final : public IAsyncFile, public ReferenceCounted((prio >> 32) + 1); } - static Future deliver(Uncancellable, Promise result, bool failed, int r, TaskPriority task) { - co_await delay(0, task); - if (failed) - result.sendError(io_timeout()); - else if (r < 0) - result.sendError(io_error()); - else - result.send(r); + template + static coro::DetachedCoroutine deliver(Promise result, bool failed, int r, TaskPriority task) { + try { + co_await delay(0, task); + if (failed) + result.sendError(io_timeout()); + else if (r < 0) + result.sendError(io_error()); + else if constexpr (std::is_same_v) + result.send(Void()); + else + result.send(r); + } catch (const Error&) { + // Typed completion errors have no result consumer. + } catch (...) { + (void)unknown_error(); + } } void setResult(int r) { @@ -554,7 +567,11 @@ class AsyncFileKAIO final : public IAsyncFile, public ReferenceCountedfilename); } - deliver(Uncancellable(), result, owner->failed, r, getTask()); + if (aio_lio_opcode == IO_CMD_PWRITE) { + deliver(std::move(writeResult), owner->failed, r, getTask()); + } else { + deliver(std::move(result), owner->failed, r, getTask()); + } delete this; } diff --git a/fdbrpc/CMakeLists.txt b/fdbrpc/CMakeLists.txt index cf41f5ff9ee..96d8c1643fb 100644 --- a/fdbrpc/CMakeLists.txt +++ b/fdbrpc/CMakeLists.txt @@ -15,17 +15,8 @@ if(NOT WIN32) endif() endif() -set(FDBRPC_SRCS_DISABLE_ACTOR_DIAGNOSTICS - ActorFuzz.actor.cpp - FlowTests.actor.cpp - dsltest.actor.cpp) - -add_flow_target(STATIC_LIBRARY NAME fdbrpc - SRCS ${FDBRPC_SRCS} - DISABLE_ACTOR_DIAGNOSTICS ${FDBRPC_SRCS_DISABLE_ACTOR_DIAGNOSTICS}) -add_flow_target(STATIC_LIBRARY NAME fdbrpc_sampling - SRCS ${FDBRPC_SRCS} - DISABLE_ACTOR_DIAGNOSTICS ${FDBRPC_SRCS_DISABLE_ACTOR_DIAGNOSTICS}) +add_flow_target(STATIC_LIBRARY NAME fdbrpc SRCS ${FDBRPC_SRCS}) +add_flow_target(STATIC_LIBRARY NAME fdbrpc_sampling SRCS ${FDBRPC_SRCS}) add_flow_target(LINK_TEST NAME fdbrpclinktest SRCS LinkTest.cpp) target_link_libraries(fdbrpclinktest PRIVATE "$" rapidjson) @@ -109,9 +100,6 @@ if(COMPILE_EIO) endif() target_compile_definitions(fdbrpc_sampling PRIVATE -DENABLE_SAMPLING) -if(WIN32) - add_dependencies(fdbrpc_sampling_actors fdbrpc_actors) -endif() if(NOT FOUNDATIONDB_CROSS_COMPILING) # FIXME(swift): make this work when # x-compiling. diff --git a/fdbrpc/DDSketchTest.cpp b/fdbrpc/DDSketchTest.cpp index 3f1a4ee7fc4..7743d41bd93 100644 --- a/fdbrpc/DDSketchTest.cpp +++ b/fdbrpc/DDSketchTest.cpp @@ -59,3 +59,38 @@ TEST_CASE("/fdbrpc/ddsketch/correctness") { ASSERT(p999 > 0 && p999 != std::numeric_limits::infinity()); return Void{}; } + +TEST_CASE("/fdbrpc/ddsketch/samplePair") { + DDSketch expectedFirst; + DDSketch expectedSecond; + DDSketch actualFirst; + DDSketch actualSecond; + + // Different histories must remain independent when adding shared samples. + expectedFirst.addSample(16.0); + actualFirst.addSample(16.0); + + constexpr double eps = DDSketch::EPS; + for (double sample : { 2.0, 1.0, 0.0, eps / 2, eps, eps * 2, 4.0, 2.0 }) { + expectedFirst.addSample(sample); + expectedSecond.addSample(sample); + actualFirst.addSamplePair(sample, actualSecond); + } + + ASSERT(expectedFirst.getSamples() == actualFirst.getSamples()); + ASSERT(expectedSecond.getSamples() == actualSecond.getSamples()); + ASSERT_EQ(expectedFirst.getPopulationSize(), actualFirst.getPopulationSize()); + ASSERT_EQ(expectedSecond.getPopulationSize(), actualSecond.getPopulationSize()); + ASSERT_EQ(expectedFirst.getSum(), actualFirst.getSum()); + ASSERT_EQ(expectedSecond.getSum(), actualSecond.getSum()); + ASSERT_EQ(expectedFirst.min(), actualFirst.min()); + ASSERT_EQ(expectedSecond.min(), actualSecond.min()); + ASSERT_EQ(expectedFirst.max(), actualFirst.max()); + ASSERT_EQ(expectedSecond.max(), actualSecond.max()); + // getSamples() omits the separate zero population. + for (double percentile : { 0.0, 0.5 }) { + ASSERT_EQ(expectedFirst.percentile(percentile), actualFirst.percentile(percentile)); + ASSERT_EQ(expectedSecond.percentile(percentile), actualSecond.percentile(percentile)); + } + return Void(); +} diff --git a/fdbrpc/FileTransfer.h b/fdbrpc/FileTransfer.h index deaa66601a2..ba501fc20e7 100644 --- a/fdbrpc/FileTransfer.h +++ b/fdbrpc/FileTransfer.h @@ -22,7 +22,6 @@ #define FDBRPC_FILE_TRANSFER_H #include -#undef loop #include #include #include diff --git a/fdbrpc/FlowGrpcTests.h b/fdbrpc/FlowGrpcTests.h index 9598141f613..4e5bcfbd701 100644 --- a/fdbrpc/FlowGrpcTests.h +++ b/fdbrpc/FlowGrpcTests.h @@ -24,9 +24,9 @@ #define FDBRPC_FLOW_GRPC_TESTS_H #include +#include #include -#undef loop #include "fdbrpc/test/echo.grpc.pb.h" #include "flow/Error.h" diff --git a/fdbrpc/FlowTests.actor.cpp b/fdbrpc/FlowTests.cpp similarity index 72% rename from fdbrpc/FlowTests.actor.cpp rename to fdbrpc/FlowTests.cpp index f53c861996a..796dc773e49 100644 --- a/fdbrpc/FlowTests.actor.cpp +++ b/fdbrpc/FlowTests.cpp @@ -1,5 +1,5 @@ /* - * FlowTests.actor.cpp + * FlowTests.cpp * * This source file is part of the FoundationDB open source project * @@ -18,13 +18,17 @@ * limitations under the License. */ -// Unit tests for the flow language and libraries +// Unit tests for the Flow runtime and libraries #include +#include #include +#include #include "flow/Arena.h" #include "flow/Error.h" +#include "flow/ProcessEvents.h" #include "flow/ProtocolVersion.h" +#include "flow/Trace.h" #include "flow/UnitTest.h" #include "flow/DeterministicRandom.h" #include "flow/IThreadPool.h" @@ -34,48 +38,31 @@ #include "flow/IAsyncFile.h" #include "flow/TLSConfig.h" #include "fdbrpc/grpc/AsyncTaskExecutor.h" -#include "flow/actorcompiler.h" // This must be the last #include. +#include "flow/CoroUtils.h" void forceLinkFlowTests() {} -constexpr int firstLine = __LINE__; -TEST_CASE("/flow/actorcompiler/lineNumbers") { - loop { - try { - ASSERT(__LINE__ == firstLine + 4); - wait(Future(Void())); - ASSERT(__LINE__ == firstLine + 6); - throw success(); - } catch (Error& e) { - ASSERT(__LINE__ == firstLine + 9); - wait(Future(Void())); - ASSERT(__LINE__ == firstLine + 11); - } - break; - } - ASSERT(__FILE__sr.endsWith("FlowTests.actor.cpp"_sr)); - return Void(); -} +extern bool g_crashOnError; TEST_CASE("/flow/buggifiedDelay") { if (FLOW_KNOBS->MAX_BUGGIFIED_DELAY == 0) { - return Void(); + co_return; } - loop { - state double x = deterministicRandom()->random01(); - state int last = 0; - state Future f1 = map(delay(x), [last = &last](const Void&) { + while (true) { + double x = deterministicRandom()->random01(); + int last = 0; + Future f1 = map(delay(x), [last = &last](const Void&) { *last = 1; return Void(); }); - state Future f2 = map(delay(x), [last = &last](const Void&) { + Future f2 = map(delay(x), [last = &last](const Void&) { *last = 2; return Void(); }); - wait(f1 && f2); + co_await (f1 && f2); if (last == 1) { CODE_PROBE(true, "Delays can become ready out of order", probe::decoration::rare); - return Void(); + co_return; } } } @@ -112,8 +99,10 @@ void onReady(Future&& f, Func&& func, ErrFunc&& errFunc) { errFunc(f.getError()); else func(f.get()); - } else - f.addCallbackAndClear(new LambdaCallback>(std::move(func), std::move(errFunc))); + } else { + f.addCallbackAndClear(new LambdaCallback>(std::forward(func), + std::forward(errFunc))); + } } template @@ -123,103 +112,82 @@ void onReady(FutureStream&& f, Func&& func, ErrFunc&& errFunc) { errFunc(f.getError()); else func(f.pop()); - } else - f.addCallbackAndClear( - new LambdaCallback>(std::move(func), std::move(errFunc))); + } else { + f.addCallbackAndClear(new LambdaCallback>(std::forward(func), + std::forward(errFunc))); + } } -ACTOR static void emptyVoidActor() {} +static Future emptyVoidActor(Uncancellable = Uncancellable()) { + co_return; +} -ACTOR [[flow_allow_discard]] static Future emptyActor() { +static Future emptyActor() { return Void(); } -ACTOR static void oneWaitVoidActor(Future f) { - wait(f); +static Future oneWaitVoidActor(Future f, Uncancellable = Uncancellable()) { + co_await f; } -ACTOR static Future oneWaitActor(Future f) { - wait(f); - return Void(); +static Future oneWaitActor(Future f) { + co_await f; } Future g_cheese; -ACTOR static Future cheeseWaitActor() { - wait(g_cheese); - return Void(); -} - -size_t cheeseWaitActorSize() { -#ifndef OPEN_FOR_IDE - return sizeof(CheeseWaitActorActor); -#else - return 0ul; -#endif +static Future cheeseWaitActor() { + // The global can change while this coroutine is suspended. + Future f = g_cheese; + co_await f; } -ACTOR static void trivialVoidActor(int* result) { +static Future trivialVoidActor(int* result, Uncancellable = Uncancellable()) { *result = 1; + co_return; } -ACTOR static Future return42Actor() { +static Future return42Actor() { return 42; } -ACTOR static void voidWaitActor(Future in, int* result) { - int i = wait(in); +static Future voidWaitActor(Future in, int* result, Uncancellable = Uncancellable()) { + int i = co_await in; *result = i; } -ACTOR static Future addOneActor(Future in) { - int i = wait(in); - return i + 1; +static Future addOneActor(Future in) { + int i = co_await in; + co_return i + 1; } -ACTOR static Future chooseTwoActor(Future f, Future g) { - choose { - when(wait(f)) {} - when(wait(g)) {} - } - return Void(); +static Future chooseTwoActor(Future f, Future g) { + co_await race(f, g); } -ACTOR static Future consumeOneActor(FutureStream in) { - int i = waitNext(in); - return i; +static Future consumeOneActor(FutureStream in) { + int i = co_await in; + co_return i; } -ACTOR static Future sumActor(FutureStream in) { - state int total = 0; +static Future sumActor(FutureStream in) { + int total = 0; try { - loop { - int i = waitNext(in); + while (true) { + int i = co_await in; total += i; } } catch (Error& e) { if (e.code() != error_code_end_of_stream) throw; } - return total; + co_return total; } -ACTOR template +template static Future templateActor(T t) { return t; } -static int destroy() { - return 666; -} -ACTOR static Future testHygeine() { - ASSERT(destroy() == 666); // Should fail to compile if SAV::destroy() is visible - return Void(); -} - -// bool expectActorCount(int x) { return actorCount == x; } -bool expectActorCount(int) { - return true; -} - struct YieldMockNetwork final : INetwork, ReferenceCounted { int ticks; Promise nextTick; @@ -307,24 +275,24 @@ struct YieldMockNetwork final : INetwork, ReferenceCounted { }; struct NonserializableThing {}; -ACTOR static Future testNonserializableThing() { +static Future testNonserializableThing() { return NonserializableThing(); } -ACTOR Future testCancelled(bool* exits, Future f) { +Future testCancelled(bool* exits, Future f) { + Error err; try { - wait(Future(Never())); + co_await Future(Never()); } catch (Error& e) { - state Error err = e; - try { - wait(Future(Never())); - } catch (Error& e) { - *exits = true; - throw; - } - throw err; + err = e; } - return Void(); + try { + co_await Future(Never()); + } catch (Error& e) { + *exits = true; + throw; + } + throw err; } TEST_CASE("/flow/flow/cancel1") { @@ -336,6 +304,9 @@ TEST_CASE("/flow/flow/cancel1") { ASSERT(exits); ASSERT(test.getPromiseReferenceCount() == 0 && test.getFutureReferenceCount() == 1 && test.isReady() && test.isError() && test.getError().code() == error_code_actor_cancelled); + // Coroutine parameters remain alive until the last future releases the frame. + ASSERT(p.getPromiseReferenceCount() == 1 && p.getFutureReferenceCount() == 1); + test = Future(); ASSERT(p.getPromiseReferenceCount() == 1 && p.getFutureReferenceCount() == 0); return Void(); @@ -352,10 +323,10 @@ TEST_CASE("/fdbrpc/asyncFileNonDurable/sendErrorOnShutdownCancellation") { return Void(); } -ACTOR static Future noteCancel(int* cancelled) { +static Future noteCancel(int* cancelled) { *cancelled = 0; try { - wait(Future(Never())); + co_await Future(Never()); throw internal_error(); } catch (...) { printf("Cancelled!\n"); @@ -391,6 +362,112 @@ struct Int { serializer(ar, value); } }; + +template +SAV* replyPromiseState(ReplyPromise& promise) { + auto* rawState = promise.extractRawPointer(); + promise = ReplyPromise(rawState); + return rawState; +} + +struct ReplyPromiseReuseRequest { + constexpr static FileIdentifier file_identifier = 1449982; + ReplyPromise reply; + SAV* defaultState; + + ReplyPromiseReuseRequest() : defaultState(replyPromiseState(reply)) {} + + template + void serialize(Ar& ar) { + serializer(ar, reply); + } +}; + +class RpcExceptionObserver : NonCopyable { + const bool previousTraceProcessEvents; + const char* expectedException; + int observed = 0; + int unexpected = 0; + ProcessEvents::Event event; + + void observe(const std::any& data) { + auto tracePtr = std::any_cast(&data); + if (!tracePtr || !*tracePtr) { + ++unexpected; + return; + } + auto* trace = *tracePtr; + int errorCode = 0; + std::string exception; + if (!expectedException || observed != 0 || trace->getSeverity() != SevError || + !trace->getFields().tryGetInt("ErrorCode", errorCode) || errorCode != error_code_unknown_error || + !trace->getFields().tryGetValue("StdException", exception) || exception != expectedException) { + ++unexpected; + return; + } + ++observed; + // Preserve the real severity check, but identify only this deliberately injected exception to trace scanners. + trace->detail("ErrorIsInjectedFault", 1); + } + +public: + explicit RpcExceptionObserver(const char* expectedException = nullptr) + : previousTraceProcessEvents(g_traceProcessEvents), expectedException(expectedException), + event("TraceEvent::SystemError"_sr, [this](StringRef, const std::any& data, const Error&) { observe(data); }) { + g_traceProcessEvents = true; + } + ~RpcExceptionObserver() { g_traceProcessEvents = previousTraceProcessEvents; } + + void check(int expectedCount) const { + ASSERT_EQ(observed, expectedCount); + ASSERT_EQ(unexpected, 0); + } +}; + +template +struct ThrowingRpcReply { + constexpr static FileIdentifier file_identifier = 1449983 + ErrorReply; + uint32_t value = 0; + + static const char* exceptionMessage() { + return ErrorReply ? "RpcUnexpectedException/networkSenderErrorReply" + : "RpcUnexpectedException/networkSenderValue"; + } + + template + void serialize(Ar& ar) { + serializer(ar, value); + if constexpr (is_fb_function) { + // Vtable collection visits the reply alternative even when ErrorOr contains an error. + throw std::runtime_error(exceptionMessage()); + } + } +}; + +Future checkNetworkSenderKnownErrors() { + RpcExceptionObserver observer; + ReplyPromise recipient; + Future received = recipient.getFuture(); + Endpoint endpoint(FlowTransport::transport().getLocalAddresses(), recipient.getEndpoint().token); + ReplyPromise sender; + sender.loadRemoteEndpoint(endpoint); + sender.sendError(operation_failed()); + ErrorOr result = co_await errorOr(timeoutError(received, 1.0)); + ASSERT(result.isError() && result.getError().code() == error_code_operation_failed); + ASSERT_EQ(sender.getFutureReferenceCount(), 0); + + ReplyPromise noReplyRecipient; + Future noReplyReceived = noReplyRecipient.getFuture(); + Endpoint noReplyEndpoint(FlowTransport::transport().getLocalAddresses(), noReplyRecipient.getEndpoint().token); + ReplyPromise noReplySender; + noReplySender.loadRemoteEndpoint(noReplyEndpoint); + noReplySender.send(Never()); + co_await orderedDelay(0, TaskPriority::DefaultPromiseEndpoint); + ASSERT(!noReplyReceived.isReady()); + ASSERT_EQ(noReplySender.getFutureReferenceCount(), 0); + observer.check(0); +} + } // namespace flow_tests_details TEST_CASE("/flow/flow/nonserializable futures") { @@ -462,10 +539,138 @@ TEST_CASE("/flow/flow/networked futures") { return Void(); } +TEST_CASE("/fdbrpc/ReplyPromise/reuse state on deserialize") { + using flow_tests_details::Int; + const Endpoint remote({ NetworkAddress(IPAddress(0x01010101), 1) }, UID(1, 2)); + + { + ReplyPromise promise; + auto* original = flow_tests_details::replyPromiseState(promise); + promise.loadRemoteEndpoint(remote); + ASSERT(flow_tests_details::replyPromiseState(promise) == original); + ASSERT(promise.getEndpoint() == remote); + Future reply = promise.getFuture(); + promise.send(Int(17)); + ASSERT(reply.isReady() && !reply.isError() && reply.get().value == 17); + } + + { + ReplyPromise promise; + Future oldReply = promise.getFuture(); + auto* original = flow_tests_details::replyPromiseState(promise); + promise.loadRemoteEndpoint(remote); + ASSERT(flow_tests_details::replyPromiseState(promise) != original); + ASSERT(oldReply.isReady() && oldReply.isError() && oldReply.getError().code() == error_code_broken_promise); + Future reply = promise.getFuture(); + promise.send(Int(23)); + ASSERT(reply.isReady() && !reply.isError() && reply.get().value == 23); + } + + { + ReplyPromise promise; + Endpoint local = promise.getEndpoint(); + ASSERT(local.isValid()); + auto* original = flow_tests_details::replyPromiseState(promise); + promise.loadRemoteEndpoint(remote); + ASSERT(flow_tests_details::replyPromiseState(promise) != original); + ASSERT(promise.getEndpoint() == remote); + Future reply = promise.getFuture(); + promise.send(Int(31)); + ASSERT(reply.isReady() && !reply.isError() && reply.get().value == 31); + } + + { + ReplyPromise promise( + PeerCompatibilityPolicy{ RequirePeer::AtLeast, ProtocolVersion::withStableInterfaces() }); + auto* original = flow_tests_details::replyPromiseState(promise); + promise.loadRemoteEndpoint(remote); + ASSERT(flow_tests_details::replyPromiseState(promise) != original); + Future reply = promise.getFuture(); + promise.send(Int(41)); + ASSERT(reply.isReady() && !reply.isError() && reply.get().value == 41); + } + + return Void(); +} + +TEST_CASE("/fdbrpc/ReplyPromise/reuse state binary deserialize") { + using flow_tests_details::ReplyPromiseReuseRequest; + ReplyPromiseReuseRequest request; + ProtocolVersion version = currentProtocolVersion(); + version.removeObjectSerializerFlag(); + BinaryWriter writer(IncludeVersion(version)); + writer << request; + + BinaryReader reader(writer.toValue(), IncludeVersion(version)); + ReplyPromiseReuseRequest received; + reader >> received; + ASSERT(flow_tests_details::replyPromiseState(received.reply) == received.defaultState); + ASSERT(received.reply.getEndpoint().token == request.reply.getEndpoint().token); + return Void(); +} + +TEST_CASE("/fdbrpc/ReplyPromise/reuse state exact reply") { + RequestStream local; + FutureStream incoming = local.getFuture(); + flow_tests_details::ReplyPromiseReuseRequest request; + Future reply = request.reply.getFuture(); + { + RequestStream remote(local.getEndpoint()); + remote.send(request); + } + + flow_tests_details::ReplyPromiseReuseRequest received = co_await incoming; + ASSERT(flow_tests_details::replyPromiseState(received.reply) == received.defaultState); + received.reply.send(flow_tests_details::Int(59)); + flow_tests_details::Int value = co_await reply; + ASSERT(value.value == 59); +} + +TEST_CASE("noSim/fdbrpc/RpcUnexpectedException/networkSenderValue") { + // Simulation counts SevError before the observer can mark an expected injection; crash-on-error must stay intact. + if (g_network->isSimulated() || g_crashOnError) { + return Void(); + } + using Reply = flow_tests_details::ThrowingRpcReply; + flow_tests_details::RpcExceptionObserver observer(Reply::exceptionMessage()); + Endpoint endpoint(FlowTransport::transport().getLocalAddresses(), UID(1, 2)); + ReplyPromise reply; + reply.loadRemoteEndpoint(endpoint); + ASSERT_GT(reply.getFutureReferenceCount(), 0); + reply.send(Reply()); + observer.check(1); + ASSERT_EQ(reply.getFutureReferenceCount(), 0); + return Void(); +} + +TEST_CASE("noSim/fdbrpc/RpcUnexpectedException/networkSenderErrorReply") { + if (g_network->isSimulated() || g_crashOnError) { + return Void(); + } + using Reply = flow_tests_details::ThrowingRpcReply; + flow_tests_details::RpcExceptionObserver observer(Reply::exceptionMessage()); + Endpoint endpoint(FlowTransport::transport().getLocalAddresses(), UID(1, 2)); + ReplyPromise reply; + reply.loadRemoteEndpoint(endpoint); + ASSERT_GT(reply.getFutureReferenceCount(), 0); + reply.sendError(operation_failed()); + observer.check(1); + ASSERT_EQ(reply.getFutureReferenceCount(), 0); + return Void(); +} + +TEST_CASE("noSim/fdbrpc/RpcUnexpectedException/networkSenderKnownErrors") { + if (g_network->isSimulated() || g_crashOnError) { + co_return; + } + co_await flow_tests_details::checkNetworkSenderKnownErrors(); +} + TEST_CASE("/flow/flow/quorum") { std::vector> ps(5); std::vector> fs; std::vector> qs; + fs.reserve(ps.size()); for (auto& p : ps) fs.push_back(p.getFuture()); @@ -631,8 +836,8 @@ TEST_CASE("/flow/flow/promisestream callbacks") { /* TEST_CASE("/flow/flow/promisestream multiple wait error") { - state int result = 0; - state PromiseStream p; + int result = 0; + PromiseStream p; try { onReady(p.getFuture(), [&result](int x) { result = x; }, [&result](Error e){ result = -1; }); result = 100; @@ -650,43 +855,34 @@ TEST_CASE("/flow/flow/promisestream multiple wait error") */ TEST_CASE("/flow/flow/trivial actors") { - ASSERT(expectActorCount(0)); - int result = 0; trivialVoidActor(&result); ASSERT(result == 1); - ASSERT(expectActorCount(0)); Future f = return42Actor(); ASSERT(f.isReady() && !f.isError() && f.get() == 42 && f.getFutureReferenceCount() == 1 && f.getPromiseReferenceCount() == 0); - ASSERT(expectActorCount(1)); f = Future(); - ASSERT(expectActorCount(0)); f = templateActor(24); ASSERT(f.isReady() && !f.isError() && f.get() == 24 && f.getFutureReferenceCount() == 1 && f.getPromiseReferenceCount() == 0); - ASSERT(expectActorCount(1)); f = Future(); - ASSERT(expectActorCount(0)); result = 0; voidWaitActor(2, &result); - ASSERT(result == 2 && expectActorCount(0)); + ASSERT(result == 2); Promise p; f = addOneActor(p.getFuture()); - ASSERT(!f.isReady() && expectActorCount(1)); + ASSERT(!f.isReady()); p.send(100); ASSERT(f.isReady() && f.get() == 101); - ASSERT(expectActorCount(1)); //< hmm f = Future(); - ASSERT(expectActorCount(0)); PromiseStream ps; f = consumeOneActor(ps.getFuture()); - ASSERT(!f.isReady() && expectActorCount(1)); + ASSERT(!f.isReady()); ps.send(101); ASSERT(f.get() == 101 && ps.isEmpty()); ps.send(102); @@ -701,7 +897,6 @@ TEST_CASE("/flow/flow/trivial actors") { ps.sendError(end_of_stream()); ASSERT(f.get() == 111); - ASSERT(testHygeine().isReady()); return Void(); } @@ -718,6 +913,7 @@ TEST_CASE("/flow/flow/yieldedFuture/progress") { Future i = success(u); std::vector> v; + v.reserve(5); for (int i = 0; i < 5; i++) v.push_back(yieldedFuture(u)); auto numReady = [&v]() { return std::count_if(v.begin(), v.end(), [](Future v) { return v.isReady(); }); }; @@ -749,6 +945,7 @@ TEST_CASE("/flow/flow/yieldedFuture/random") { Future i = success(u); std::vector> v; + v.reserve(25); for (int i = 0; i < 25; i++) v.push_back(yieldedFuture(u)); auto numReady = [&v]() { @@ -797,6 +994,7 @@ TEST_CASE("/flow/perf/yieldedFuture") { std::vector> ys; start = timer(); + ys.reserve(N); for (int i = 0; i < N; i++) ys.push_back(yieldedFuture(f)); printf("yieldedFuture(f) create: %0.1f M/sec\n", N / 1e6 / (timer() - start)); @@ -818,16 +1016,14 @@ TEST_CASE("/flow/perf/yieldedFuture") { } TEST_CASE("/flow/flow/chooseTwoActor") { - ASSERT(expectActorCount(0)); - Promise a, b; Future c = chooseTwoActor(a.getFuture(), b.getFuture()); - ASSERT(a.getFutureReferenceCount() == 2 && b.getFutureReferenceCount() == 2 && !c.isReady()); + // Parameters, race inputs, and callbacks each retain a reference while suspended. + ASSERT(a.getFutureReferenceCount() == 3 && b.getFutureReferenceCount() == 3 && !c.isReady()); b.send(Void()); - ASSERT(a.getFutureReferenceCount() == 0 && b.getFutureReferenceCount() == 0 && c.isReady() && !c.isError() && - expectActorCount(1)); + ASSERT(a.getFutureReferenceCount() == 1 && b.getFutureReferenceCount() == 1 && c.isReady() && !c.isError()); c = Future(); - ASSERT(a.getFutureReferenceCount() == 0 && b.getFutureReferenceCount() == 0 && expectActorCount(0)); + ASSERT(a.getFutureReferenceCount() == 0 && b.getFutureReferenceCount() == 0); return Void(); } @@ -835,23 +1031,17 @@ TEST_CASE("#flow/flow/perf/actor patterns") { double start; int N = 1000000; - ASSERT(expectActorCount(0)); - start = timer(); for (int i = 0; i < N; i++) emptyVoidActor(); printf("emptyVoidActor(): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - ASSERT(expectActorCount(0)); - start = timer(); for (int i = 0; i < N; i++) { emptyActor(); } printf("emptyActor(): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - ASSERT(expectActorCount(0)); - Promise neverSet; Future never = neverSet.getFuture(); Future already = Void(); @@ -861,8 +1051,6 @@ TEST_CASE("#flow/flow/perf/actor patterns") { oneWaitVoidActor(already); printf("oneWaitVoidActor(already): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - ASSERT(expectActorCount(0)); - /*start = timer(); for (int i = 0; i < N; i++) oneWaitVoidActor(never); @@ -884,7 +1072,6 @@ TEST_CASE("#flow/flow/perf/actor patterns") { ASSERT(!f.isReady()); } printf("(cancelled) oneWaitActor(never): %0.1f M/sec\n", N / 1e6 / (timer() - start)); - ASSERT(expectActorCount(0)); } { @@ -959,7 +1146,6 @@ TEST_CASE("#flow/flow/perf/actor patterns") { Future f = chooseTwoActor(never, never); ASSERT(!f.isReady()); } - // ASSERT(expectActorCount(0)); printf("(cancelled) chooseTwoActor(never, never): %0.1f M/sec\n", N / 1e6 / (timer() - start)); } @@ -1105,7 +1291,6 @@ TEST_CASE("#flow/flow/perf/actor patterns") { ASSERT(out2[i].isReady()); } printf("2xcheeseActor(chooseTwoActor(cheeseActor(fifo), never)): %0.2f M/sec\n", N / 1e6 / (timer() - start)); - printf("sizeof(CheeseWaitActorActor) == %zu\n", cheeseWaitActorSize()); } { @@ -1161,7 +1346,7 @@ struct YAMRandom { } else if (op == 1) { onchanges.push_back(trigger([this]() { this->randomOp(); }, yam.onChange(k))); } else if (op == 2) { - if (onchanges.size()) { + if (!onchanges.empty()) { int i = deterministicRandom()->randomInt(0, onchanges.size()); onchanges[i] = onchanges.back(); onchanges.pop_back(); @@ -1182,57 +1367,51 @@ struct YAMRandom { }; TEST_CASE("/flow/flow/YieldedAsyncMap/randomized") { - state YAMRandom> yamr; - state int it; - for (it = 0; it < 100000; it++) { + YAMRandom> yamr; + for (int it = 0; it < 100000; it++) { yamr.randomOp(); - wait(yield()); + co_await yield(); } - return Void(); } TEST_CASE("/flow/flow/AsyncMap/randomized") { - state YAMRandom> yamr; - state int it; - for (it = 0; it < 100000; it++) { + YAMRandom> yamr; + for (int it = 0; it < 100000; it++) { yamr.randomOp(); - wait(yield()); + co_await yield(); } - return Void(); } TEST_CASE("/flow/flow/YieldedAsyncMap/basic") { - state YieldedAsyncMap yam; - state Future y0 = yam.onChange(1); + YieldedAsyncMap yam; + Future y0 = yam.onChange(1); yam.setUnconditional(1, 0); - state Future y1 = yam.onChange(1); - state Future y1a = yam.onChange(1); - state Future y1b = yam.onChange(1); + Future y1 = yam.onChange(1); + Future y1a = yam.onChange(1); + Future y1b = yam.onChange(1); yam.set(1, 1); // while (!check_yield()) {} // yam.triggerRange(0, 4); - state Future y2 = yam.onChange(1); - wait(reportErrors(y0, "Y0")); - wait(reportErrors(y1, "Y1")); - wait(reportErrors(y1a, "Y1a")); - wait(reportErrors(y1b, "Y1b")); - wait(reportErrors(timeout(y2, 5, Void()), "Y2")); - - return Void(); + Future y2 = yam.onChange(1); + co_await reportErrors(y0, "Y0"); + co_await reportErrors(y1, "Y1"); + co_await reportErrors(y1a, "Y1a"); + co_await reportErrors(y1b, "Y1b"); + co_await reportErrors(timeout(y2, 5, Void()), "Y2"); } TEST_CASE("/flow/flow/YieldedAsyncMap/cancel") { - state YieldedAsyncMap yam; + YieldedAsyncMap yam; // ASSERT(yam.count(1) == 0); - // state Future y0 = yam.onChange(1); + // Future y0 = yam.onChange(1); // ASSERT(yam.count(1) == 1); // yam.setUnconditional(1, 0); ASSERT(yam.count(1) == 0); - state Future y1 = yam.onChange(1); - state Future y1a = yam.onChange(1); - state Future y1b = yam.onChange(1); + Future y1 = yam.onChange(1); + Future y1a = yam.onChange(1); + Future y1b = yam.onChange(1); ASSERT(yam.count(1) == 1); y1.cancel(); ASSERT(!y1a.isReady()); @@ -1247,10 +1426,10 @@ TEST_CASE("/flow/flow/YieldedAsyncMap/cancel") { } TEST_CASE("/flow/flow/YieldedAsyncMap/cancel2") { - state YieldedAsyncMap yam; + YieldedAsyncMap yam; - state Future y1 = yam.onChange(1); - state Future y2 = yam.onChange(2); + Future y1 = yam.onChange(1); + Future y2 = yam.onChange(2); auto* pyam = &yam; uncancellable(trigger( @@ -1260,11 +1439,9 @@ TEST_CASE("/flow/flow/YieldedAsyncMap/cancel2") { }, delay(1))); - wait(y1); + co_await y1; printf("Got y1\n"); y2.cancel(); - - return Void(); } TEST_CASE("/flow/flow/AsyncVar/basic") { @@ -1284,15 +1461,18 @@ TEST_CASE("/flow/flow/AsyncVar/basic") { return Void(); } -ACTOR static Future waitAfterCancel(int* output) { +static Future waitAfterCancel(int* output) { *output = 0; + bool cancelled = false; try { - wait(Never()); + co_await Future(Never()); } catch (...) { - wait((*output = 1, Future(Void()))); + cancelled = true; + } + if (cancelled) { + co_await (*output = 1, Future(Void())); } ASSERT(false); - return Void(); } TEST_CASE("/fdbrpc/flow/wait_expression_after_cancel_flow") { @@ -1301,88 +1481,10 @@ TEST_CASE("/fdbrpc/flow/wait_expression_after_cancel_flow") { ASSERT(a == 0); f.cancel(); ASSERT(a == 1); + ASSERT(f.isReady() && f.isError() && f.getError().code() == error_code_actor_cancelled); return Void(); } -// Tests for https://github.com/apple/foundationdb/issues/1226 - -template -struct ShouldNotGoIntoClassContextStack; - -class Foo1 { -public: - explicit Foo1(int x) : x(x) {} - Future foo() { return fooActor(this); } - ACTOR static Future fooActor(Foo1* self); - -private: - int x; -}; -ACTOR Future Foo1::fooActor(Foo1* self) { - wait(Future()); - return self->x; -} - -class [[nodiscard]] Foo2 { -public: - explicit Foo2(int x) : x(x) {} - Future foo() { return fooActor(this); } - ACTOR static Future fooActor(Foo2* self); - -private: - int x; -}; -ACTOR Future Foo2::fooActor(Foo2* self) { - wait(Future()); - return self->x; -} - -class alignas(4) Foo3 { -public: - explicit Foo3(int x) : x(x) {} - Future foo() { return fooActor(this); } - ACTOR static Future fooActor(Foo3* self); - -private: - int x; -}; -ACTOR Future Foo3::fooActor(Foo3* self) { - wait(Future()); - return self->x; -} - -struct Super {}; - -class Foo4 : Super { -public: - explicit Foo4(int x) : x(x) {} - Future foo() { return fooActor(this); } - ACTOR static Future fooActor(Foo4* self); - -private: - int x; -}; -ACTOR Future Foo4::fooActor(Foo4* self) { - wait(Future()); - return self->x; -} - -struct Outer { - class Foo5 : Super { - public: - explicit Foo5(int x) : x(x) {} - Future foo() { return fooActor(this); } - ACTOR static Future fooActor(Foo5* self); - - private: - int x; - }; -}; -ACTOR Future Outer::Foo5::fooActor(Outer::Foo5* self) { - wait(Future()); - return self->x; -} - // Meant to be run with -fsanitize=undefined TEST_CASE("/flow/DeterministicRandom/SignedOverflow") { deterministicRandom()->randomInt(std::numeric_limits::min(), 0); @@ -1429,32 +1531,31 @@ struct Tracker { } ~Tracker() = default; - ACTOR static Future listen(FutureStream stream) { - Tracker movedTracker = waitNext(stream); + static Future listen(FutureStream stream, int expectedCopies) { + Tracker movedTracker = co_await stream; ASSERT(!movedTracker.moved); - ASSERT(movedTracker.copied == 0); - return Void(); + ASSERT(movedTracker.copied == expectedCopies); } }; TEST_CASE("/flow/flow/PromiseStream/move") { - state PromiseStream stream; - state Future listener; + PromiseStream stream; + Future listener; { // This tests the case when a callback is added before // a movable value is sent - listener = Tracker::listen(stream.getFuture()); + listener = Tracker::listen(stream.getFuture(), 0); stream.send(Tracker{}); - wait(listener); + co_await listener; } { // This tests the case when a callback is added before - // a unmovable value is sent - listener = Tracker::listen(stream.getFuture()); + // an lvalue is copied into the coroutine's awaiter storage. + listener = Tracker::listen(stream.getFuture(), 1); Tracker namedTracker; stream.send(namedTracker); - wait(listener); + co_await listener; } { // This tests the case when no callback is added until @@ -1462,12 +1563,12 @@ TEST_CASE("/flow/flow/PromiseStream/move") { stream.send(Tracker{}); stream.send(Tracker{}); { - state Tracker movedTracker = waitNext(stream.getFuture()); + Tracker movedTracker = co_await stream.getFuture(); ASSERT(!movedTracker.moved); ASSERT(movedTracker.copied == 0); } { - Tracker movedTracker = waitNext(stream.getFuture()); + Tracker movedTracker = co_await stream.getFuture(); ASSERT(!movedTracker.moved); ASSERT(movedTracker.copied == 0); } @@ -1480,48 +1581,46 @@ TEST_CASE("/flow/flow/PromiseStream/move") { stream.send(namedTracker1); stream.send(namedTracker2); { - state Tracker copiedTracker = waitNext(stream.getFuture()); + Tracker copiedTracker = co_await stream.getFuture(); ASSERT(!copiedTracker.moved); // must copy onto queue ASSERT(copiedTracker.copied == 1); } { - Tracker copiedTracker = waitNext(stream.getFuture()); + Tracker copiedTracker = co_await stream.getFuture(); ASSERT(!copiedTracker.moved); // must copy onto queue ASSERT(copiedTracker.copied == 1); } } - - return Void(); } TEST_CASE("/flow/flow/PromiseStream/move2") { PromiseStream stream; stream.send(Tracker{}); - Tracker tracker = waitNext(stream.getFuture()); + Tracker tracker = co_await stream.getFuture(); Tracker movedTracker = std::move(tracker); ASSERT( tracker.moved); // NOLINT(bugprone-use-after-move): this test intentionally checks the moved-from Tracker state. ASSERT(!movedTracker.moved); ASSERT(movedTracker.copied == 0); - return Void(); } constexpr double mutexTestDelay = 0.00001; -ACTOR Future mutexTest(int id, FlowMutex* mutex, int n, bool allowError, bool* verbose) { +Future mutexTest(int id, FlowMutex* mutex, int n, bool allowError, bool* verbose) { + FlowMutex::Lock lock; while (n-- > 0) { - state double d = deterministicRandom()->random01() * mutexTestDelay; + double d = deterministicRandom()->random01() * mutexTestDelay; if (*verbose) { printf("%d:%d wait %f while unlocked\n", id, n, d); } - wait(delay(d)); + co_await delay(d); if (*verbose) { printf("%d:%d locking\n", id, n); } - state FlowMutex::Lock lock = wait(mutex->take()); + lock = co_await mutex->take(); if (*verbose) { printf("%d:%d locked\n", id, n); } @@ -1530,7 +1629,7 @@ ACTOR Future mutexTest(int id, FlowMutex* mutex, int n, bool allowError, b if (*verbose) { printf("%d:%d wait %f while locked\n", id, n, d); } - wait(delay(d)); + co_await delay(d); // On the last iteration, send an error or drop the lock if allowError is true if (n == 0 && allowError) { @@ -1557,61 +1656,60 @@ ACTOR Future mutexTest(int id, FlowMutex* mutex, int n, bool allowError, b if (*verbose) { printf("%d Returning\n", id); } - return Void(); } TEST_CASE("/flow/flow/FlowMutex") { - state int count = 100000; + int count = 100000; // Default verboseness - state bool verboseSetting = false; + bool verboseSetting = false; // Useful for debugging, enable verbose mode for this iteration number - state int verboseTestIteration = -1; + int verboseTestIteration = -1; try { - state bool verbose = verboseSetting || count == verboseTestIteration; + bool verbose = verboseSetting || count == verboseTestIteration; while (--count > 0) { if (count % 1000 == 0) { printf("%d tests left\n", count); } - state FlowMutex mutex; - state std::vector> tests; + FlowMutex mutex; + std::vector> tests; - state bool allowErrors = deterministicRandom()->coinflip(); + bool allowErrors = deterministicRandom()->coinflip(); if (verbose) { printf("\nTesting allowErrors=%d\n", allowErrors); } - state Optional error; + Optional error; try { for (int i = 0; i < 10; ++i) { tests.push_back(mutexTest(i, &mutex, 10, allowErrors, &verbose)); } - wait(waitForAll(tests)); + co_await waitForAll(tests); if (allowErrors) { if (verbose) { printf("Final wait in case error was injected by the last actor to finish\n"); } - wait(success(mutex.take())); + co_await success(mutex.take()); } } catch (Error& e) { if (verbose) { printf("Caught error %s\n", e.what()); } error = e; - + } + if (error.present()) { // Some actors can still be running, waiting while locked or unlocked, // but all should become ready, some with errors. - state int i; if (verbose) { printf("Waiting for completions. Future end states:\n"); } - for (i = 0; i < tests.size(); ++i) { - ErrorOr f = wait(errorOr(tests[i])); + for (int i = 0; i < tests.size(); ++i) { + ErrorOr f = co_await errorOr(tests[i]); if (verbose) { printf(" %d: %s\n", i, f.isError() ? f.getError().what() : "done"); } @@ -1626,18 +1724,16 @@ TEST_CASE("/flow/flow/FlowMutex") { printf("Error at count=%d\n", count + 1); ASSERT(false); } - - return Void(); } using namespace std::chrono_literals; TEST_CASE("/flow/thread/ThreadReturnPromiseStream_Simple") { - state AsyncTaskExecutor exc(1); + AsyncTaskExecutor exc(1); noUnseed = true; ThreadReturnPromiseStream stream; - state ThreadFutureStream f = stream.getFuture(); - state Future t = exc.post([stream = std::move(stream)]() mutable { + ThreadFutureStream f = stream.getFuture(); + Future t = exc.post([stream = std::move(stream)]() mutable { for (int i = 0; i < 10; ++i) { stream.send(i); std::cout << "Sent: " << i << std::endl; @@ -1647,12 +1743,12 @@ TEST_CASE("/flow/thread/ThreadReturnPromiseStream_Simple") { return Void(); }); - state int n = 0; + int n = 0; while (true) { try { - state int i = waitNext(f); + int i = co_await f; std::cout << "Got: " << i << std::endl; - wait(delay(0.1)); + co_await delay(0.1); ASSERT(i == n); ++n; } catch (Error& e) { @@ -1662,16 +1758,15 @@ TEST_CASE("/flow/thread/ThreadReturnPromiseStream_Simple") { } } - wait(t); - return Void(); + co_await t; } TEST_CASE("/flow/thread/ThreadReturnPromiseStream_Seq") { - state AsyncTaskExecutor exc(1); + AsyncTaskExecutor exc(1); noUnseed = true; ThreadReturnPromiseStream stream; - state ThreadFutureStream f = stream.getFuture(); - state Future t = exc.post([stream = std::move(stream)]() mutable { + ThreadFutureStream f = stream.getFuture(); + Future t = exc.post([stream = std::move(stream)]() mutable { for (int i = 0; i <= 3; ++i) { stream.send(i); std::this_thread::sleep_for(100ms); @@ -1680,61 +1775,58 @@ TEST_CASE("/flow/thread/ThreadReturnPromiseStream_Seq") { return Void(); }); - int r0 = waitNext(f); + int r0 = co_await f; ASSERT(r0 == 0); - int r1 = waitNext(f); + int r1 = co_await f; ASSERT(r1 == 1); - int r2 = waitNext(f); + int r2 = co_await f; ASSERT(r2 == 2); - wait(delay(1.0)); - int r3 = waitNext(f); + co_await delay(1.0); + int r3 = co_await f; ASSERT(r3 == 3); - wait(t); - return Void(); + co_await t; } TEST_CASE("/flow/thread/ThreadReturnPromiseStream_Error") { - state AsyncTaskExecutor exc(1); + AsyncTaskExecutor exc(1); noUnseed = true; { ThreadReturnPromiseStream s1; - state ThreadFutureStream f1 = s1.getFuture(); - state Future t1 = exc.post([s1 = std::move(s1)]() mutable { + ThreadFutureStream f1 = s1.getFuture(); + Future t1 = exc.post([s1 = std::move(s1)]() mutable { std::this_thread::sleep_for(2s); return Void(); }); try { - int _ = waitNext(f1); + co_await f1; ASSERT(false); } catch (Error& e) { ASSERT(e.code() == error_code_broken_promise); } - wait(t1); + co_await t1; } { ThreadReturnPromiseStream s2; - state ThreadFutureStream f2 = s2.getFuture(); - state Future t2 = exc.post([s2 = std::move(s2)]() mutable { + ThreadFutureStream f2 = s2.getFuture(); + Future t2 = exc.post([s2 = std::move(s2)]() mutable { std::this_thread::sleep_for(2s); s2.sendError(transaction_too_old()); return Void(); }); try { - int _ = waitNext(f2); + co_await f2; ASSERT(false); } catch (Error& e) { ASSERT(e.code() == error_code_transaction_too_old); } - wait(t2); + co_await t2; } - - return Void(); } TEST_CASE("/fdbrpc/waitValueOrSignal/peerDisconnect") { @@ -1746,17 +1838,16 @@ TEST_CASE("/fdbrpc/waitValueOrSignal/peerDisconnect") { // peer->disconnect, and PeerHolder only touches outstandingReplies. Note that Peer construction // also updates the global failure monitor status for fakeAddr. NetworkAddress fakeAddr = NetworkAddress::parse("1.2.3.4:1234"); - state Reference peer = makeReference(nullptr, fakeAddr); + Reference peer = makeReference(nullptr, fakeAddr); // Create a value future that never resolves (simulating a stuck RPC to unreachable storage server) - state Promise neverReply; + Promise neverReply; // No failure signal either (simulating failure monitor not yet detecting the failure) - state Endpoint ep; + Endpoint ep; // Call waitValueOrSignal with the peer - state Future> result = - waitValueOrSignal(neverReply.getFuture(), Never(), ep, ReplyPromise(), peer); + Future> result = waitValueOrSignal(neverReply.getFuture(), Never(), ep, ReplyPromise(), peer); // Result should not be ready yet - the reply hasn't come and disconnect hasn't fired ASSERT(!result.isReady()); @@ -1780,7 +1871,7 @@ TEST_CASE("/flow/IThreadPool/ThreadReturnPromiseStream_DestroyPromise") { std::cout << "ThreadReturnPromiseStream with future > 0, promise == 0, end_of_stream sent\n"; // After all references to PromiseStream are gone, FutureStream should still be able to get // all the values if PromiseStream had reached end_of_stream. - state ThreadFutureStream fs1 = ([]() { + ThreadFutureStream fs1 = ([]() { ThreadReturnPromiseStream p; auto ret = p.getFuture(); for (int i = 0; i < 10; i++) { @@ -1791,12 +1882,12 @@ TEST_CASE("/flow/IThreadPool/ThreadReturnPromiseStream_DestroyPromise") { return ret; })(); - wait(delay(1)); + co_await delay(1); - state int recvd = 0; + int recvd = 0; try { while (true) { - int _ = waitNext(fs1); + co_await fs1; recvd += 1; } } catch (Error& err) { @@ -1812,7 +1903,7 @@ TEST_CASE("/flow/IThreadPool/ThreadReturnPromiseStream_DestroyPromise") { { // After all references to PromiseStream are gone, but the stream was not ended cleanly // by sending end_of_stream, we should instead get broken_promise. - state ThreadFutureStream fs2 = ([]() { + ThreadFutureStream fs2 = ([]() { ThreadReturnPromiseStream p; auto ret = p.getFuture(); for (int i = 0; i < 10; i++) { @@ -1822,13 +1913,15 @@ TEST_CASE("/flow/IThreadPool/ThreadReturnPromiseStream_DestroyPromise") { return ret; })(); - wait(delay(1)); + co_await delay(1); try { while (true) { - choose { - when(int _ = waitNext(fs2)) {} - when(wait(delay(1))) { + if (fs2.isReady()) { + co_await fs2; + } else { + auto res = co_await race(fs2, delay(1)); + if (res.index() == 1) { ASSERT(false); // shouldn't hangup if end_of_stream is not sent. break; } @@ -1838,8 +1931,6 @@ TEST_CASE("/flow/IThreadPool/ThreadReturnPromiseStream_DestroyPromise") { ASSERT(err.code() == error_code_broken_promise); } } - - return Void(); } TEST_CASE("/fdbrpc/waitValueOrSignal/noPeerFallback") { @@ -1847,10 +1938,10 @@ TEST_CASE("/fdbrpc/waitValueOrSignal/noPeerFallback") { // The broken_promise path should still be handled: when the reply promise breaks, // the endpoint should be marked as not found and value set to Never(). - state Promise reply; + Promise reply; // Call waitValueOrSignal without a peer (default behavior) - state Future> result = waitValueOrSignal(reply.getFuture(), Never(), Endpoint()); + Future> result = waitValueOrSignal(reply.getFuture(), Never(), Endpoint()); ASSERT(!result.isReady()); @@ -1860,10 +1951,8 @@ TEST_CASE("/fdbrpc/waitValueOrSignal/noPeerFallback") { // waitValueOrSignal should handle broken_promise by setting value = Never() and looping. // Since signal is Never() and there's no peer disconnect, it should now wait forever. // We verify it doesn't crash and the result is NOT ready (it's stuck in the loop). - wait(delay(0.1)); + co_await delay(0.1); ASSERT(!result.isReady()); - - return Void(); } TEST_CASE("/fdbrpc/waitValueOrSignal/retryOnDisconnect") { @@ -1878,17 +1967,17 @@ TEST_CASE("/fdbrpc/waitValueOrSignal/retryOnDisconnect") { NetworkAddress addr1 = NetworkAddress::parse("1.2.3.4:1234"); NetworkAddress addr2 = NetworkAddress::parse("1.2.3.5:1234"); - state Reference peer1 = makeReference(nullptr, addr1); - state Reference peer2 = makeReference(nullptr, addr2); + Reference peer1 = makeReference(nullptr, addr1); + Reference peer2 = makeReference(nullptr, addr2); - state int numAttempts = 0; + int numAttempts = 0; // --- Attempt 1: peer1 disconnects mid-request (simulating K8s NAT timeout) --- { - state Promise reply1; - state Endpoint ep1; + Promise reply1; + Endpoint ep1; - state Future> result1 = + Future> result1 = waitValueOrSignal(reply1.getFuture(), Never(), ep1, ReplyPromise(), peer1); numAttempts++; @@ -1907,10 +1996,10 @@ TEST_CASE("/fdbrpc/waitValueOrSignal/retryOnDisconnect") { // --- Attempt 2: retry to peer2, which responds successfully --- // This is what loadBalance does: on request_maybe_delivered, pick next alternative and retry { - state Promise reply2; - state Endpoint ep2; + Promise reply2; + Endpoint ep2; - state Future> result2 = + Future> result2 = waitValueOrSignal(reply2.getFuture(), Never(), ep2, ReplyPromise(), peer2); numAttempts++; diff --git a/fdbrpc/FlowTransport.cpp b/fdbrpc/FlowTransport.cpp index 6f72e816ce1..10eb220a634 100644 --- a/fdbrpc/FlowTransport.cpp +++ b/fdbrpc/FlowTransport.cpp @@ -27,6 +27,7 @@ #include #include +#include #include #include #if VALGRIND @@ -48,7 +49,10 @@ #include "flow/TDMetric.h" #include "flow/ObjectSerializer.h" #include "flow/Platform.h" +#include "flow/ProcessEvents.h" #include "flow/ProtocolVersion.h" +#include "flow/ScopeExit.h" +#include "flow/UnitTest.h" #include "flow/WatchFile.h" #include "flow/IConnection.h" #define XXH_INLINE_ALL @@ -65,7 +69,7 @@ namespace { NetworkAddressList g_currentDeliveryPeerAddress = NetworkAddressList(); bool g_currentDeliverPeerAddressTrusted = false; -Future g_currentDeliveryPeerDisconnect; +const Future* g_currentDeliveryPeerDisconnect = nullptr; } // namespace @@ -617,7 +621,7 @@ static Future connectionReader(TransportData* transport, static void sendLocal(TransportData* self, ISerializeSource const& what, const Endpoint& destination); static ReliablePacket* sendPacket(TransportData* self, - Reference peer, + const Reference& peer, ISerializeSource const& what, const Endpoint& destination, bool reliable); @@ -1169,51 +1173,35 @@ static bool checkCompatible(const PeerCompatibilityPolicy& policy, ProtocolVersi } } -// This actor looks up the task associated with an endpoint -// and sends the message to it. The actual deserialization will +// Looks up the task associated with an endpoint and sends the message to it. The actual deserialization will // be done by that task (see NetworkMessageReceiver). -static Future deliver(Uncancellable, - TransportData* self, - Endpoint destination, - TaskPriority priority, - ArenaReader reader, - NetworkAddress peerAddress, - bool isTrustedPeer, - InReadSocket inReadSocket, - Future disconnect) { - // We want to run the task at the right priority. If the priority is higher than the current priority (which is - // ReadSocket) we can just upgrade. Otherwise we'll context switch so that we don't block other tasks that might run - // with a higher priority. ReplyPromiseStream needs to guarantee that messages are received in the order they were - // sent, so we are using orderedDelay. - // NOTE: don't skip delay(0) when it's local deliver since it could cause out of order object deconstruction. - if (priority < TaskPriority::ReadSocket || !inReadSocket) { - co_await orderedDelay(0, priority); - } else { - g_network->setCurrentTask(priority); - } - +static void deliverNow(TransportData* self, + const Endpoint& destination, + ArenaReader reader, + const NetworkAddress& peerAddress, + bool isTrustedPeer, + const Future& disconnect) { auto receiver = self->endpoints.get(destination.token); if (receiver && (isTrustedPeer || receiver->isPublic())) { if (!checkCompatible(receiver->peerCompatibilityPolicy(), reader.protocolVersion())) { - co_return; + return; } try { ASSERT(g_currentDeliveryPeerAddress == NetworkAddressList()); ASSERT(!g_currentDeliverPeerAddressTrusted); g_currentDeliveryPeerAddress = destination.addresses; g_currentDeliverPeerAddressTrusted = isTrustedPeer; - g_currentDeliveryPeerDisconnect = disconnect; + g_currentDeliveryPeerDisconnect = &disconnect; + auto clearDeliveryContext = ScopeExit([]() { + g_currentDeliveryPeerAddress = NetworkAddressList(); + g_currentDeliverPeerAddressTrusted = false; + g_currentDeliveryPeerDisconnect = nullptr; + }); StringRef data = reader.arenaReadAll(); ASSERT(data.size() > 8); - ArenaObjectReader objReader(reader.arena(), reader.arenaReadAll(), AssumeVersion(reader.protocolVersion())); + ArenaObjectReader objReader(std::move(reader.arena()), data, AssumeVersion(reader.protocolVersion())); receiver->receive(objReader); - g_currentDeliveryPeerAddress = NetworkAddressList(); - g_currentDeliverPeerAddressTrusted = false; - g_currentDeliveryPeerDisconnect = Future(); } catch (Error& e) { - g_currentDeliveryPeerAddress = NetworkAddressList(); - g_currentDeliverPeerAddressTrusted = false; - g_currentDeliveryPeerDisconnect = Future(); TraceEvent(SevError, "ReceiverError") .error(e) .detail("Token", destination.token.toString()) @@ -1257,6 +1245,48 @@ static Future deliver(Uncancellable, } } +static coro::DetachedCoroutine deliverAfterDelay(TransportData* self, + Endpoint destination, + TaskPriority priority, + ArenaReader reader, + NetworkAddress peerAddress, + bool isTrustedPeer, + Future disconnect) { + try { + co_await orderedDelay(0, priority); + deliverNow(self, destination, std::move(reader), peerAddress, isTrustedPeer, disconnect); + } catch (const Error&) { + // Typed delivery errors have no result consumer. + } catch (...) { + (void)unknown_error(); + } +} + +static void deliver(TransportData* self, + Endpoint destination, + TaskPriority priority, + ArenaReader reader, + NetworkAddress peerAddress, + bool isTrustedPeer, + InReadSocket inReadSocket, + const Future& disconnect) { + // Preserve ordering and the asynchronous local/lower-priority delivery boundary. Incoming reads at or above + // ReadSocket can be delivered synchronously without allocating a coroutine frame. + if (priority < TaskPriority::ReadSocket || !inReadSocket) { + deliverAfterDelay(self, destination, priority, std::move(reader), peerAddress, isTrustedPeer, disconnect); + return; + } + + try { + g_network->setCurrentTask(priority); + deliverNow(self, destination, std::move(reader), peerAddress, isTrustedPeer, disconnect); + } catch (const Error&) { + // Typed delivery errors have no result consumer. + } catch (...) { + (void)unknown_error(); + } +} + static void scanPackets(TransportData* transport, uint8_t*& unprocessed_begin, // FIXME: why isn't this called `start`? const uint8_t* e, // FIXME: why isn't this called `end`? @@ -1264,7 +1294,7 @@ static void scanPackets(TransportData* transport, NetworkAddress const& peerAddress, bool isTrustedPeer, ProtocolVersion peerProtocolVersion, - Future disconnect, + Future const& disconnect, IsStableConnection isStableConnection) { // Find each complete packet in the given byte range and queue a ready task to deliver it. // Remove the complete packets from the range by increasing unprocessed_begin. @@ -1389,12 +1419,10 @@ static void scanPackets(TransportData* transport, // we ignore packets to unknown endpoints if they're not going to a stream anyways, so we can just // return here. The main place where this seems to happen is if a ReplyPromise is not waited on // long enough. - // It would be slightly more elegant/readable to put this if-block into the deliver actor, but if - // we have many messages to UnknownEndpoint we want to optimize earlier. As deliver is an actor it - // will allocate some state on the heap and this prevents it from doing that. + // It would be slightly more elegant/readable to put this if-block into deliver, but if we have many messages + // to UnknownEndpoint we want to optimize earlier and avoid scheduling unnecessary delivery work. if (priority != TaskPriority::UnknownEndpoint || (token.first() & TOKEN_STREAM_FLAG) != 0) { - deliver(Uncancellable(), - transport, + deliver(transport, Endpoint({ peerAddress }, token), priority, std::move(reader), @@ -1446,6 +1474,7 @@ static Future connectionReader(TransportData* transport, bool incompatiblePeerCounted = false; NetworkAddress peerAddress; ProtocolVersion peerProtocolVersion; + Future disconnect; bool trusted = transport->allowList(conn->getPeerAddress().ip) && conn->hasTrustedPeer(); peerAddress = conn->getPeerAddress(); @@ -1603,6 +1632,7 @@ static Future connectionReader(TransportData* transport, co_await delay(0); // Check for cancellation } peer->protocolVersion->set(peerProtocolVersion); + disconnect = peer->disconnect.getFuture(); } } @@ -1615,7 +1645,7 @@ static Future connectionReader(TransportData* transport, peerAddress, trusted, peerProtocolVersion, - peer->disconnect.getFuture(), + disconnect, IsStableConnection(g_network->isSimulated() && conn->isStableConnection())); } else { unprocessed_begin = unprocessed_end; @@ -1885,7 +1915,7 @@ Endpoint FlowTransport::loadedEndpoint(const UID& token) { } Future FlowTransport::loadedDisconnect() { - return g_currentDeliveryPeerDisconnect; + return g_currentDeliveryPeerDisconnect != nullptr ? *g_currentDeliveryPeerDisconnect : Future(); } void FlowTransport::addPeerReference(const Endpoint& endpoint, bool isStream) { @@ -1961,8 +1991,7 @@ static void sendLocal(TransportData* self, ISerializeSource const& what, const E ASSERT(!copy.empty()); TaskPriority priority = self->endpoints.getPriority(destination.token); if (priority != TaskPriority::UnknownEndpoint || (destination.token.first() & TOKEN_STREAM_FLAG) != 0) { - deliver(Uncancellable(), - self, + deliver(self, destination, priority, ArenaReader(copy.arena(), copy, AssumeVersion(currentProtocolVersion())), @@ -1974,7 +2003,7 @@ static void sendLocal(TransportData* self, ISerializeSource const& what, const E } static ReliablePacket* sendPacket(TransportData* self, - Reference peer, + const Reference& peer, ISerializeSource const& what, const Endpoint& destination, bool reliable) { @@ -2273,3 +2302,116 @@ static Future watchPublicKeyJwksFile(std::string filePath, TransportData* void FlowTransport::watchPublicKeyFile(const std::string& publicKeyFilePath) { self->publicKeyFileWatch = watchPublicKeyJwksFile(publicKeyFilePath, self); } + +extern bool g_crashOnError; + +namespace { + +class ThrowingDeliveryReceiver : public NetworkMessageReceiver { + const char* exceptionMessage; + int received = 0; + bool validPayload = false; + +public: + explicit ThrowingDeliveryReceiver(const char* exceptionMessage) : exceptionMessage(exceptionMessage) {} + bool isPublic() const override { return true; } + void receive(ArenaObjectReader& reader) override { + UID payload; + reader.deserialize(payload); + validPayload = payload == UID(1, 2); + ++received; + throw std::runtime_error(exceptionMessage); + } + int receivedCount() const { return received; } + bool receivedValidPayload() const { return validPayload; } +}; + +Future checkRpcDeliveryException(Uncancellable, bool deferred) { + // Simulation counts SevError before trace observers can mark expected injections. + if (g_network->isSimulated() || g_crashOnError) { + co_return; + } + auto previousTask = g_network->getCurrentTask(); + auto restoreTask = ScopeExit([previousTask]() { g_network->setCurrentTask(previousTask); }); + TransportData transport(1, WLTOKEN_FIRST_AVAILABLE, nullptr); + // These constructor-started actors borrow transport; stop them before exercising or destroying the fixture. + transport.pingLogger.cancel(); + transport.pingLogger = Future(); + transport.connectionHistoryLoggerF.cancel(); + transport.connectionHistoryLoggerF = Future(); + if (transport.connectionLogWriterThread) { + co_await transport.connectionLogWriterThread->stop(); + transport.connectionLogWriterThread = Reference(); + } + + const char* message = + deferred ? "RpcUnexpectedException/deferredDelivery" : "RpcUnexpectedException/directDelivery"; + int observed = 0; + int unexpected = 0; + bool previousTraceProcessEvents = g_traceProcessEvents; + auto restoreTraceProcessEvents = + ScopeExit([previousTraceProcessEvents]() { g_traceProcessEvents = previousTraceProcessEvents; }); + ProcessEvents::Event observer("TraceEvent::SystemError"_sr, [&](StringRef, const std::any& data, const Error&) { + auto tracePtr = std::any_cast(&data); + if (!tracePtr || !*tracePtr) { + ++unexpected; + return; + } + auto* trace = *tracePtr; + int errorCode = 0; + std::string exception; + if (observed != 0 || trace->getSeverity() != SevError || + !trace->getFields().tryGetInt("ErrorCode", errorCode) || errorCode != error_code_unknown_error || + !trace->getFields().tryGetValue("StdException", exception) || exception != message) { + ++unexpected; + return; + } + ++observed; + trace->detail("ErrorIsInjectedFault", 1); + }); + g_traceProcessEvents = true; + + ThrowingDeliveryReceiver receiver(message); + TaskPriority priority = deferred ? TaskPriority::DefaultPromiseEndpoint : TaskPriority::ReadSocket; + UID token(2, 0); + transport.endpoints.insert(&receiver, token, priority); + auto removeEndpoint = ScopeExit([&]() { transport.endpoints.remove(token, &receiver); }); + NetworkAddress peer = NetworkAddress::parse("127.0.0.1:54321"); + Endpoint destination({ peer }, token); + ProtocolVersion version = g_network->protocolVersion(); + auto bytes = ObjectWriter::toValue(UID(1, 2), AssumeVersion(version)); + Future disconnect = Never(); + deliver(&transport, + destination, + priority, + ArenaReader(bytes.arena(), bytes, AssumeVersion(version)), + peer, + true, + deferred ? InReadSocket::False : InReadSocket::True, + disconnect); + int immediateReceives = receiver.receivedCount(); + int immediateObservations = observed; + if (deferred) { + // The uncancellable helper keeps the receiver and transport alive until detached delivery has finished. + co_await orderedDelay(0, priority); + } + ASSERT_EQ(immediateReceives, deferred ? 0 : 1); + ASSERT_EQ(immediateObservations, deferred ? 0 : 1); + ASSERT_EQ(receiver.receivedCount(), 1); + ASSERT(receiver.receivedValidPayload()); + ASSERT_EQ(observed, 1); + ASSERT_EQ(unexpected, 0); + ASSERT(g_currentDeliveryPeerAddress == NetworkAddressList()); + ASSERT(!g_currentDeliverPeerAddressTrusted); + ASSERT(g_currentDeliveryPeerDisconnect == nullptr); +} + +} // namespace + +TEST_CASE("noSim/fdbrpc/RpcUnexpectedException/directDelivery") { + return checkRpcDeliveryException(Uncancellable(), false); +} + +TEST_CASE("noSim/fdbrpc/RpcUnexpectedException/deferredDelivery") { + return checkRpcDeliveryException(Uncancellable(), true); +} diff --git a/fdbrpc/LoadBalance.cpp b/fdbrpc/LoadBalance.cpp index a9478cf411c..bfb4a130a72 100644 --- a/fdbrpc/LoadBalance.cpp +++ b/fdbrpc/LoadBalance.cpp @@ -18,7 +18,7 @@ * limitations under the License. */ -#include "fdbrpc/LoadBalance.actor.h" +#include "fdbrpc/LoadBalance.h" #include "flow/CoroUtils.h" #include "flow/UnitTest.h" #include "flow/flow.h" diff --git a/fdbrpc/Stats.cpp b/fdbrpc/Stats.cpp index b1791d2be12..88c5ec2693f 100644 --- a/fdbrpc/Stats.cpp +++ b/fdbrpc/Stats.cpp @@ -238,6 +238,10 @@ void LatencySample::addMeasurement(double measurement) { sketch.addSample(measurement); } +void LatencySample::addMeasurementPair(double measurement, LatencySample& other) { + sketch.addSamplePair(measurement, other.sketch); +} + void LatencySample::logSample() { if (skipTraceOnSilentInterval && sketch.getPopulationSize() == 0) { return; diff --git a/fdbrpc/actorFuzz.py b/fdbrpc/actorFuzz.py deleted file mode 100755 index 3f41415ebd0..00000000000 --- a/fdbrpc/actorFuzz.py +++ /dev/null @@ -1,494 +0,0 @@ -# -# actorFuzz.py -# -# This source file is part of the FoundationDB open source project -# -# Copyright 2013-2026 Apple Inc. and the FoundationDB project authors -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -# - -import random -import copy - - -class Context: - tok = 0 - inLoop = False - indent = 0 - - def __init__(self): - self.random = random.Random() - - def uniqueID(self): - return self.random.randint(100000, 999999) - - -class InfiniteLoop(Exception): - pass - - -class ExecContext: - iterationsLeft = 1000 - ifstate = 0 - - def __init__(self, inputSeq): - self.input = iter(inputSeq) - self.output = [] - - def inp(self): - return next(self.input) - - def out(self, x): - self.output.append(x) - - def infinityCheck(self): - self.iterationsLeft -= 1 - if self.iterationsLeft <= 0: - raise InfiniteLoop() - - -OK = 1 -BREAK = 2 -THROW = 3 -RETURN = 4 -CONTINUE = 5 - - -def indent(cx): - return "\t" * cx.indent - - -class F(object): - def unreachable(self): - return False - - def containsbreak(self): - return False - - -class hashF(F): - def __init__(self, cx): - self.cx = cx - self.uniqueID = cx.uniqueID() - - def __str__(self): - return indent(self.cx) + "outputStream.send( %d );\n" % self.uniqueID - - def eval(self, ecx): - ecx.infinityCheck() - ecx.out(self.uniqueID) - return OK - - -class compoundF(F): - def __init__(self, cx, children): - self.cx = cx - self.children = [] - for c in children: - self.children.append(c) - if c.unreachable(): - self.unreachable = lambda: 1 - break - - def __str__(self): - return "".join(str(c) for c in self.children) - - def eval(self, ecx): - for c in self.children: - ecx.infinityCheck() - result = c.eval(ecx) - if result != OK: - break - return result - - def containsbreak(self): - return any(c.containsbreak() for c in self.children) - - -class loopF(F): - def __init__(self, cx): - self.cx = cx - ccx = copy.copy(cx) - ccx.indent += 1 - ccx.inLoop = True - self.body = compoundF(ccx, [hashF(ccx)] + [fuzzCode(ccx)(ccx)] + [hashF(ccx)]) - self.uniqueID = cx.uniqueID() - self.forever = cx.random.random() < 0.1 - - def __str__(self): - if self.forever: - return ( - indent(self.cx) + "loop {\n" + str(self.body) + indent(self.cx) + "}\n" - ) - else: - return ( - indent(self.cx) - + "state int i%d; for(i%d = 0; i%d < 5; i%d++) {\n" - % ((self.uniqueID,) * 4) - + str(self.body) - + indent(self.cx) - + "}\n" - ) - - def eval(self, ecx): - if self.forever: - while True: - ecx.infinityCheck() - result = self.body.eval(ecx) - if result == BREAK: - break - elif result not in (OK, CONTINUE): - return result - else: - for i in range(5): - ecx.infinityCheck() - result = self.body.eval(ecx) - if result == BREAK: - break - elif result not in (OK, CONTINUE): - return result - return OK - - def unreachable(self): - return self.forever and not self.body.containsbreak() - - -class rangeForF(F): - def __init__(self, cx): - self.cx = cx - ccx = copy.copy(cx) - ccx.indent += 1 - ccx.inLoop = True - self.body = compoundF(ccx, [hashF(ccx)] + [fuzzCode(ccx)(ccx)] + [hashF(ccx)]) - self.uniqueID = cx.uniqueID() - - def __str__(self): - return ( - indent(self.cx) - + ("\n" + indent(self.cx)) - .join( - [ - "state std::vector V;", - "V.push_back(1);", - "V.push_back(2);", - "V.push_back(3);", - "for( auto i : V ) {\n", - ] - ) - .replace("V", "list%d" % self.uniqueID) - + indent(self.cx) - + "\t(void)i;\n" - + str(self.body) # Suppress -Wunused-variable warning in generated code - + indent(self.cx) - + "}\n" - ) - - def eval(self, ecx): - for i in range(1, 4): - ecx.infinityCheck() - result = self.body.eval(ecx) - if result == BREAK: - break - elif result not in (OK, CONTINUE): - return result - return OK - - def unreachable(self): - return False - - -class ifF(F): - def __init__(self, cx): - self.cx = cx - ccx = copy.copy(cx) - ccx.indent += 1 - self.toggle = cx.random.randint(0, 1) - self.ifbody = compoundF(ccx, [hashF(ccx)] + [fuzzCode(ccx)(ccx)] + [hashF(ccx)]) - if cx.random.random() < 0.5: - ccx = copy.copy(cx) - ccx.indent += 1 - self.elsebody = compoundF( - ccx, [hashF(ccx)] + [fuzzCode(ccx)(ccx)] + [hashF(ccx)] - ) - else: - self.elsebody = None - - def __str__(self): - s = ( - indent(self.cx) - + "if ( (++ifstate&1) == %d ) {\n" % self.toggle - + str(self.ifbody) - ) - if self.elsebody: - s += indent(self.cx) + "} else {\n" + str(self.elsebody) - s += indent(self.cx) + "}\n" - return s - - def eval(self, ecx): - ecx.infinityCheck() - ecx.ifstate = ecx.ifstate + 1 - if (ecx.ifstate & 1) == self.toggle: - return self.ifbody.eval(ecx) - elif self.elsebody: - return self.elsebody.eval(ecx) - else: - return OK - - def unreachable(self): - return ( - self.elsebody and self.ifbody.unreachable() and self.elsebody.unreachable() - ) - - def containsbreak(self): - return self.ifbody.containsbreak() or ( - self.elsebody and self.elsebody.containsbreak() - ) - - -class tryF(F): - def __init__(self, cx): - self.cx = cx - ccx = copy.copy(cx) - ccx.indent += 1 - self.body = compoundF(ccx, [hashF(ccx)] + [fuzzCode(ccx)(ccx)] + [hashF(ccx)]) - ccx = copy.copy(cx) - ccx.indent += 1 - self.catch = compoundF(ccx, [hashF(ccx)] + [fuzzCode(ccx)(ccx)] + [hashF(ccx)]) - - def __str__(self): - return ( - indent(self.cx) - + "try {\n" - + str(self.body) - + indent(self.cx) - + "} catch (...) {\n" - + str(self.catch) - + indent(self.cx) - + "}\n" - ) - - def eval(self, ecx): - ecx.infinityCheck() - result = self.body.eval(ecx) - if result != THROW: - return result - return self.catch.eval(ecx) - - def unreachable(self): - return self.body.unreachable() and self.catch.unreachable() - - def containsbreak(self): - return self.body.containsbreak() or self.catch.containsbreak() - - -def doubleF(cx): - return compoundF(cx, [fuzzCode(cx)(cx)] + [hashF(cx)] + [fuzzCode(cx)(cx)]) - - -class breakF(F): - def __init__(self, cx): - self.cx = cx - - def __str__(self): - return indent(self.cx) + "break;\n" - - def unreachable(self): - return True - - def eval(self, ecx): - ecx.infinityCheck() - return BREAK - - def containsbreak(self): - return True - - -class continueF(F): - def __init__(self, cx): - self.cx = cx - - def __str__(self): - return indent(self.cx) + "continue;\n" - - def unreachable(self): - return True - - def eval(self, ecx): - ecx.infinityCheck() - return CONTINUE - - -class waitF(F): - def __init__(self, cx): - self.cx = cx - self.uniqueID = cx.uniqueID() - - def __str__(self): - return ( - indent(self.cx) - + "int input = waitNext( inputStream );\n" - + indent(self.cx) - + "outputStream.send( input + %d );\n" % self.uniqueID - ) - - def eval(self, ecx): - ecx.infinityCheck() - input = ecx.inp() - ecx.out((input + self.uniqueID) & 0xFFFFFFFF) - return OK - - -class throwF(F): - def __init__(self, cx): - self.cx = cx - - def __str__(self): - return indent(self.cx) + "throw operation_failed();\n" - - def unreachable(self): - return True - - def eval(self, ecx): - ecx.infinityCheck() - return THROW - - -class throwF2(throwF): - def __str__(self): - return indent(self.cx) + "throw_operation_failed();\n" - - def unreachable(self): - return False # The actor compiler doesn't know the function never returns - - -class throwF3(throwF): - def __str__(self): - return indent(self.cx) + "wait( error ); // throw operation_failed()\n" - - def unreachable(self): - return False # The actor compiler doesn't know that 'error' always contains an error - - -class returnF(F): - def __init__(self, cx): - self.cx = cx - self.uniqueID = cx.uniqueID() - - def __str__(self): - return indent(self.cx) + "return %d;\n" % self.uniqueID - - def unreachable(self): - return True - - def eval(self, ecx): - ecx.infinityCheck() - ecx.returnValue = self.uniqueID - return RETURN - - -def fuzzCode(cx): - choices = [loopF, rangeForF, tryF, doubleF, ifF] - if cx.indent < 2: - choices = choices * 2 - choices += [waitF, returnF] - if cx.inLoop: - choices += [breakF, continueF] - choices = choices * 3 + [throwF, throwF2, throwF3] - return cx.random.choice(choices) - - -def randomActor(index): - while 1: - cx = Context() - cx.indent += 1 - actor = fuzzCode(cx)(cx) - actor = compoundF( - cx, [actor, returnF(cx)] - ) # Add a return at the end if the end is reachable - name = "actorFuzz%d" % index - text = ( - "ACTOR Future %s( FutureStream inputStream, PromiseStream outputStream, Future error ) {\n" - % name - + "\tstate int ifstate = 0;\n" - + str(actor) - + "}" - ) - ecx = actor.ecx = ExecContext((i + 1) * 1000 for i in range(1000000)) - try: - result = actor.eval(ecx) - except InfiniteLoop: - print("Infinite loop for actor %s" % name) - continue - if result == RETURN: - ecx.out(ecx.returnValue) - elif result == THROW: - ecx.out(1000) - else: - print(text) - raise Exception("Invalid eval result: " + str(result)) - actor.name = name - actor.text = text - - return actor - - -header = """ -/* - * ActorFuzz.actor.cpp - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -""" - - -testCaseCount = 30 -outputFile = open("ActorFuzz.actor.cpp", "wt") -print(header, file=outputFile) -print( - "// THIS FILE WAS GENERATED BY actorFuzz.py; DO NOT MODIFY IT DIRECTLY\n", - file=outputFile, -) -print('#include "ActorFuzz.h"\n', file=outputFile) -print("#ifndef WIN32\n", file=outputFile) - -actors = [randomActor(i) for i in range(testCaseCount)] - -for actor in actors: - print(actor.text + "\n", file=outputFile) - -print("std::pair actorFuzzTests() {\n\tint testsOK = 0;", file=outputFile) -for actor in actors: - print( - '\ttestsOK += testFuzzActor( &%s, "%s", {%s} );' - % (actor.name, actor.name, ",".join(str(e) for e in actor.ecx.output)), - file=outputFile, - ) -print("\treturn std::make_pair(testsOK, %d);\n}" % len(actors), file=outputFile) -print("#endif // WIN32\n", file=outputFile) -outputFile.close() diff --git a/fdbrpc/bench/BenchIONet2.actor.cpp b/fdbrpc/bench/BenchIONet2.cpp similarity index 63% rename from fdbrpc/bench/BenchIONet2.actor.cpp rename to fdbrpc/bench/BenchIONet2.cpp index 4b3edacf53d..007c4be31b3 100644 --- a/fdbrpc/bench/BenchIONet2.actor.cpp +++ b/fdbrpc/bench/BenchIONet2.cpp @@ -1,5 +1,5 @@ /* - * BenchIONet2.actor.cpp + * BenchIONet2.cpp * * This source file is part of the FoundationDB open source project * @@ -28,20 +28,6 @@ #include "flow/ThreadHelper.h" #include "flow/IAsyncFile.h" -#include "flow/actorcompiler.h" // This must be the last #include. - -ACTOR static Future incrementActor(TaskPriority priority, uint32_t* sum) { - wait(delay(0, priority)); - DeterministicRandom rand(1); - int randSum = 0; - for (int i = 0; i < 1e6; ++i) { - randSum += rand.randomInt(0, 3); - } - benchmark::DoNotOptimize(randSum); - ++(*sum); - return Void(); -} - static Future incrementCoroutine(TaskPriority priority, uint32_t* sum, NoThrowOnCancel = {}) { co_await delay(0, priority); DeterministicRandom rand(1); @@ -57,32 +43,6 @@ static inline TaskPriority getRandomTaskPriority(DeterministicRandom& rand) { return static_cast(rand.randomInt(0, 100)); } -ACTOR static Future benchIONet2Actor(benchmark::State* benchState) { - state size_t actorCount = benchState->range(0); - state uint32_t sum; - state uint64_t seed = 1; - state std::unique_ptr data(new char[4096]); - memset(data.get(), 0, 4096); - while (benchState->KeepRunning()) { - sum = 0; - state std::vector> futures; - futures.reserve(actorCount); - DeterministicRandom rand(seed); - for (int i = 0; i < actorCount; ++i) { - futures.push_back(incrementActor(getRandomTaskPriority(rand), &sum)); - } - state Reference f = wait(IAsyncFileSystem::filesystem()->open( - "/tmp/__test-benchmark-file__", - IAsyncFile::OPEN_ATOMIC_WRITE_AND_CREATE | IAsyncFile::OPEN_CREATE | IAsyncFile::OPEN_READWRITE, - 0600)); - wait(f->write(data.get(), 4096, 0)); - wait(f->sync()); - benchmark::DoNotOptimize(sum); - } - benchState->SetItemsProcessed(actorCount * static_cast(benchState->iterations())); - return Void(); -} - static Future benchIONet2Coroutine(benchmark::State* benchState) { size_t actorCount = benchState->range(0); uint32_t sum = 0; @@ -96,7 +56,7 @@ static Future benchIONet2Coroutine(benchmark::State* benchState) { memset(data.get(), 0, 4096); while (benchState->KeepRunning()) { sum = 0; - // Match the ACTOR's declaration-point reset, including releasing capacity. + // Each iteration includes allocation of a fresh batch. futures = std::vector>(); futures.reserve(actorCount); DeterministicRandom rand(seed); @@ -114,13 +74,8 @@ static Future benchIONet2Coroutine(benchmark::State* benchState) { benchState->SetItemsProcessed(actorCount * static_cast(benchState->iterations())); } -static void bench_ionet2_actor(benchmark::State& benchState) { - onMainThread([&benchState] { return benchIONet2Actor(&benchState); }).blockUntilReady(); -} - static void bench_ionet2_coroutine(benchmark::State& benchState) { onMainThread([&benchState] { return benchIONet2Coroutine(&benchState); }).blockUntilReady(); } -BENCHMARK(bench_ionet2_actor)->Range(1, 1 << 16)->ReportAggregatesOnly(true); BENCHMARK(bench_ionet2_coroutine)->Range(1, 1 << 16)->ReportAggregatesOnly(true); diff --git a/fdbrpc/bench/BenchSelectReplicas.cpp b/fdbrpc/bench/BenchSelectReplicas.cpp index 453fe5b6d66..aa18c9279d2 100644 --- a/fdbrpc/bench/BenchSelectReplicas.cpp +++ b/fdbrpc/bench/BenchSelectReplicas.cpp @@ -54,6 +54,7 @@ static void bench_select_replicas(int repCount, benchmark::State& state) { // fromServersSet->DisplayEntries(); + localityGroupEntries.reserve(serverCount > 0 ? serverCount : 0); for (int i = 0; i < serverCount; i++) { localityGroupEntries.push_back(fromServersGroup->getEntry(i)); } diff --git a/fdbrpc/dsltest.actor.cpp b/fdbrpc/dsltest.actor.cpp deleted file mode 100644 index ece5deffb38..00000000000 --- a/fdbrpc/dsltest.actor.cpp +++ /dev/null @@ -1,1500 +0,0 @@ -/* - * dsltest.actor.cpp - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -#include -#include -#include "flow/FastRef.h" -#undef ERROR -#include "fdbrpc/simulator.h" -#include "ActorFuzz.h" -#include "flow/DeterministicRandom.h" -#include "flow/ThreadHelper.h" -#include "flow/actorcompiler.h" // This must be the last #include. - -// -// TODO: IS THIS FILE NEEDED? IS IT RUN ON A RECURRING BASIS? IF NOT, DELETE IT. -// - -void* allocateLargePages(int total); - -bool testFuzzActor(Future (*actor)(FutureStream const&, PromiseStream const&, Future const&), - const char* desc, - std::vector const& expectedOutput) { - // Run the test 5 times with different "timing" - int i, outCount; - bool ok = true; - for (int trial = 0; trial < 5; trial++) { - PromiseStream in, out; - Promise err; - int before = deterministicRandom()->randomInt(0, 4); - int errorBefore = before + deterministicRandom()->randomInt(0, 4); - // printf("\t\tTrial #%d: %d, %d\n", trial, before, errorBefore); - if (errorBefore <= before) - err.sendError(operation_failed()); - for (i = 0; i < before; i++) { - in.send((i + 1) * 1000); - } - Future ret = (*actor)(in.getFuture(), out, err.getFuture()); - while (i < 1000000 && !ret.isReady()) { - i++; - if (errorBefore == i) - err.sendError(operation_failed()); - in.send(i * 1000); - } - if (ret.isReady()) { - if (ret.isError()) - out.send(ret.getError().code()); - else - out.send(ret.get()); - } else { - printf("\tERROR: %s did not return after consuming %d input values\n", desc, i); - if (trial) - printf("\t\tResult was inconsistent between runs! (Trial %d)\n", trial); - ok = false; - // return false; - } - - outCount = -1; - while (out.getFuture().isReady()) { - int o = out.getFuture().pop(); - outCount++; - if (outCount < expectedOutput.size() && expectedOutput[outCount] != o) { - printf("\tERROR: %s output #%d incorrect: %d != expected %d\n", - desc, - outCount, - o, - expectedOutput[outCount]); - if (trial) - printf("\t\tResult was inconsistent between runs!\n"); - ok = false; - // return false; - } - } - if (outCount + 1 != expectedOutput.size()) { - printf( - "\tERROR: %s output length incorrect: %d != expected %zu\n", desc, outCount + 1, expectedOutput.size()); - if (trial) - printf("\t\tResult was inconsistent between runs!\n"); - ok = false; - // return false; - } - - // We might have put in values that weren't actually consumed... - while (in.getFuture().isReady()) { - in.getFuture().pop(); - i--; - } - } - // printf("\t%s: OK, %d input values -> %d output values\n", desc, i, outCount); - return ok; -} - -#if 0 -void memoryTest2() { - const int Size = 2000 << 20; - const int Reads = 4 << 20; - const int MaxThreads = 4; - - char* block = new char[Size]; - memset(block, 0, Size); - - char** random = new char*[ Reads * MaxThreads ]; - random[0] = block; - for(int i=1; irandomInt(0, Size) ]; - random[i++] = s; - /*for(int j=0; j<10 && irandomInt(0, 4096); - if (random[i] >= block+Size) random[i] -= Size; - }*/ - } - - for(int threads=1; threads<=MaxThreads; threads++) { - double tstart = timer(); - - std::vector> done; - for(int t=0; t( [r,Reads] () -> Void { - for(int i=0; i= 1; n--) { - int k = deterministicRandom()->randomInt(0, n); //random.IRandomX(0, n-1); - std::swap( x[k], x[n] ); - } - } else { - std::cout <<" Sequential permutation" << std::endl; - // Sequential - for(int i=0; irandomInt(0,N) ]; - } - for(int T=1; T<=MT; T+=T) { - double start = timer(); - std::vector< Future > done; - for(int t=0; t( [start,N2,TraversalsPerThread] () -> double { - void **p[MaxTraversalsPerThread]; - for(int j=0; j 1) - _mm_prefetch( (const char*)p[j], _MM_HINT_T0 ); - } - for(int j=0; j -[[flow_allow_discard]] Future addN(Future in) { - X i = wait(in); - return i + N; -} - -ACTOR template -[[flow_allow_discard]] Future switchTest(FutureStream as, Future oneb) { - loop choose { - when(A a = waitNext(as)) { - std::cout << "A " << a << std::endl; - } - when(B b = wait(oneb)) { - std::cout << "B " << b << std::endl; - break; - } - } - loop { - std::cout << "Done!" << std::endl; - return Void(); - } -} - -class TestBuffer : public ReferenceCounted { -public: - static TestBuffer* create(int length) { -#if defined(__INTEL_COMPILER) - return new TestBuffer(length); -#else - auto b = (TestBuffer*)new int[(length + 7) / 4]; - new (b) TestBuffer(length); - return b; -#endif - } -#if !defined(__INTEL_COMPILER) - void operator delete(void* buf) { - std::cout << "Freeing buffer" << std::endl; - delete[] (int*)buf; - } -#endif - - int size() const { return length; } - uint8_t* begin() { return data; } - uint8_t* end() { return data + length; } - const uint8_t* begin() const { return data; } - const uint8_t* end() const { return data + length; } - -private: - explicit TestBuffer(int length) noexcept : length(length) {} - int length; - uint8_t data[1]; -}; - -int fastKeyCount = 0; - -class FastKey : public FastAllocated, public ReferenceCounted { -public: - FastKey() : length(0) {} - FastKey(char* b, int length) : length(length) { - ASSERT(length <= sizeof(data)); - memcpy(data, b, length); - } - ~FastKey() { fastKeyCount++; } - int size() const { return length; } - uint8_t* begin() { return data; } - uint8_t* end() { return data + length; } - const uint8_t* begin() const { return data; } - const uint8_t* end() const { return data + length; } - -private: - int length; - uint8_t data[252]; -}; - -struct TestB : FastAllocated { - char x[65]; -}; - -void fastAllocTest() { - double t; - - std::vector d; - for (int i = 0; i < 1000000; i++) { - d.push_back(FastAllocator<64>::allocate()); - int r = deterministicRandom()->randomInt(0, 1000000); - if (r < d.size()) { - FastAllocator<64>::release(d[r]); - d[r] = d.back(); - d.pop_back(); - } - } - std::sort(d.begin(), d.end()); - if (std::unique(d.begin(), d.end()) != d.end()) - std::cout << "Pointer returned twice!?" << std::endl; - - for (int i = 0; i < 2; i++) { - void* p = FastAllocator<64>::allocate(); - void* q = FastAllocator<64>::allocate(); - std::cout << (intptr_t)p << " " << (intptr_t)q << std::endl; - FastAllocator<64>::release(p); - FastAllocator<64>::release(q); - } - - t = timer(); - for (int i = 0; i < 1000000; i++) - (void)FastAllocator<64>::allocate(); - t = timer() - t; - std::cout << "Allocations: " << (1 / t) << "M/sec" << std::endl; - - t = timer(); - for (int i = 0; i < 1000000; i++) - FastAllocator<64>::release(FastAllocator<64>::allocate()); - t = timer() - t; - std::cout << "Allocate/Release pairs: " << (1 / t) << "M/sec" << std::endl; - - t = timer(); - void* pp[100]; - for (int i = 0; i < 10000; i++) { - for (int j = 0; j < 100; j++) - pp[j] = FastAllocator<64>::allocate(); - for (int j = 0; j < 100; j++) - FastAllocator<64>::release(pp[j]); - } - t = timer() - t; - std::cout << "Allocate/Release interleaved(100): " << (1 / t) << "M/sec" << std::endl; - - t = timer(); - for (int i = 0; i < 1000000; i++) - delete new TestB; - t = timer() - t; - std::cout << "Allocate/Release TestB pairs: " << (1 / t) << "M/sec" << std::endl; - -#if FLOW_THREAD_SAFE - t = timer(); - std::vector> results; - for (int i = 0; i < 4; i++) - results.push_back(inThread([]() -> bool { - TestB* pp[100]; - for (int i = 0; i < 10000; i++) { - for (int j = 0; j < 100; j++) - pp[j] = new TestB; - for (int j = 0; j < 100; j++) - delete pp[j]; - } - return true; - })); - waitForAll(results).getBlocking(); - t = timer() - t; - std::cout << "Threaded Allocate/Release TestB interleaved (100): " << results.size() << " x " << (1 / t) << "M/sec" - << std::endl; -#endif - - volatile int32_t v = 0; - - t = timer(); - for (int i = 0; i < 10000000; i++) - interlockedIncrement(&v); - t = timer() - t; - std::cout << "interlocked increment: " << 10.0 / t << "M/sec " << v << std::endl; - - v = 5; - t = timer(); - for (int i = 0; i < 10000000; i++) { - interlockedCompareExchange(&v, 5, 5); - } - t = timer() - t; - std::cout << "1 state machine: " << 10.0 / t << "M/sec " << v << std::endl; - - v = 0; - t = timer(); - for (int i = 0; i < 10000000; i++) - v++; - t = timer() - t; - std::cout << "volatile increment: " << 10.0 / t << "M/sec " << v << std::endl; - - { - Reference b(TestBuffer::create(1000)); - memcpy(b->begin(), "Hello, world!", 14); - - t = timer(); - for (int i = 0; i < 10000000; i++) { - Reference r = std::move(b); - b = std::move(r); - } - t = timer() - t; - std::cout << "move Reference: " << 10.0 / t << "M/sec " << std::endl; - - t = timer(); - for (int i = 0; i < 10000000; i++) { - Reference r = b; - } - t = timer() - t; - std::cout << "copy (1) Reference: " << 10.0 / t << "M/sec " << std::endl; - - Reference c = b; - t = timer(); - for (int i = 0; i < 10000000; i++) { - Reference r = b; - } - t = timer() - t; - std::cout << "copy (2) Reference: " << 10.0 / t << "M/sec " << std::endl; - - std::cout << (const char*)b->begin() << std::endl; - } - t = timer(); - for (int i = 0; i < 10000000; i++) { - delete new FastKey; - } - t = timer() - t; - std::cout << "delete new FastKey: " << 10.0 / t << "M/sec " << fastKeyCount << std::endl; - - t = timer(); - for (int i = 0; i < 10000000; i++) { - Reference r(new FastKey); - } - t = timer() - t; - std::cout << "new Reference: " << 10.0 / t << "M/sec " << fastKeyCount << std::endl; -} - -template -Future threadSafetySender(std::vector& v, Event& start, Event& ready, int iterations) { - for (int i = 0; i < iterations; i++) { - start.block(); - if (v.size() == 0) - return Void(); - for (int i = 0; i < v.size(); i++) - v[i].send(Void()); - ready.set(); - } - return Void(); -} - -ACTOR [[flow_allow_discard]] void threadSafetyWaiter(Future f, int32_t* count) { - wait(f); - interlockedIncrement(count); -} -ACTOR [[flow_allow_discard]] void threadSafetyWaiter(FutureStream f, int n, int32_t* count) { - while (n--) { - waitNext(f); - interlockedIncrement(count); - } -} - -#if 0 -void threadSafetyTest() { - double t = timer(); - - int N = 10000, V = 100; - - std::vector> v; - Event start, ready; - Future sender = inThread( [&] { return threadSafetySender( v, start, ready, N ); } ); - - for(int i=0; i()); - std::vector> f( v.size() ); - for(int i=0; i> streams( 100 ); - std::vector> streams; - for (int i = 0; i < 100; i++) - streams.push_back(PromiseStream()); - std::vector> v; - Event start, ready; - Future sender = inThread( [&] { return threadSafetySender( v, start, ready, N ); } ); - - for(int i=0; i counts( streams.size() ); - v.clear(); - for(int k=0; krandomInt(0, (int)streams.size()); - counts[i]++; - v.push_back( streams[i] ); - } - - start.set(); - int32_t count = 0; - for(int i=0; i returnCancelRacer( Future f ) { - try { - wait(f); - } catch ( Error& ) { - interlockedIncrement( &cancelled ); - throw; - } - interlockedIncrement( &returned ); - return Void(); -} - -void returnCancelRaceTest() { - int N = 100, M = 100; - for(int i=0; i > promises; - std::vector< Future > futures; - for(int i=0; i < M; i++) { - promises.push_back( Promise() ); - futures.push_back( returnCancelRacer( promises.back().getFuture() ) ); - } - std::random_shuffle( futures.begin(), futures.end() ); - - // FIXME: Doesn't work as written with auto-reset - // events. Probably not particularly racy as written. Test may - // FAIL or PASS at whim. - - Event ev1, ev2; - ThreadFuture b = inThread( [&] ()->Void { - ev1.block(); - for(int i=0; i(); - return Void(); - } ); - ThreadFuture a = inThread([&]()->Void { - ev2.block(); - for(int i=0; i chooseTest(Future a, Future b) { - choose { - when(int A = wait(a)) { - return A; - } - when(int B = wait(b)) { - return B; - } - } -} - -void showArena(ArenaBlock* a, ArenaBlock* parent) { - printf("ArenaBlock %p (<-%p): %d bytes, %d refs\n", a, parent, a->size(), a->debugGetReferenceCount()); - if (!a->isTiny()) { - int o = a->nextBlockOffset; - while (o) { - auto* r = (ArenaBlockRef*)((char*)a->getData() + o); - - // If alignedBuffer is valid then print its pointer and size, else recurse - if (r->aligned4kBufferSize != 0) { - printf("AlignedBuffer %p (<-%p) %u bytes\n", r->aligned4kBuffer, a, r->aligned4kBufferSize); - } else { - showArena(r->next, a); - } - - o = r->nextBlockOffset; - } - } -} - -void arenaTest() { - BinaryWriter wr(AssumeVersion(g_network->protocolVersion())); - { - Arena arena; - VectorRef test; - test.push_back(arena, StringRef(arena, "Hello"_sr)); - test.push_back(arena, StringRef(arena, ", "_sr)); - test.push_back(arena, StringRef(arena, "World!"_sr)); - - for (auto i = test.begin(); i != test.end(); ++i) - for (auto j = i->begin(); j != i->end(); ++j) - std::cout << *j; - std::cout << std::endl; - - wr << test; - } - { - Arena arena2; - VectorRef test2; - BinaryReader reader(wr.getData(), wr.getLength(), AssumeVersion(g_network->protocolVersion())); - reader >> test2 >> arena2; - - for (auto i = test2.begin(); i != test2.end(); ++i) - for (auto j = i->begin(); j != i->end(); ++j) - std::cout << *j; - std::cout << std::endl; - } - - double t = timer(); - for (int i = 0; i < 100; i++) { - Arena ar; - for (int i = 0; i < 10000000; i++) - new (ar) char[10]; - } - printf("100 x 10M x 10B allocated+freed from Arenas: %f sec\n", timer() - t); - - // printf("100M x 8bytes allocations: %d bytes used\n", 0);//ar.getSize()); - // showArena( ar.impl.getPtr(), 0 ); -}; - -ACTOR [[flow_allow_discard]] void testStream(FutureStream xs) { - loop { - int x = waitNext(xs); - std::cout << x << std::endl; - } -} - -ACTOR [[flow_allow_discard]] Future actorTest1(bool b) { - printf("1"); - if (b) - throw future_version(); - return Void(); -} - -ACTOR [[flow_allow_discard]] void actorTest2(bool b) { - printf("2"); - if (b) - throw future_version(); -} - -ACTOR [[flow_allow_discard]] Future actorTest3(bool b) { - try { - if (b) - throw future_version(); - } catch (Error&) { - printf("3"); - return Void(); - } - printf("\nactorTest3 failed\n"); - return Void(); -} - -ACTOR [[flow_allow_discard]] Future actorTest4(bool b) { - state double tstart = now(); - try { - if (b) - throw operation_failed(); - } catch (...) { - wait(delay(1)); - } - if (now() < tstart + 1) - printf("actorTest4 failed"); - else - printf("4"); - return Void(); -} - -ACTOR [[flow_allow_discard]] Future actorTest5() { - state bool caught = false; - - loop { - loop { - state bool inloop = false; - if (caught) { - printf("5"); - return true; - } - try { - loop { - if (inloop) { - printf("\nactorTest5 failed\n"); - return false; - } - inloop = true; - if (1) - throw operation_failed(); - } - } catch (Error&) { - caught = true; - } - } - } -} - -ACTOR [[flow_allow_discard]] Future actorTest6() { - state bool caught = false; - loop { - if (caught) { - printf("6"); - return true; - } - try { - if (1) - throw operation_failed(); - } catch (Error&) { - caught = true; - } - } -} - -ACTOR [[flow_allow_discard]] Future actorTest7() { - try { - loop { - loop { - if (1) - throw operation_failed(); - if (1) { - printf("actorTest7 failed (1)\n"); - return false; - } - if (0) - break; - } - if (1) { - printf("actorTest7 failed (2)\n"); - return false; - } - } - } catch (Error&) { - printf("7"); - return true; - } -} - -ACTOR [[flow_allow_discard]] Future actorTest8() { - state bool caught = false; - state Future set = true; - - loop { - state bool inloop = false; - if (caught) { - printf("8"); - return true; - } - try { - loop { - if (inloop) { - printf("\nactorTest8 failed\n"); - return false; - } - wait(success(set)); - inloop = true; - if (1) - throw operation_failed(); - } - } catch (Error&) { - caught = true; - } - } -} - -ACTOR [[flow_allow_discard]] Future actorTest9A(Future setAfterCalling) { - state int count = 0; - loop { - if (count == 4) { - printf("9"); - return true; - } - if (count && count != 4) { - printf("\nactorTest9 failed\n"); - return false; - } - loop { - loop { - wait(setAfterCalling); - loop { - loop { - count++; - break; - } - wait(Future(Void())); - count++; - break; - } - count++; - break; - } - count++; - break; - } - // loopDepth < 0 ??? - } -} - -Future actorTest9() { - Promise p; - Future f = actorTest9A(p.getFuture()); - p.send(Void()); - return f; -} - -ACTOR [[flow_allow_discard]] Future actorTest10A(FutureStream inputStream, Future go) { - state int i; - for (i = 0; i < 5; i++) { - wait(go); - int input = waitNext(inputStream); - (void)input; - } - return Void(); -} - -void actorTest10() { - PromiseStream ins; - Promise go; - for (int x = 0; x < 2; x++) - ins.send(x); - Future a = actorTest10A(ins.getFuture(), go.getFuture()); - go.send(Void()); - for (int x = 0; x < 3; x++) - ins.send(x); - if (!a.isReady()) - printf("\nactorTest10 failed\n"); - else - printf("10"); -} - -ACTOR [[flow_allow_discard]] Future cancellable() { - wait(Never()); - return Void(); -} - -ACTOR [[flow_allow_discard]] Future simple() { - return Void(); -} - -ACTOR [[flow_allow_discard]] Future simpleWait() { - wait(Future(Void())); - return Void(); -} - -ACTOR [[flow_allow_discard]] Future simpleRet(Future x) { - int i = wait(x); - return i; -} - -template -Future chain(Future const& x); - -ACTOR template -[[flow_allow_discard]] Future achain(Future x) { - int k = wait(chain(x)); - return k + 1; -} - -template -Future chain(Future const& x) { - return achain(x); -} - -template <> -Future chain<0>(Future const& x) { - return x; -} - -ACTOR [[flow_allow_discard]] Future chain2(Future x, int i); - -ACTOR [[flow_allow_discard]] Future chain2(Future x, int i) { - if (i > 1) { - int k = wait(chain2(x, i - 1)); - return k + 1; - } else { - int k = wait(x); - return k + i; - } -} - -ACTOR [[flow_allow_discard]] Future cancellable2() { - try { - wait(Never()); - return Void(); - } catch (Error& e) { - throw; - } -} - -ACTOR [[flow_allow_discard]] Future introLoadValueFromDisk(Future filename) { - std::string file = wait(filename); - - if (file == "/dev/threes") - return 3; - else - ASSERT(false); - return 0; // does not happen -} - -ACTOR [[flow_allow_discard]] Future introAdd(Future a, Future b) { - state int x = wait(a); - int y = wait(b); - return x + y; // x would be undefined here if it was not "state" -} - -ACTOR [[flow_allow_discard]] Future introFirst(Future a, Future b) { - choose { - when(int x = wait(a)) { - return x; - } - when(int x = wait(b)) { - return x; - } - } -} - -struct AddReply { - int sum; - AddReply() = default; - explicit(false) AddReply(int x) : sum(x) {} - - template - void serialize(Ar& ar) { - serializer(ar, sum); - } -}; - -struct AddRequest { - int a, b; - Promise reply; // Self-addressed envelope - - AddRequest() = default; - AddRequest(int a, int b) : a(a), b(b) {} - - template - void serialize(Ar& ar) { - serializer(ar, a, b, reply); - } -}; - -ACTOR [[flow_allow_discard]] void introAddServer(PromiseStream add) { - loop choose { - when(AddRequest req = waitNext(add.getFuture())) { - printf("%d + %d = %d\n", req.a, req.b, req.a + req.b); - req.reply.send(req.a + req.b); - } - } -} - -void introPromiseFuture() { - Promise myPromise; - - Future myFuture = myPromise.getFuture(); - - myPromise.send(12345); - - ASSERT(myFuture.isReady() && myFuture.get() == 12345); -} - -void introActor() { - Future f = introLoadValueFromDisk(std::string("/dev/threes")); - ASSERT(f.get() == 3); - - Promise a, b; - Future sum = introAdd(a.getFuture(), b.getFuture()); - b.send(3); - ASSERT(!sum.isReady()); - a.send(2); - ASSERT(sum.get() == 5); - - Promise c, d; - Future first = introFirst(c.getFuture(), d.getFuture()); - ASSERT(!first.isReady()); - // d.send(100); - d.sendError(operation_failed()); - ASSERT(first.isError() && first.getError().code() == error_code_operation_failed); - // ASSERT( first.getBlocking() == 100 ); - - PromiseStream addInterface; - introAddServer(addInterface); - - Future reply = addInterface.getReply(AddRequest(5, 2)); - ASSERT(reply.get().sum == 7); - - printf("OK\n"); -} - -template -void chainTest() { - auto startt = timer(); - for (int i = 0; i < 100000; i++) { - Promise p; - Future f = chain(p.getFuture()); - p.send(i); - ASSERT(f.get() == i + N); - } - auto endt = timer(); - printf("chain<%d>: %0.3f M/sec\n", N, 0.1 / (endt - startt)); - - startt = timer(); - for (int i = 0; i < 100000; i++) { - Promise p; - Future f = chain2(p.getFuture(), N); - p.send(i); - ASSERT(f.get() == i + N); - } - endt = timer(); - printf("chain2<%d>: %0.3f M/sec\n", N, 0.1 / (endt - startt)); -} - -ACTOR [[flow_allow_discard]] void cycle(FutureStream in, PromiseStream out, int* ptotal) { - loop { - waitNext(in); - (*ptotal)++; - out.send(Void()); - } -} - -ACTOR [[flow_allow_discard]] Future cycleTime(int nodes, int times) { - state std::vector> n(nodes); - state int total = 0; - - // 1->2, 2->3, ..., n-1->0 - for (int i = 1; i < nodes; i++) - cycle(n[i].getFuture(), n[(i + 1) % nodes], &total); - - state double startT = timer(); - n[1].send(Void()); - loop { - waitNext(n[0].getFuture()); - if (!--times) - break; - n[1].send(Void()); - } - - printf("Ring test: %d nodes, %d total ops, %.3f seconds\n", nodes, total, timer() - startT); - return Void(); -} - -void asyncMapTest() { - Future c; - - { - AsyncMap m1; - m1.set(10, 1); - ASSERT(m1.get(10) == 1); - ASSERT(m1.get(20) == 0); - Future a = m1.onChange(10); - Future b = m1.onChange(20); - c = m1.onChange(30); - ASSERT(!a.isReady() && !b.isReady()); - m1.set(10, 0); - ASSERT(a.isReady() && !a.isError() && !b.isReady() && m1.get(10) == 0); - m1.set(20, 5); - ASSERT(b.isReady() && !b.isError() && m1.get(20) == 5); - - a = m1.onChange(10); - b = m1.onChange(20); - m1.triggerRange(15, 25); - ASSERT(!a.isReady() && b.isReady() && !b.isError() && m1.get(20) == 5); - } - ASSERT(c.isReady() && c.isError() && c.getError().code() == error_code_broken_promise); - - printf("AsyncMap: OK\n"); - - double startt; - AsyncMap m2; - startt = timer(); - for (int i = 0; i < 1000000; i++) { - m2.set(5, 0); - m2.set(5, 1); - } - printf(" set(not present/present): %0.1fM/sec\n", 2.0 / (timer() - startt)); - startt = timer(); - for (int i = 0; i < 1000000; i++) { - m2.set(5, 1); - m2.set(5, 2); - } - printf(" set(present/present): %0.1fM/sec\n", 2.0 / (timer() - startt)); - startt = timer(); - for (int i = 0; i < 1000000; i++) { - m2.set(5, 1); - } - printf(" set(no change): %0.1fM/sec\n", 1.0 / (timer() - startt)); - - m2.set(5, 5); - startt = timer(); - for (int i = 0; i < 1000000; i++) - m2.onChange(5); - printf(" onChange(present, cancelled): %0.1fM/sec\n", 1.0 / (timer() - startt)); - startt = timer(); - for (int i = 0; i < 1000000; i++) - m2.onChange(10); - printf(" onChange(not present, cancelled): %0.1fM/sec\n", 1.0 / (timer() - startt)); - startt = timer(); - for (int i = 0; i < 1000000; i++) { - auto f = m2.onChange(10); - m2.set(10, 1); - m2.set(10, 0); - } - printf(" onChange(not present, set): %0.1fM/sec\n", 1.0 / (timer() - startt)); - startt = timer(); - for (int i = 0; i < 1000000; i++) { - auto f = m2.onChange(5); - m2.set(5, i + 1); - } - printf(" onChange(present, set): %0.1fM/sec\n", 1.0 / (timer() - startt)); -} - -void dsltest() { - double startt, endt; - - setThreadLocalDeterministicRandomSeed(40); - - asyncMapTest(); - - Future ctf = cycleTime(1000, 1000); - ctf.get(); - - introPromiseFuture(); - introActor(); - // return; - - printf("Actor control flow tests: "); - actorTest1(true); - actorTest2(true); - actorTest3(true); - // if (g_network == g_simulator) - // g_simulator->run( actorTest4(true) ); - actorTest5(); - actorTest6(); - actorTest7(); - actorTest8(); - actorTest9(); - actorTest10(); - - printf("\n"); - - printf("Running actor fuzz tests:\n"); - // Only include this test outside of Windows because of MSVC compiler bug -#ifndef WIN32 - auto afResults = actorFuzzTests(); -#else - std::pair afResults(0, 0); -#endif - printf("Actor fuzz tests: %d/%d passed\n", afResults.first, afResults.second); - startt = timer(); - for (int i = 0; i < 1000000; i++) - deterministicRandom()->random01(); - endt = timer(); - printf("Random01: %0.2f M/sec\n", 1.0 / (endt - startt)); - - startt = timer(); - for (int i = 0; i < 1000000; i++) - Promise(); - endt = timer(); - printf("Promises: %0.2f M/sec\n", 1.0 / (endt - startt)); - - startt = timer(); - for (int i = 0; i < 1000000; i++) - Promise().send(Void()); - endt = timer(); - printf("Promises (with send): %0.2f M/sec\n", 1.0 / (endt - startt)); - - startt = timer(); - for (int i = 0; i < 1000000; i++) { - Promise p; - Future f = p.getFuture(); - p.send(Void()); - f.get(); - } - endt = timer(); - printf("Promise/Future/send roundtrip: %0.2f M/sec\n", 1.0 / (endt - startt)); - - Promise p; - - startt = timer(); - for (int i = 0; i < 1000000; i++) - p.getFuture(); - endt = timer(); - printf("Futures: %0.2f M/sec\n", 1.0 / (endt - startt)); - - startt = timer(); - for (int i = 0; i < 1000000; i++) - PromiseStream(); - endt = timer(); - printf("PromiseStreams: %0.2f M/sec\n", 1.0 / (endt - startt)); - - startt = timer(); - for (int i = 0; i < 1000000; i++) - PromiseStream().send(Void()); - endt = timer(); - printf("PromiseStreams (with send): %0.2f M/sec\n", 1.0 / (endt - startt)); - - startt = timer(); - for (int i = 0; i < 1000000; i++) { - PromiseStream p; - FutureStream f = p.getFuture(); - p.send(Void()); - f.pop(); - } - endt = timer(); - printf("PromiseStream/FutureStream/send/popBlocking roundtrip: %0.2f M/sec\n", 1.0 / (endt - startt)); - - startt = timer(); - { - PromiseStream ps; - for (int i = 0; i < 1000000; i++) { - ps.send(i); - } - } - endt = timer(); - printf("PromiseStream queued send: %0.2f M/sec\n", 1.0 / (endt - startt)); - - startt = timer(); - for (int i = 0; i < 1000000; i++) - cancellable(); - endt = timer(); - printf("Cancellations: %0.2f M/sec\n", 1.0 / (endt - startt)); - - startt = timer(); - for (int i = 0; i < 1000000; i++) - cancellable2(); - endt = timer(); - printf("Cancellations with catch: %0.2f M/sec\n", 1.0 / (endt - startt)); - - startt = timer(); - for (int i = 0; i < 1000000; i++) - simple(); - endt = timer(); - printf("Actor creation: %0.2f M/sec\n", 1.0 / (endt - startt)); - - startt = timer(); - for (int i = 0; i < 1000000; i++) - simpleWait(); - endt = timer(); - printf("With trivial wait: %0.2f M/sec\n", 1.0 / (endt - startt)); - - startt = timer(); - for (int i = 0; i < 1000000; i++) { - Promise p; - Future f = simpleRet(p.getFuture()); - p.send(i); - ASSERT(f.get() == i); - } - endt = timer(); - printf("Bounce int through actor: %0.2f M/sec\n", 1.0 / (endt - startt)); - - startt = timer(); - for (int i = 0; i < 1000000; i++) { - Promise p; - Future f = simpleRet(p.getFuture()); - Future g = simpleRet(p.getFuture()); - p.send(i); - ASSERT(f.get() == i); - ASSERT(g.get() == i); - } - endt = timer(); - printf("Bounce int through two actors in parallel: %0.2f M/sec\n", 1.0 / (endt - startt)); - - /*chainTest<1>(); - chainTest<4>(); - chainTest<16>(); - chainTest<64>(); - - startt = timer(); - for(int i=0; i<1000000; i++) - try { - throw success(); - } catch (Error&) { - } - endt = timer(); - printf("C++ exception: %0.2f M/sec\n", 1.0/(endt-startt));*/ - - arenaTest(); - - { - Promise a, b; - Future c = chooseTest(a.getFuture(), b.getFuture()); - a.send(1); - b.send(2); - std::cout << "c=" << c.get() << std::endl; - } - - { - Promise i; - Future d = addN<20>(i.getFuture()); - i.send(1.1); - std::cout << d.get() << std::endl; - } - - { - Promise i; - i.sendError(operation_failed()); - Future d = addN<20>(i.getFuture()); - if (d.isError() && d.getError().code() == error_code_operation_failed) - std::cout << "Error transmitted OK" << std::endl; - else - std::cout << "Error not transmitted!" << std::endl; - } - - /*{ - int na = Actor::allActors.size(); - PromiseStream t; - testStream(t.getFuture()); - if (Actor::allActors.size() != na+1) - std::cout << "Actor not created!" << std::endl; - t = PromiseStream(); - if (Actor::allActors.size() != na) - std::cout << "Actor not cleaned up!" << std::endl; - }*/ - - PromiseStream as; - Promise bs; - as.send(4); - Future sT = switchTest(as.getFuture(), bs.getFuture()); - as.send(5); - // sT = move(Future()); - as.send(6); - bs.send(10.1); - as.send(7); - - fastAllocTest(); - -#if FLOW_THREAD_SAFE - returnCancelRaceTest(); - threadSafetyTest(); - threadSafetyTest2(); -#else - printf("Thread safety disabled.\n"); -#endif -} - -/*ACTOR Future pingServer( FutureStream> requests, int rate ) { - state int count = 0; - loop { - Promise req = waitNext( requests ); - req.send( (++count)%rate != 0 ); - } -} - -ACTOR Future ping( PromiseStream> server ) { - state int count = 0; - loop { - bool result = wait( server.getReply() ); - - count++; - if (!result) - break; - } - return count; -} - -void pingtest() { - double start = timer(); - PromiseStream> serverInterface; - Future pS = pingServer( serverInterface.getFuture(), 5000000 ); - Future count = ping( serverInterface ); - double end = timer(); - std::cout << count.get() << " pings completed in " << (end-start) << " sec" << std::endl; -}*/ - -void copyTest() { - double start, elapsed; - - Arena arena; - StringRef s(new (arena) uint8_t[10 << 20], 10 << 20); - - { - start = timer(); - for (int i = 0; i < 100; i++) { - StringRef k = s; - (void)k; - } - elapsed = timer() - start; - - printf("StringRef->StringRef: %fs/GB\n", elapsed); - } - - { - start = timer(); - for (int i = 0; i < 100; i++) - Standalone a = s; - elapsed = timer() - start; - - printf("StringRef->Standalone: %fs/GB\n", elapsed); - } - - { - Standalone sa = s; - start = timer(); - for (int i = 0; i < 100; i++) - Standalone a = sa; - elapsed = timer() - start; - - printf("Standalone->Standalone: %fs/GB\n", elapsed); - } - - { - Standalone sa = s, sb; - start = timer(); - for (int i = 0; i < 50; i++) { - sb = std::move(sa); - sa = std::move(sb); - } - elapsed = timer() - start; - printf("move(Standalone)->Standalone: %fs/GB\n", elapsed); - } -} - -/*ACTOR void badTest( FutureStream is ) { - state PromiseStream js; - - loop choose { - when( int j = waitNext( js.getFuture() ) ) { - std::cout << "J" << j << std::endl; - } - when( int i = waitNext( is ) ) { - std::cout << "I" << i << std::endl; - js.send( i ); - std::cout << "-I" << i << std::endl; - } - } -} - -void dsltest() { - PromiseStream is; - badTest( is.getFuture() ); - is.send(1); - is.send(2); - is.send(3); - throw not_implemented(); -} -void pingtest() {}*/ diff --git a/fdbrpc/include/fdbrpc/DDSketch.h b/fdbrpc/include/fdbrpc/DDSketch.h index 481a8af40c1..0be9cea4d2a 100644 --- a/fdbrpc/include/fdbrpc/DDSketch.h +++ b/fdbrpc/include/fdbrpc/DDSketch.h @@ -112,6 +112,35 @@ class DDSketchBase { return *this; } + void addSamplePair(T sample, DDSketchBase& other) { + ASSERT(this != &other && fabs(errorGuarantee - other.errorGuarantee) < EPS && + buckets.size() == other.buckets.size()); + + if (!populationSize) + minValue = maxValue = sample; + if (!other.populationSize) + other.minValue = other.maxValue = sample; + + if (sample <= EPS) { + ++zeroPopulationSize; + ++other.zeroPopulationSize; + } else { + size_t index = static_cast(this)->getIndex(sample); + ASSERT(index < buckets.size()); + ++buckets[index]; + ++other.buckets[index]; + } + + ++populationSize; + ++other.populationSize; + sum += sample; + other.sum += sample; + maxValue = std::max(maxValue, sample); + other.maxValue = std::max(other.maxValue, sample); + minValue = std::min(minValue, sample); + other.minValue = std::min(other.minValue, sample); + } + double mean() const { if (populationSize == 0) return 0; diff --git a/fdbrpc/include/fdbrpc/FailureMonitor.h b/fdbrpc/include/fdbrpc/FailureMonitor.h index 1fde1ee0ac7..3e4f9e80252 100644 --- a/fdbrpc/include/fdbrpc/FailureMonitor.h +++ b/fdbrpc/include/fdbrpc/FailureMonitor.h @@ -168,8 +168,6 @@ class SimpleFailureMonitor : public IFailureMonitor { YieldedAsyncMap endpointKnownFailed; AsyncMap disconnectTriggers; std::unordered_map failedEndpoints; - - friend class OnStateChangedActorActor; }; #endif diff --git a/fdbrpc/include/fdbrpc/LoadBalance.actor.h b/fdbrpc/include/fdbrpc/LoadBalance.actor.h deleted file mode 100644 index 4931b3e76a0..00000000000 --- a/fdbrpc/include/fdbrpc/LoadBalance.actor.h +++ /dev/null @@ -1,765 +0,0 @@ -/* - * LoadBalance.actor.h - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -#pragma once - -// When actually compiled (NO_INTELLISENSE), include the generated version of this file. In intellisense use the source -// version. -#if defined(NO_INTELLISENSE) && !defined(FLOW_LOADBALANCE_ACTOR_G_H) -#define FLOW_LOADBALANCE_ACTOR_G_H -#include "fdbrpc/LoadBalance.actor.g.h" -#elif !defined(FLOW_LOADBALANCE_ACTOR_H) -#define FLOW_LOADBALANCE_ACTOR_H - -#include "flow/BooleanParam.h" -#include "flow/flow.h" -#include "flow/Error.h" -#include "flow/Knobs.h" - -#include "fdbrpc/FailureMonitor.h" -#include "fdbrpc/fdbrpc.h" -#include "fdbrpc/Locality.h" -#include "fdbrpc/QueueModel.h" -#include "fdbrpc/MultiInterface.h" -#include "flow/actorcompiler.h" // This must be the last #include. - -Future allAlternativesFailedDelay(Future okFuture); - -enum RequiredReplicas { BEST_EFFORT = -2, ALL_REPLICAS = -1 }; - -FDB_BOOLEAN_PARAM(CleanRequest); -FDB_BOOLEAN_PARAM(FutureVersion); -FDB_BOOLEAN_PARAM(MeasureLatency); - -struct ModelHolder : NonCopyable, public ReferenceCounted { - QueueModel* model; - bool released; - double startTime; - double delta; - uint64_t token; - - ModelHolder(QueueModel* model, uint64_t token) : model(model), released(false), startTime(now()), token(token) { - if (model) { - delta = model->addRequest(token); - } - } - - void release(CleanRequest clean, - FutureVersion futureVersion, - double penalty, - MeasureLatency measureLatency = MeasureLatency::True) { - if (model && !released) { - released = true; - double latency = (clean || measureLatency) ? now() - startTime : 0.0; - model->endRequest(token, latency, penalty, delta, clean, futureVersion); - } - } - - ~ModelHolder() { release(CleanRequest::False, FutureVersion::False, -1.0, MeasureLatency::False); } -}; - -// Subclasses must initialize all members in their default constructors -// Subclasses must serialize all members -struct LoadBalancedReply { - double penalty; - Optional error; - LoadBalancedReply() : penalty(1.0) {} -}; - -Optional getLoadBalancedReply(const LoadBalancedReply* reply); -Optional getLoadBalancedReply(const void*); - -// Storage-server-specific layers can specialize these hooks to attach request duplication or post-response comparison -// behavior without teaching the generic RPC load balancer about storage semantics. -template -struct LoadBalanceRequestHooks { - static_assert(!LoadBalanceHooksRequired::value, - "LoadBalanceRequestHooks specialization required for this queue model"); - - static void maybeDuplicate(RequestStream const* stream, - Request& request, - Model* model, - Future> response, - Reference> alternatives, - RequestStream Interface::* channel) {} - - static Future maybeCompare(Request& request, - Model* model, - RequestStream const* requestStream, - Future> response, - Reference> alternatives, - RequestStream Interface::* channel, - bool compareReplicas, - int requiredReplicas) { - return Void(); - } -}; - -FDB_BOOLEAN_PARAM(AtMostOnce); -FDB_BOOLEAN_PARAM(TriedAllOptions); - -// Stores state for a request made by the load balancer -template -struct RequestData : NonCopyable { - typedef ErrorOr Reply; - - Future response; - Reference modelHolder; - TriedAllOptions triedAllOptions{ false }; - RequestStream const* requestStream = nullptr; - - bool requestStarted = false; // true once the request has been sent to an alternative - bool requestProcessed = false; // true once a response has been received and handled by checkAndProcessResult - - bool compareReplicas = false; - Future comparisonResult; - - explicit RequestData(bool compareReplicas = false) : compareReplicas(compareReplicas) {} - - // Whether or not the response future is valid - // This is true once setupRequest is called, even though at that point the response is Never(). - bool isValid() { return response.isValid(); } - - Future maybeDoPostRequestComparison(Request& request, - Model* model, - Reference> alternatives, - RequestStream Interface::* channel, - int requiredReplicas) { - ASSERT(requestStream != nullptr); - return LoadBalanceRequestHooks::maybeCompare( - request, model, requestStream, response, alternatives, channel, compareReplicas, requiredReplicas); - } - - // Initializes the request state and starts it, possibly after a backoff delay - void startRequest( - double backoff, - TriedAllOptions triedAllOptions, - RequestStream const* stream, - Request& request, - Model* model, - Reference> alternatives, // alternatives and channel passed through to request hooks - RequestStream Interface::* channel) { - modelHolder = Reference(); - requestStream = stream; - requestStarted = false; - - if (backoff > 0) { - response = mapAsync(delay(backoff), [this, stream, &request, model, alternatives, channel](Void _) { - requestStarted = true; - modelHolder = makeReference(model, stream->getEndpoint().token.first()); - Future resp = stream->tryGetReply(request); - LoadBalanceRequestHooks::maybeDuplicate( - stream, request, model, resp, alternatives, channel); - return resp; - }); - } else { - requestStarted = true; - modelHolder = makeReference(model, stream->getEndpoint().token.first()); - response = stream->tryGetReply(request); - LoadBalanceRequestHooks::maybeDuplicate( - stream, request, model, response, alternatives, channel); - } - - requestProcessed = false; - this->triedAllOptions = triedAllOptions; - } - - // Implementation of the logic to handle a response. - // Checks the state of the response, updates the queue model, and returns one of the following outcomes: - // A return value of true means that the request completed successfully - // A return value of false means that the request failed but should be retried - // A return value with an error means that the error should be thrown back to original caller - static ErrorOr checkAndProcessResultImpl(Reply const& result, - Reference modelHolder, - AtMostOnce atMostOnce, - TriedAllOptions triedAllOptions) { - ASSERT(modelHolder); - - Optional loadBalancedReply; - if (!result.isError()) { - loadBalancedReply = getLoadBalancedReply(&result.get()); - } - - int errCode; - if (loadBalancedReply.present()) { - errCode = loadBalancedReply.get().error.present() ? loadBalancedReply.get().error.get().code() - : error_code_success; - } else { - errCode = result.isError() ? result.getError().code() : error_code_success; - } - - bool maybeDelivered = errCode == error_code_broken_promise || errCode == error_code_request_maybe_delivered; - bool receivedResponse = - loadBalancedReply.present() ? !loadBalancedReply.get().error.present() : result.present(); - receivedResponse = receivedResponse || (!maybeDelivered && errCode != error_code_process_behind); - FutureVersion futureVersion{ errCode == error_code_future_version || errCode == error_code_process_behind }; - - modelHolder->release(CleanRequest{ receivedResponse }, - futureVersion, - loadBalancedReply.present() ? loadBalancedReply.get().penalty : -1.0); - - if (errCode == error_code_server_overloaded) { - return false; - } - - if (loadBalancedReply.present() && !loadBalancedReply.get().error.present()) { - return true; - } - - if (!loadBalancedReply.present() && result.present()) { - return true; - } - - if (receivedResponse) { - return loadBalancedReply.present() ? loadBalancedReply.get().error.get() : result.getError(); - } - - if (atMostOnce && maybeDelivered) { - return request_maybe_delivered(); - } - - if (triedAllOptions && errCode == error_code_process_behind) { - return process_behind(); - } - - return false; - } - - // Checks the state of the response, updates the queue model, and returns one of the following outcomes: - // A return value of true means that the request completed successfully - // A return value of false means that the request failed but should be retried - // In the event of a non-retryable failure, an error is thrown indicating the failure - bool checkAndProcessResult(AtMostOnce atMostOnce) { - ASSERT(response.isReady()); - requestProcessed = true; - - ErrorOr outcome = - checkAndProcessResultImpl(response.get(), std::move(modelHolder), atMostOnce, triedAllOptions); - - if (outcome.isError()) { - throw outcome.getError(); - } else if (!outcome.get()) { - response = Future(); - } - - return outcome.get(); - } - - // Convert this request to a lagging request. Such a request is no longer being waited on, but it still needs to be - // processed so we can update the queue model. - void makeLaggingRequest() { - ASSERT(response.isValid()); - ASSERT(modelHolder); - ASSERT(modelHolder->model); - - QueueModel* model = modelHolder->model; - if (model->laggingRequestCount > FLOW_KNOBS->MAX_LAGGING_REQUESTS_OUTSTANDING || - model->laggingRequests.isReady()) { - model->laggingRequests.cancel(); - model->laggingRequestCount = 0; - model->addActor = PromiseStream>(); - model->laggingRequests = actorCollection(model->addActor.getFuture(), &model->laggingRequestCount); - } - - // We need to process the lagging request in order to update the queue model - Reference holderCapture = std::move(modelHolder); - auto triedAllOptionsCapture = triedAllOptions; - Future updateModel = map(response, [holderCapture, triedAllOptionsCapture](Reply result) { - checkAndProcessResultImpl(result, holderCapture, AtMostOnce::False, triedAllOptionsCapture); - return Void(); - }); - model->addActor.send(updateModel); - } - - ~RequestData() { - // If the request has been started but hasn't completed, mark it as a lagging request - if (requestStarted && !requestProcessed && modelHolder && modelHolder->model) { - makeLaggingRequest(); - } - } -}; - -// Try to get a reply from one of the alternatives until success, cancellation, or certain errors. -// Load balancing has a budget to race requests to a second alternative if the first request is slow. -// Tries to take into account failMon's information for load balancing and avoiding failed servers. -// If ALL the servers are failed and the list of servers is not fresh, throws an exception to let the caller refresh the -// list of servers. -// When model is set, load balance among alternatives in the same DC aims to balance request queue length on these -// interfaces. If too many interfaces in the same DC are bad, try remote interfaces. -// If compareReplicas is set, does a consistency check by fetching and comparing results from storage -// replicas (as many as specified by "requiredReplicas") and throws an exception if an inconsistency is found. -ACTOR template -Future loadBalanceImpl( - Reference> alternatives, - RequestStream Interface::* channel, - Request request = Request(), - TaskPriority taskID = TaskPriority::DefaultPromiseEndpoint, - // If true, throws request_maybe_delivered() instead of retrying automatically. - AtMostOnce atMostOnce = AtMostOnce::False, - Model* model = nullptr, - bool compareReplicas = false, - int requiredReplicas = 0) { - - state RequestData firstRequestData(compareReplicas); - state RequestData secondRequestData(compareReplicas); - - state Optional firstRequestEndpoint; - state Future secondDelay = Never(); - - state Promise requestFinished; - state double startTime = now(); - - state TriedAllOptions triedAllOptions = TriedAllOptions::False; - - setReplyPriority(request, taskID); - if (!alternatives) - return Never(); - - ASSERT(alternatives->size()); - - state int bestAlt = deterministicRandom()->randomInt(0, alternatives->countBest()); - state int nextAlt = deterministicRandom()->randomInt(0, std::max(alternatives->size() - 1, 1)); - if (nextAlt >= bestAlt) - nextAlt++; - - if (model) { - double bestMetric = 1e9; // Storage server with the least outstanding requests. - double nextMetric = 1e9; - double bestTime = 1e9; // The latency to the server with the least outstanding requests. - double nextTime = 1e9; - int badServers = 0; - - for (int i = 0; i < alternatives->size(); i++) { - // countBest(): the number of alternatives in the same locality (i.e., DC by default) as alternatives[0]. - // if the if-statement is correct, it won't try to send requests to the remote ones. - if (badServers < std::min(i, FLOW_KNOBS->LOAD_BALANCE_MAX_BAD_OPTIONS + 1) && - i == alternatives->countBest()) { - // When we have at least one healthy local server, and the bad - // server count is within "LOAD_BALANCE_MAX_BAD_OPTIONS". We - // do not need to consider any remote servers. - break; - } else if (badServers == alternatives->countBest() && i == badServers) { - TraceEvent("AllLocalAlternativesFailed") - .suppressFor(1.0) - .detail("Alternatives", alternatives->description()) - .detail("Total", alternatives->size()) - .detail("Best", alternatives->countBest()); - } - - RequestStream const* thisStream = &alternatives->get(i, channel); - if (!IFailureMonitor::failureMonitor().getState(thisStream->getEndpoint()).failed) { - auto const& qd = model->getMeasurement(thisStream->getEndpoint().token.first()); - if (now() > qd.failedUntil) { - double thisMetric = qd.smoothOutstanding.smoothTotal(); - double thisTime = qd.latency; - if (FLOW_KNOBS->LOAD_BALANCE_PENALTY_IS_BAD && qd.penalty > 1.001) { - // When a server wants to penalize itself (the default - // penalty value is 1.0), consider this server as bad. - // penalty is sent from server. - ++badServers; - } - - if (thisMetric < bestMetric) { - if (i != bestAlt) { - nextAlt = bestAlt; - nextMetric = bestMetric; - nextTime = bestTime; - } - bestAlt = i; - bestMetric = thisMetric; - bestTime = thisTime; - } else if (thisMetric < nextMetric) { - nextAlt = i; - nextMetric = thisMetric; - nextTime = thisTime; - } - } else { - ++badServers; - } - } else { - ++badServers; - } - } - if (nextMetric > 1e8) { - // If we still don't have a second best choice to issue request to, - // go through all the remote servers again, since we may have - // skipped it. - for (int i = alternatives->countBest(); i < alternatives->size(); i++) { - RequestStream const* thisStream = &alternatives->get(i, channel); - if (!IFailureMonitor::failureMonitor().getState(thisStream->getEndpoint()).failed) { - auto const& qd = model->getMeasurement(thisStream->getEndpoint().token.first()); - if (now() > qd.failedUntil) { - double thisMetric = qd.smoothOutstanding.smoothTotal(); - double thisTime = qd.latency; - - if (thisMetric < nextMetric) { - nextAlt = i; - nextMetric = thisMetric; - nextTime = thisTime; - } - } - } - } - } - - if (nextTime < 1e9) { - // Decide when to send the request to the second best choice. - if (bestTime > FLOW_KNOBS->INSTANT_SECOND_REQUEST_MULTIPLIER * - (model->secondMultiplier * (nextTime) + FLOW_KNOBS->BASE_SECOND_REQUEST_TIME)) { - secondDelay = Void(); - } else { - secondDelay = delay(model->secondMultiplier * nextTime + FLOW_KNOBS->BASE_SECOND_REQUEST_TIME); - } - } else { - secondDelay = Never(); - } - } - - state int startAlt = nextAlt; - state int startDistance = (bestAlt + alternatives->size() - startAlt) % alternatives->size(); - - state int numAttempts = 0; - state double backoff = 0; - // Issue requests to selected servers. - loop { - if (now() - startTime > (g_network->isSimulated() ? 30.0 : 600.0)) { - TraceEvent ev(g_network->isSimulated() ? SevWarn : SevWarnAlways, "LoadBalanceTooLong"); - ev.suppressFor(1.0); - ev.detail("Duration", now() - startTime); - ev.detail("NumAttempts", numAttempts); - ev.detail("Backoff", backoff); - ev.detail("TriedAllOptions", triedAllOptions); - if (ev.isEnabled()) { - ev.log(); - for (int alternativeNum = 0; alternativeNum < alternatives->size(); alternativeNum++) { - RequestStream const* thisStream = &alternatives->get(alternativeNum, channel); - TraceEvent(SevWarn, "LoadBalanceTooLongEndpoint") - .detail("Addr", thisStream->getEndpoint().getPrimaryAddress()) - .detail("Token", thisStream->getEndpoint().token) - .detail("Failed", IFailureMonitor::failureMonitor().getState(thisStream->getEndpoint()).failed); - } - } - } - - // Find an alternative, if any, that is not failed, starting with - // nextAlt. This logic matters only if model == nullptr. Otherwise, the - // bestAlt and nextAlt have been decided. - state RequestStream const* stream = nullptr; - state LBDistance::Type distance; - for (int alternativeNum = 0; alternativeNum < alternatives->size(); alternativeNum++) { - int useAlt = nextAlt; - if (nextAlt == startAlt) - useAlt = bestAlt; - else if ((nextAlt + alternatives->size() - startAlt) % alternatives->size() <= startDistance) - useAlt = (nextAlt + alternatives->size() - 1) % alternatives->size(); - - stream = &alternatives->get(useAlt, channel); - distance = alternatives->getDistance(useAlt); - if (!IFailureMonitor::failureMonitor().getState(stream->getEndpoint()).failed && - (!firstRequestEndpoint.present() || stream->getEndpoint().token.first() != firstRequestEndpoint.get())) - break; - nextAlt = (nextAlt + 1) % alternatives->size(); - if (nextAlt == startAlt) - triedAllOptions = TriedAllOptions::True; - stream = nullptr; - distance = LBDistance::DISTANT; - } - - if (!stream && !firstRequestData.isValid()) { - // Everything is down! Wait for someone to be up. - - std::vector> ok(alternatives->size()); - for (int i = 0; i < ok.size(); i++) { - ok[i] = IFailureMonitor::failureMonitor().onStateEqual(alternatives->get(i, channel).getEndpoint(), - FailureStatus(false)); - } - - Future okFuture = quorum(ok, 1); - - if (!alternatives->alwaysFresh()) { - // Making this SevWarn means a lot of clutter - if (now() - g_network->networkInfo.newestAlternativesFailure > 1 || - deterministicRandom()->random01() < 0.01) { - TraceEvent("AllAlternativesFailed").detail("Alternatives", alternatives->description()); - } - wait(allAlternativesFailedDelay(okFuture)); - } else { - wait(okFuture); - } - - numAttempts = 0; // now that we've got a server back, reset the backoff - } else if (!stream) { - // Only the first location is available. - wait(success(firstRequestData.response)); - if (firstRequestData.checkAndProcessResult(atMostOnce)) { - // Do consistency check, if requested. - wait(firstRequestData.maybeDoPostRequestComparison( - request, model, alternatives, channel, requiredReplicas)); - - ASSERT(firstRequestData.response.isReady()); - return firstRequestData.response.get().get(); - } - - firstRequestEndpoint = Optional(); - } else if (firstRequestData.isValid()) { - // Issue a second request, the first one is taking a long time. - if (distance == LBDistance::DISTANT) { - TraceEvent("LBDistant2nd") - .suppressFor(0.1) - .detail("Distance", (int)distance) - .detail("BackOff", backoff) - .detail("TriedAllOptions", triedAllOptions) - .detail("Alternatives", alternatives->description()) - .detail("Token", stream->getEndpoint().token) - .detail("Total", alternatives->size()) - .detail("Best", alternatives->countBest()) - .detail("Attempts", numAttempts); - } - secondRequestData.startRequest(backoff, triedAllOptions, stream, request, model, alternatives, channel); - - state bool firstRequestSuccessful = false; - state bool secondRequestSuccessful = false; - - loop choose { - when(wait(success(firstRequestData.response.isValid() ? firstRequestData.response : Never()))) { - if (firstRequestData.checkAndProcessResult(atMostOnce)) { - firstRequestSuccessful = true; - break; - } - - firstRequestEndpoint = Optional(); - } - when(wait(success(secondRequestData.response))) { - if (secondRequestData.checkAndProcessResult(atMostOnce)) { - secondRequestSuccessful = true; - } - - break; - } - } - - if (firstRequestSuccessful || secondRequestSuccessful) { - // Do consistency check, by comparing results from storage replicas, if requested. - state RequestData* requestData = - firstRequestSuccessful ? &firstRequestData : &secondRequestData; - wait( - requestData->maybeDoPostRequestComparison(request, model, alternatives, channel, requiredReplicas)); - - ASSERT(requestData->response.isReady()); - return requestData->response.get().get(); - } - - if (++numAttempts >= alternatives->size()) { - backoff = std::min( - FLOW_KNOBS->LOAD_BALANCE_MAX_BACKOFF, - std::max(FLOW_KNOBS->LOAD_BALANCE_START_BACKOFF, backoff * FLOW_KNOBS->LOAD_BALANCE_BACKOFF_RATE)); - } - } else { - // Issue a request, if it takes too long to get a reply, go around the loop - if (distance == LBDistance::DISTANT) { - TraceEvent("LBDistant") - .suppressFor(0.1) - .detail("Distance", (int)distance) - .detail("BackOff", backoff) - .detail("TriedAllOptions", triedAllOptions) - .detail("Alternatives", alternatives->description()) - .detail("Token", stream->getEndpoint().token) - .detail("Total", alternatives->size()) - .detail("Best", alternatives->countBest()) - .detail("Attempts", numAttempts); - } - firstRequestData.startRequest(backoff, triedAllOptions, stream, request, model, alternatives, channel); - firstRequestEndpoint = stream->getEndpoint().token.first(); - - loop { - choose { - when(wait(success(firstRequestData.response))) { - if (model) { - model->secondMultiplier = - std::max(model->secondMultiplier - FLOW_KNOBS->SECOND_REQUEST_MULTIPLIER_DECAY, 1.0); - model->secondBudget = - std::min(model->secondBudget + FLOW_KNOBS->SECOND_REQUEST_BUDGET_GROWTH, - FLOW_KNOBS->SECOND_REQUEST_MAX_BUDGET); - } - - if (firstRequestData.checkAndProcessResult(atMostOnce)) { - // Do consistency check, by comparing results from storage replicas, if requested. - wait(firstRequestData.maybeDoPostRequestComparison( - request, model, alternatives, channel, requiredReplicas)); - - ASSERT(firstRequestData.response.isReady()); - return firstRequestData.response.get().get(); - } - - firstRequestEndpoint = Optional(); - break; - } - when(wait(secondDelay)) { - secondDelay = Never(); - if (model && model->secondBudget >= 1.0) { - model->secondMultiplier += FLOW_KNOBS->SECOND_REQUEST_MULTIPLIER_GROWTH; - model->secondBudget -= 1.0; - break; - } - } - } - } - - if (++numAttempts >= alternatives->size()) { - backoff = std::min( - FLOW_KNOBS->LOAD_BALANCE_MAX_BACKOFF, - std::max(FLOW_KNOBS->LOAD_BALANCE_START_BACKOFF, backoff * FLOW_KNOBS->LOAD_BALANCE_BACKOFF_RATE)); - } - } - - nextAlt = (nextAlt + 1) % alternatives->size(); - if (nextAlt == startAlt) - triedAllOptions = TriedAllOptions::True; - resetReply(request, taskID); - secondDelay = Never(); - } -} - -// Subclasses must initialize all members in their default constructors -// Subclasses must serialize all members -struct BasicLoadBalancedReply { - int processBusyTime; - BasicLoadBalancedReply() : processBusyTime(0) {} -}; - -Optional getBasicLoadBalancedReply(const BasicLoadBalancedReply* reply); -Optional getBasicLoadBalancedReply(const void*); - -// A simpler version of LoadBalance that does not send second requests where the list of servers are always fresh -// -// If |alternativeChosen| is not null, then atMostOnce must be True, and if the returned future completes successfully -// then *alternativeChosen will be the alternative to which the message was sent. *alternativeChosen must outlive the -// returned future. -ACTOR template -Future basicLoadBalance(Reference> alternatives, - RequestStream Interface::* channel, - Request request = Request(), - TaskPriority taskID = TaskPriority::DefaultPromiseEndpoint, - AtMostOnce atMostOnce = AtMostOnce::False, - int* alternativeChosen = nullptr) { - ASSERT(alternativeChosen == nullptr || atMostOnce == AtMostOnce::True); - setReplyPriority(request, taskID); - if (!alternatives) - return Never(); - - ASSERT(alternatives->size() && alternatives->alwaysFresh()); - - state int bestAlt = alternatives->getBest(); - state int nextAlt = deterministicRandom()->randomInt(0, std::max(alternatives->size() - 1, 1)); - if (nextAlt >= bestAlt) - nextAlt++; - - state int startAlt = nextAlt; - state int startDistance = (bestAlt + alternatives->size() - startAlt) % alternatives->size(); - - state int numAttempts = 0; - state double backoff = 0; - state int useAlt; - loop { - // Find an alternative, if any, that is not failed, starting with nextAlt - state RequestStream const* stream = nullptr; - for (int alternativeNum = 0; alternativeNum < alternatives->size(); alternativeNum++) { - useAlt = nextAlt; - if (nextAlt == startAlt) - useAlt = bestAlt; - else if ((nextAlt + alternatives->size() - startAlt) % alternatives->size() <= startDistance) - useAlt = (nextAlt + alternatives->size() - 1) % alternatives->size(); - - stream = &alternatives->get(useAlt, channel); - if (alternativeChosen != nullptr) { - *alternativeChosen = useAlt; - } - if (!IFailureMonitor::failureMonitor().getState(stream->getEndpoint()).failed) - break; - nextAlt = (nextAlt + 1) % alternatives->size(); - stream = nullptr; - } - - if (!stream) { - // Everything is down! Wait for someone to be up. - - std::vector> ok(alternatives->size()); - for (int i = 0; i < ok.size(); i++) { - ok[i] = IFailureMonitor::failureMonitor().onStateEqual(alternatives->get(i, channel).getEndpoint(), - FailureStatus(false)); - } - wait(quorum(ok, 1)); - - numAttempts = 0; // now that we've got a server back, reset the backoff - } else { - if (backoff > 0.0) { - wait(delay(backoff)); - } - - ErrorOr result = wait(stream->tryGetReply(request)); - - if (result.present()) { - Optional loadBalancedReply = getBasicLoadBalancedReply(&result.get()); - if (loadBalancedReply.present()) { - alternatives->updateRecent(useAlt, loadBalancedReply.get().processBusyTime); - } - - return result.get(); - } - - if (result.getError().code() != error_code_broken_promise && - result.getError().code() != error_code_request_maybe_delivered) { - throw result.getError(); - } - - if (atMostOnce) { - throw request_maybe_delivered(); - } - - if (++numAttempts >= alternatives->size()) { - backoff = std::min( - FLOW_KNOBS->LOAD_BALANCE_MAX_BACKOFF, - std::max(FLOW_KNOBS->LOAD_BALANCE_START_BACKOFF, backoff * FLOW_KNOBS->LOAD_BALANCE_BACKOFF_RATE)); - } - } - - nextAlt = (nextAlt + 1) % alternatives->size(); - resetReply(request, taskID); - } -} - -// Actor-generated headers do not preserve default template arguments, so this wrapper preserves both the default model -// and the concrete model type supplied by a caller. -template -Future loadBalance(Reference> alternatives, - RequestStream Interface::* channel, - Request request = Request(), - TaskPriority taskID = TaskPriority::DefaultPromiseEndpoint, - AtMostOnce atMostOnce = AtMostOnce::False, - Model* model = nullptr, - bool compareReplicas = false, - int requiredReplicas = 0) { - return loadBalanceImpl( - alternatives, channel, std::move(request), taskID, atMostOnce, model, compareReplicas, requiredReplicas); -} - -#include "flow/unactorcompiler.h" - -#endif diff --git a/fdbrpc/include/fdbrpc/LoadBalance.h b/fdbrpc/include/fdbrpc/LoadBalance.h index 2b3d3a42bce..ec6c96c65c2 100644 --- a/fdbrpc/include/fdbrpc/LoadBalance.h +++ b/fdbrpc/include/fdbrpc/LoadBalance.h @@ -18,4 +18,760 @@ * limitations under the License. */ -#include "fdbrpc/LoadBalance.actor.h" \ No newline at end of file +#pragma once + +#include "flow/BooleanParam.h" +#include "flow/flow.h" +#include "flow/CoroUtils.h" +#include "flow/Error.h" +#include "flow/Knobs.h" + +#include "fdbrpc/FailureMonitor.h" +#include "fdbrpc/fdbrpc.h" +#include "fdbrpc/Locality.h" +#include "fdbrpc/QueueModel.h" +#include "fdbrpc/MultiInterface.h" + +Future allAlternativesFailedDelay(Future okFuture); + +enum RequiredReplicas { BEST_EFFORT = -2, ALL_REPLICAS = -1 }; + +FDB_BOOLEAN_PARAM(CleanRequest); +FDB_BOOLEAN_PARAM(FutureVersion); +FDB_BOOLEAN_PARAM(MeasureLatency); + +struct ModelHolder : NonCopyable, public ReferenceCounted { + QueueModel* model; + bool released; + double startTime; + double delta; + uint64_t token; + + ModelHolder(QueueModel* model, uint64_t token) : model(model), released(false), startTime(now()), token(token) { + if (model) { + delta = model->addRequest(token); + } + } + + void release(CleanRequest clean, + FutureVersion futureVersion, + double penalty, + MeasureLatency measureLatency = MeasureLatency::True) { + if (model && !released) { + released = true; + double latency = (clean || measureLatency) ? now() - startTime : 0.0; + model->endRequest(token, latency, penalty, delta, clean, futureVersion); + } + } + + ~ModelHolder() { release(CleanRequest::False, FutureVersion::False, -1.0, MeasureLatency::False); } +}; + +// Subclasses must initialize all members in their default constructors +// Subclasses must serialize all members +struct LoadBalancedReply { + double penalty; + Optional error; + LoadBalancedReply() : penalty(1.0) {} +}; + +Optional getLoadBalancedReply(const LoadBalancedReply* reply); +Optional getLoadBalancedReply(const void*); + +// Storage-server-specific layers can specialize these hooks to attach request duplication or post-response comparison +// behavior without teaching the generic RPC load balancer about storage semantics. +template +struct LoadBalanceRequestHooks { + static_assert(!LoadBalanceHooksRequired::value, + "LoadBalanceRequestHooks specialization required for this queue model"); + + static void maybeDuplicate(RequestStream const* stream, + Request& request, + Model* model, + Future> response, + Reference> alternatives, + RequestStream Interface::* channel) {} + + static Future maybeCompare(Request& request, + Model* model, + RequestStream const* requestStream, + Future> response, + Reference> alternatives, + RequestStream Interface::* channel, + bool compareReplicas, + int requiredReplicas) { + return Void(); + } +}; + +FDB_BOOLEAN_PARAM(AtMostOnce); +FDB_BOOLEAN_PARAM(TriedAllOptions); + +inline const Future& neverSecondRequest() { + // Constructing the Future initializes allocator TLS first, so this Future is destroyed before allocator teardown. + static thread_local const Future result = Never(); + return result; +} + +// Stores state for a request made by the load balancer +template +struct RequestData : NonCopyable { + using Reply = ErrorOr; + + Future response; + Reference modelHolder; + TriedAllOptions triedAllOptions{ false }; + RequestStream const* requestStream = nullptr; + + bool requestStarted = false; // true once the request has been sent to an alternative + bool requestProcessed = false; // true once a response has been received and handled by checkAndProcessResult + + bool compareReplicas = false; + Future comparisonResult; + + explicit RequestData(bool compareReplicas = false) : compareReplicas(compareReplicas) {} + + // Whether or not the response future is valid + // This is true once setupRequest is called, even though at that point the response is Never(). + bool isValid() { return response.isValid(); } + + Future maybeDoPostRequestComparison(Request& request, + Model* model, + Reference> alternatives, + RequestStream Interface::* channel, + int requiredReplicas) { + ASSERT(requestStream != nullptr); + return LoadBalanceRequestHooks::maybeCompare( + request, model, requestStream, response, alternatives, channel, compareReplicas, requiredReplicas); + } + + // Initializes the request state and starts it, possibly after a backoff delay + void startRequest( + double backoff, + TriedAllOptions triedAllOptions, + RequestStream const* stream, + Request& request, + Model* model, + Reference> alternatives, // alternatives and channel passed through to request hooks + RequestStream Interface::* channel) { + modelHolder = Reference(); + requestStream = stream; + requestStarted = false; + + if (backoff > 0) { + response = mapAsync(delay(backoff), [this, stream, &request, model, alternatives, channel](Void _) { + requestStarted = true; + modelHolder = makeReference(model, stream->getEndpoint().token.first()); + Future resp = stream->tryGetReply(request); + LoadBalanceRequestHooks::maybeDuplicate( + stream, request, model, resp, alternatives, channel); + return resp; + }); + } else { + requestStarted = true; + modelHolder = makeReference(model, stream->getEndpoint().token.first()); + response = stream->tryGetReply(request); + LoadBalanceRequestHooks::maybeDuplicate( + stream, request, model, response, alternatives, channel); + } + + requestProcessed = false; + this->triedAllOptions = triedAllOptions; + } + + // Implementation of the logic to handle a response. + // Checks the state of the response, updates the queue model, and returns one of the following outcomes: + // A return value of true means that the request completed successfully + // A return value of false means that the request failed but should be retried + // A return value with an error means that the error should be thrown back to original caller + static ErrorOr checkAndProcessResultImpl(Reply const& result, + Reference modelHolder, + AtMostOnce atMostOnce, + TriedAllOptions triedAllOptions) { + ASSERT(modelHolder); + + Optional loadBalancedReply; + if (!result.isError()) { + loadBalancedReply = getLoadBalancedReply(&result.get()); + } + + int errCode; + if (loadBalancedReply.present()) { + errCode = loadBalancedReply.get().error.present() ? loadBalancedReply.get().error.get().code() + : error_code_success; + } else { + errCode = result.isError() ? result.getError().code() : error_code_success; + } + + bool maybeDelivered = errCode == error_code_broken_promise || errCode == error_code_request_maybe_delivered; + bool receivedResponse = + loadBalancedReply.present() ? !loadBalancedReply.get().error.present() : result.present(); + receivedResponse = receivedResponse || (!maybeDelivered && errCode != error_code_process_behind); + FutureVersion futureVersion{ errCode == error_code_future_version || errCode == error_code_process_behind }; + + modelHolder->release(CleanRequest{ receivedResponse }, + futureVersion, + loadBalancedReply.present() ? loadBalancedReply.get().penalty : -1.0); + + if (errCode == error_code_server_overloaded) { + return false; + } + + if (loadBalancedReply.present() && !loadBalancedReply.get().error.present()) { + return true; + } + + if (!loadBalancedReply.present() && result.present()) { + return true; + } + + if (receivedResponse) { + return loadBalancedReply.present() ? loadBalancedReply.get().error.get() : result.getError(); + } + + if (atMostOnce && maybeDelivered) { + return request_maybe_delivered(); + } + + if (triedAllOptions && errCode == error_code_process_behind) { + return process_behind(); + } + + return false; + } + + // Checks the state of the response, updates the queue model, and returns one of the following outcomes: + // A return value of true means that the request completed successfully + // A return value of false means that the request failed but should be retried + // In the event of a non-retryable failure, an error is thrown indicating the failure + bool checkAndProcessResult(AtMostOnce atMostOnce) { + ASSERT(response.isReady()); + requestProcessed = true; + + ErrorOr outcome = + checkAndProcessResultImpl(response.get(), std::move(modelHolder), atMostOnce, triedAllOptions); + + if (outcome.isError()) { + throw outcome.getError(); + } else if (!outcome.get()) { + response = Future(); + } + + return outcome.get(); + } + + // Convert this request to a lagging request. Such a request is no longer being waited on, but it still needs to be + // processed so we can update the queue model. + void makeLaggingRequest() { + ASSERT(response.isValid()); + ASSERT(modelHolder); + ASSERT(modelHolder->model); + + QueueModel* model = modelHolder->model; + if (model->laggingRequestCount > FLOW_KNOBS->MAX_LAGGING_REQUESTS_OUTSTANDING || + model->laggingRequests.isReady()) { + model->laggingRequests.cancel(); + model->laggingRequestCount = 0; + model->addActor = PromiseStream>(); + model->laggingRequests = actorCollection(model->addActor.getFuture(), &model->laggingRequestCount); + } + + // We need to process the lagging request in order to update the queue model + Reference holderCapture = std::move(modelHolder); + auto triedAllOptionsCapture = triedAllOptions; + Future updateModel = map(response, [holderCapture, triedAllOptionsCapture](Reply result) { + checkAndProcessResultImpl(result, holderCapture, AtMostOnce::False, triedAllOptionsCapture); + return Void(); + }); + model->addActor.send(updateModel); + } + + ~RequestData() { + // If the request has been started but hasn't completed, mark it as a lagging request + if (requestStarted && !requestProcessed && modelHolder && modelHolder->model) { + makeLaggingRequest(); + } + } +}; + +// Try to get a reply from one of the alternatives until success, cancellation, or certain errors. +// Load balancing has a budget to race requests to a second alternative if the first request is slow. +// Tries to take into account failMon's information for load balancing and avoiding failed servers. +// If ALL the servers are failed and the list of servers is not fresh, throws an exception to let the caller refresh the +// list of servers. +// When model is set, load balance among alternatives in the same DC aims to balance request queue length on these +// interfaces. If too many interfaces in the same DC are bad, try remote interfaces. +// If compareReplicas is set, does a consistency check by fetching and comparing results from storage +// replicas (as many as specified by "requiredReplicas") and throws an exception if an inconsistency is found. +template +Future loadBalanceImpl(Reference> alternativesInput, + RequestStream Interface::* channel, + Request requestInput, + TaskPriority taskID, + AtMostOnce atMostOnce, + Model* model, + bool compareReplicas, + int requiredReplicas, + ExplicitVoid = {}) { + // Release owned request resources and server references before publishing completion. + Reference> alternatives(std::move(alternativesInput)); + Request request(std::move(requestInput)); + RequestData firstRequestData(compareReplicas); + RequestData secondRequestData(compareReplicas); + + Optional firstRequestEndpoint; + Future secondDelay = neverSecondRequest(); + + double startTime = now(); + + TriedAllOptions triedAllOptions = TriedAllOptions::False; + + ASSERT(alternatives->size()); + + int bestAlt = deterministicRandom()->randomInt(0, alternatives->countBest()); + int nextAlt = deterministicRandom()->randomInt(0, std::max(alternatives->size() - 1, 1)); + if (nextAlt >= bestAlt) + nextAlt++; + + if (model) { + double bestMetric = 1e9; // Storage server with the least outstanding requests. + double nextMetric = 1e9; + double bestTime = 1e9; // The latency to the server with the least outstanding requests. + double nextTime = 1e9; + int badServers = 0; + + for (int i = 0; i < alternatives->size(); i++) { + // countBest(): the number of alternatives in the same locality (i.e., DC by default) as alternatives[0]. + // if the if-statement is correct, it won't try to send requests to the remote ones. + if (badServers < std::min(i, FLOW_KNOBS->LOAD_BALANCE_MAX_BAD_OPTIONS + 1) && + i == alternatives->countBest()) { + // When we have at least one healthy local server, and the bad + // server count is within "LOAD_BALANCE_MAX_BAD_OPTIONS". We + // do not need to consider any remote servers. + break; + } else if (badServers == alternatives->countBest() && i == badServers) { + TraceEvent("AllLocalAlternativesFailed") + .suppressFor(1.0) + .detail("Alternatives", alternatives->description()) + .detail("Total", alternatives->size()) + .detail("Best", alternatives->countBest()); + } + + RequestStream const* thisStream = &alternatives->get(i, channel); + if (!IFailureMonitor::failureMonitor().getState(thisStream->getEndpoint()).failed) { + auto const& qd = model->getMeasurement(thisStream->getEndpoint().token.first()); + if (now() > qd.failedUntil) { + double thisMetric = qd.smoothOutstanding.smoothTotal(); + double thisTime = qd.latency; + if (FLOW_KNOBS->LOAD_BALANCE_PENALTY_IS_BAD && qd.penalty > 1.001) { + // When a server wants to penalize itself (the default + // penalty value is 1.0), consider this server as bad. + // penalty is sent from server. + ++badServers; + } + + if (thisMetric < bestMetric) { + if (i != bestAlt) { + nextAlt = bestAlt; + nextMetric = bestMetric; + nextTime = bestTime; + } + bestAlt = i; + bestMetric = thisMetric; + bestTime = thisTime; + } else if (thisMetric < nextMetric) { + nextAlt = i; + nextMetric = thisMetric; + nextTime = thisTime; + } + } else { + ++badServers; + } + } else { + ++badServers; + } + } + if (nextMetric > 1e8) { + // If we still don't have a second best choice to issue request to, + // go through all the remote servers again, since we may have + // skipped it. + for (int i = alternatives->countBest(); i < alternatives->size(); i++) { + RequestStream const* thisStream = &alternatives->get(i, channel); + if (!IFailureMonitor::failureMonitor().getState(thisStream->getEndpoint()).failed) { + auto const& qd = model->getMeasurement(thisStream->getEndpoint().token.first()); + if (now() > qd.failedUntil) { + double thisMetric = qd.smoothOutstanding.smoothTotal(); + double thisTime = qd.latency; + + if (thisMetric < nextMetric) { + nextAlt = i; + nextMetric = thisMetric; + nextTime = thisTime; + } + } + } + } + } + + if (nextTime < 1e9) { + // Decide when to send the request to the second best choice. + if (bestTime > FLOW_KNOBS->INSTANT_SECOND_REQUEST_MULTIPLIER * + (model->secondMultiplier * (nextTime) + FLOW_KNOBS->BASE_SECOND_REQUEST_TIME)) { + secondDelay = Void(); + } else { + secondDelay = delay(model->secondMultiplier * nextTime + FLOW_KNOBS->BASE_SECOND_REQUEST_TIME); + } + } else { + secondDelay = neverSecondRequest(); + } + } + + int startAlt = nextAlt; + int startDistance = (bestAlt + alternatives->size() - startAlt) % alternatives->size(); + + int numAttempts = 0; + double backoff = 0; + // Issue requests to selected servers. + while (true) { + if (now() - startTime > (g_network->isSimulated() ? 30.0 : 600.0)) { + TraceEvent ev(g_network->isSimulated() ? SevWarn : SevWarnAlways, "LoadBalanceTooLong"); + ev.suppressFor(1.0); + ev.detail("Duration", now() - startTime); + ev.detail("NumAttempts", numAttempts); + ev.detail("Backoff", backoff); + ev.detail("TriedAllOptions", triedAllOptions); + if (ev.isEnabled()) { + ev.log(); + for (int alternativeNum = 0; alternativeNum < alternatives->size(); alternativeNum++) { + RequestStream const* thisStream = &alternatives->get(alternativeNum, channel); + TraceEvent(SevWarn, "LoadBalanceTooLongEndpoint") + .detail("Addr", thisStream->getEndpoint().getPrimaryAddress()) + .detail("Token", thisStream->getEndpoint().token) + .detail("Failed", IFailureMonitor::failureMonitor().getState(thisStream->getEndpoint()).failed); + } + } + } + + // Find an alternative, if any, that is not failed, starting with + // nextAlt. This logic matters only if model == nullptr. Otherwise, the + // bestAlt and nextAlt have been decided. + RequestStream const* stream = nullptr; + LBDistance::Type distance; + for (int alternativeNum = 0; alternativeNum < alternatives->size(); alternativeNum++) { + int useAlt = nextAlt; + if (nextAlt == startAlt) + useAlt = bestAlt; + else if ((nextAlt + alternatives->size() - startAlt) % alternatives->size() <= startDistance) + useAlt = (nextAlt + alternatives->size() - 1) % alternatives->size(); + + stream = &alternatives->get(useAlt, channel); + distance = alternatives->getDistance(useAlt); + if (!IFailureMonitor::failureMonitor().getState(stream->getEndpoint()).failed && + (!firstRequestEndpoint.present() || stream->getEndpoint().token.first() != firstRequestEndpoint.get())) + break; + nextAlt = (nextAlt + 1) % alternatives->size(); + if (nextAlt == startAlt) + triedAllOptions = TriedAllOptions::True; + stream = nullptr; + distance = LBDistance::DISTANT; + } + + if (!stream && !firstRequestData.isValid()) { + // Everything is down! Wait for someone to be up. + + std::vector> ok(alternatives->size()); + for (int i = 0; i < ok.size(); i++) { + ok[i] = IFailureMonitor::failureMonitor().onStateEqual(alternatives->get(i, channel).getEndpoint(), + FailureStatus(false)); + } + + Future okFuture = quorum(ok, 1); + + if (!alternatives->alwaysFresh()) { + // Making this SevWarn means a lot of clutter + if (now() - g_network->networkInfo.newestAlternativesFailure > 1 || + deterministicRandom()->random01() < 0.01) { + TraceEvent("AllAlternativesFailed").detail("Alternatives", alternatives->description()); + } + co_await allAlternativesFailedDelay(okFuture); + } else { + co_await okFuture; + } + + numAttempts = 0; // now that we've got a server back, reset the backoff + } else if (!stream) { + // Only the first location is available. + co_await firstRequestData.response; + if (firstRequestData.checkAndProcessResult(atMostOnce)) { + // Do consistency check, if requested. + co_await firstRequestData.maybeDoPostRequestComparison( + request, model, alternatives, channel, requiredReplicas); + + ASSERT(firstRequestData.response.isReady()); + co_return firstRequestData.response.get().get(); + } + + firstRequestEndpoint = Optional(); + } else if (firstRequestData.isValid()) { + // Issue a second request, the first one is taking a long time. + if (distance == LBDistance::DISTANT) { + TraceEvent("LBDistant2nd") + .suppressFor(0.1) + .detail("Distance", (int)distance) + .detail("BackOff", backoff) + .detail("TriedAllOptions", triedAllOptions) + .detail("Alternatives", alternatives->description()) + .detail("Token", stream->getEndpoint().token) + .detail("Total", alternatives->size()) + .detail("Best", alternatives->countBest()) + .detail("Attempts", numAttempts); + } + secondRequestData.startRequest(backoff, triedAllOptions, stream, request, model, alternatives, channel); + + bool firstRequestSuccessful = false; + bool secondRequestSuccessful = false; + + while (true) { + auto res = co_await race(firstRequestData.response.isValid() ? firstRequestData.response : Never(), + secondRequestData.response); + if (res.index() == 0) { + if (firstRequestData.checkAndProcessResult(atMostOnce)) { + firstRequestSuccessful = true; + break; + } + + firstRequestEndpoint = Optional(); + } else { + if (secondRequestData.checkAndProcessResult(atMostOnce)) { + secondRequestSuccessful = true; + } + + break; + } + } + + if (firstRequestSuccessful || secondRequestSuccessful) { + // Do consistency check, by comparing results from storage replicas, if requested. + RequestData* requestData = + firstRequestSuccessful ? &firstRequestData : &secondRequestData; + co_await requestData->maybeDoPostRequestComparison( + request, model, alternatives, channel, requiredReplicas); + + ASSERT(requestData->response.isReady()); + co_return requestData->response.get().get(); + } + + if (++numAttempts >= alternatives->size()) { + backoff = std::min( + FLOW_KNOBS->LOAD_BALANCE_MAX_BACKOFF, + std::max(FLOW_KNOBS->LOAD_BALANCE_START_BACKOFF, backoff * FLOW_KNOBS->LOAD_BALANCE_BACKOFF_RATE)); + } + } else { + // Issue a request, if it takes too long to get a reply, go around the loop + if (distance == LBDistance::DISTANT) { + TraceEvent("LBDistant") + .suppressFor(0.1) + .detail("Distance", (int)distance) + .detail("BackOff", backoff) + .detail("TriedAllOptions", triedAllOptions) + .detail("Alternatives", alternatives->description()) + .detail("Token", stream->getEndpoint().token) + .detail("Total", alternatives->size()) + .detail("Best", alternatives->countBest()) + .detail("Attempts", numAttempts); + } + firstRequestData.startRequest(backoff, triedAllOptions, stream, request, model, alternatives, channel); + firstRequestEndpoint = stream->getEndpoint().token.first(); + + while (true) { + auto res = co_await race(firstRequestData.response, secondDelay); + if (res.index() == 0) { + if (model) { + model->secondMultiplier = + std::max(model->secondMultiplier - FLOW_KNOBS->SECOND_REQUEST_MULTIPLIER_DECAY, 1.0); + model->secondBudget = std::min(model->secondBudget + FLOW_KNOBS->SECOND_REQUEST_BUDGET_GROWTH, + FLOW_KNOBS->SECOND_REQUEST_MAX_BUDGET); + } + + if (firstRequestData.checkAndProcessResult(atMostOnce)) { + // Do consistency check, by comparing results from storage replicas, if requested. + co_await firstRequestData.maybeDoPostRequestComparison( + request, model, alternatives, channel, requiredReplicas); + + ASSERT(firstRequestData.response.isReady()); + co_return firstRequestData.response.get().get(); + } + + firstRequestEndpoint = Optional(); + break; + } else { + secondDelay = neverSecondRequest(); + if (model && model->secondBudget >= 1.0) { + model->secondMultiplier += FLOW_KNOBS->SECOND_REQUEST_MULTIPLIER_GROWTH; + model->secondBudget -= 1.0; + break; + } + } + } + + if (++numAttempts >= alternatives->size()) { + backoff = std::min( + FLOW_KNOBS->LOAD_BALANCE_MAX_BACKOFF, + std::max(FLOW_KNOBS->LOAD_BALANCE_START_BACKOFF, backoff * FLOW_KNOBS->LOAD_BALANCE_BACKOFF_RATE)); + } + } + + nextAlt = (nextAlt + 1) % alternatives->size(); + if (nextAlt == startAlt) + triedAllOptions = TriedAllOptions::True; + resetReply(request, taskID); + secondDelay = neverSecondRequest(); + } +} + +template +Future loadBalance(Reference> alternatives, + RequestStream Interface::* channel, + Request request = Request(), + TaskPriority taskID = TaskPriority::DefaultPromiseEndpoint, + // If true, throws request_maybe_delivered() instead of retrying automatically. + AtMostOnce atMostOnce = AtMostOnce::False, + Model* model = nullptr, + bool compareReplicas = false, + int requiredReplicas = 0) { + setReplyPriority(request, taskID); + if (!alternatives) { + return Never(); + } + return loadBalanceImpl(std::move(alternatives), + channel, + std::move(request), + taskID, + atMostOnce, + model, + compareReplicas, + requiredReplicas); +} + +// Subclasses must initialize all members in their default constructors +// Subclasses must serialize all members +struct BasicLoadBalancedReply { + int processBusyTime; + BasicLoadBalancedReply() : processBusyTime(0) {} +}; + +Optional getBasicLoadBalancedReply(const BasicLoadBalancedReply* reply); +Optional getBasicLoadBalancedReply(const void*); + +// A simpler version of LoadBalance that does not send second requests where the list of servers are always fresh +// +// If |alternativeChosen| is not null, then atMostOnce must be True, and if the returned future completes successfully +// then *alternativeChosen will be the alternative to which the message was sent. *alternativeChosen must outlive the +// returned future. +template +Future basicLoadBalanceImpl(Reference> alternativesInput, + RequestStream Interface::* channel, + Request requestInput, + TaskPriority taskID, + AtMostOnce atMostOnce, + int* alternativeChosen, + ExplicitVoid = {}) { + // Release owned request resources and server references before publishing completion. + Reference> alternatives(std::move(alternativesInput)); + Request request(std::move(requestInput)); + ASSERT(alternatives->size() && alternatives->alwaysFresh()); + + int bestAlt = alternatives->getBest(); + int nextAlt = deterministicRandom()->randomInt(0, std::max(alternatives->size() - 1, 1)); + if (nextAlt >= bestAlt) + nextAlt++; + + int startAlt = nextAlt; + int startDistance = (bestAlt + alternatives->size() - startAlt) % alternatives->size(); + + int numAttempts = 0; + double backoff = 0; + int useAlt{ 0 }; + while (true) { + // Find an alternative, if any, that is not failed, starting with nextAlt + RequestStream const* stream = nullptr; + for (int alternativeNum = 0; alternativeNum < alternatives->size(); alternativeNum++) { + useAlt = nextAlt; + if (nextAlt == startAlt) + useAlt = bestAlt; + else if ((nextAlt + alternatives->size() - startAlt) % alternatives->size() <= startDistance) + useAlt = (nextAlt + alternatives->size() - 1) % alternatives->size(); + + stream = &alternatives->get(useAlt, channel); + if (alternativeChosen != nullptr) { + *alternativeChosen = useAlt; + } + if (!IFailureMonitor::failureMonitor().getState(stream->getEndpoint()).failed) + break; + nextAlt = (nextAlt + 1) % alternatives->size(); + stream = nullptr; + } + + if (!stream) { + // Everything is down! Wait for someone to be up. + + std::vector> ok(alternatives->size()); + for (int i = 0; i < ok.size(); i++) { + ok[i] = IFailureMonitor::failureMonitor().onStateEqual(alternatives->get(i, channel).getEndpoint(), + FailureStatus(false)); + } + co_await quorum(ok, 1); + + numAttempts = 0; // now that we've got a server back, reset the backoff + } else { + if (backoff > 0.0) { + co_await delay(backoff); + } + + ErrorOr result = co_await stream->tryGetReply(request); + + if (result.present()) { + Optional loadBalancedReply = getBasicLoadBalancedReply(&result.get()); + if (loadBalancedReply.present()) { + alternatives->updateRecent(useAlt, loadBalancedReply.get().processBusyTime); + } + + co_return result.get(); + } + + if (result.getError().code() != error_code_broken_promise && + result.getError().code() != error_code_request_maybe_delivered) { + throw result.getError(); + } + + if (atMostOnce) { + throw request_maybe_delivered(); + } + + if (++numAttempts >= alternatives->size()) { + backoff = std::min( + FLOW_KNOBS->LOAD_BALANCE_MAX_BACKOFF, + std::max(FLOW_KNOBS->LOAD_BALANCE_START_BACKOFF, backoff * FLOW_KNOBS->LOAD_BALANCE_BACKOFF_RATE)); + } + } + + nextAlt = (nextAlt + 1) % alternatives->size(); + resetReply(request, taskID); + } +} + +template +Future basicLoadBalance(Reference> alternatives, + RequestStream Interface::* channel, + Request request = Request(), + TaskPriority taskID = TaskPriority::DefaultPromiseEndpoint, + AtMostOnce atMostOnce = AtMostOnce::False, + int* alternativeChosen = nullptr) { + ASSERT(alternativeChosen == nullptr || atMostOnce == AtMostOnce::True); + setReplyPriority(request, taskID); + if (!alternatives) { + return Never(); + } + return basicLoadBalanceImpl( + std::move(alternatives), channel, std::move(request), taskID, atMostOnce, alternativeChosen); +} diff --git a/fdbrpc/include/fdbrpc/Stats.h b/fdbrpc/include/fdbrpc/Stats.h index c8c9e1774c4..5fa942355d9 100644 --- a/fdbrpc/include/fdbrpc/Stats.h +++ b/fdbrpc/include/fdbrpc/Stats.h @@ -257,6 +257,7 @@ class LatencySample : public IMetric { double accuracy, bool skipTraceOnSilentInterval = false); void addMeasurement(double measurement); + void addMeasurementPair(double measurement, LatencySample& other); private: std::string name; diff --git a/fdbrpc/include/fdbrpc/fdbrpc.h b/fdbrpc/include/fdbrpc/fdbrpc.h index 0b8b42943c0..f5924bb8b44 100644 --- a/fdbrpc/include/fdbrpc/fdbrpc.h +++ b/fdbrpc/include/fdbrpc/fdbrpc.h @@ -86,6 +86,7 @@ class FlowReceiver : public NetworkMessageReceiver, public NonCopyable { } void setPeerCompatibilityPolicy(const PeerCompatibilityPolicy& policy) { peerCompatibilityPolicy_ = policy; } + bool hasPeerCompatibilityPolicy() const { return peerCompatibilityPolicy_.present(); } PeerCompatibilityPolicy peerCompatibilityPolicy() const override { return peerCompatibilityPolicy_.orDefault(NetworkMessageReceiver::peerCompatibilityPolicy()); @@ -120,7 +121,7 @@ struct NetSAV final : SAV, FlowReceiver, FastAllocated> { if (message.isError()) { SAV::sendErrorAndDelPromiseRef(message.getError()); } else { - SAV::sendAndDelPromiseRef(message.get().asUnderlyingType()); + SAV::sendAndDelPromiseRef(std::move(message.get().asUnderlyingType())); } } @@ -134,18 +135,16 @@ class ReplyPromise final : public ComposedIdentifier { void send(U&& value) const { sav->send(std::forward(value)); } - // Swift can't call method that takes in a universal references (U&&), - // so provide a callable `send` method that copies the value. - void sendCopy(const T& valueCopy) const SWIFT_NAME(send(_:)) { - sav->send(valueCopy); - } + // Swift can't call method that takes in a universal references (U&&), + // so provide a callable `send` method that copies the value. + void sendCopy(const T& valueCopy) const SWIFT_NAME(send(_:)) { sav->send(valueCopy); } template void sendError(const E& exc) const { sav->sendError(exc); } void send(Never) { sendError(never_reply()); } - // SWIFT: Convenience method, since there is also a Swift.Never, so Never() could be confusing + // SWIFT: Convenience method, since there is also a Swift.Never, so Never() could be confusing void sendNever() const { send(Never()); } Future getFuture() const { @@ -169,6 +168,18 @@ class ReplyPromise final : public ComposedIdentifier { const Endpoint& getEndpoint(TaskPriority taskID = TaskPriority::DefaultPromiseEndpoint) const { return sav->getEndpoint(taskID); } + void loadRemoteEndpoint(const Endpoint& endpoint) { + // Request deserialization first creates an unused local reply state. Reuse it when it has no observers, + // aliases, endpoint, or compatibility policy; all other states retain the original replacement semantics. + if (sav != nullptr && sav->canBeSet() && sav->getFutureReferenceCount() == 0 && + sav->getPromiseReferenceCount() == 1 && !sav->isLocalEndpoint() && !sav->isRemoteEndpoint() && + !sav->hasPeerCompatibilityPolicy()) { + sav->setRemoteEndpoint(endpoint, false); + } else { + *this = ReplyPromise(endpoint); + } + networkSender(Uncancellable(), getFuture(), &sav->getRawEndpoint()); + } void operator=(const ReplyPromise& rhs) { if (rhs.sav) @@ -214,8 +225,7 @@ void load(Ar& ar, ReplyPromise& value) { UID token; ar >> token; Endpoint endpoint = FlowTransport::transport().loadedEndpoint(token); - value = ReplyPromise(endpoint); - networkSender(Uncancellable(), value.getFuture(), endpoint); + value.loadRemoteEndpoint(endpoint); } template @@ -226,8 +236,7 @@ struct serializable_traits> : std::true_type { UID token; serializer(ar, token); auto endpoint = FlowTransport::transport().loadedEndpoint(token); - p = ReplyPromise(endpoint); - networkSender(Uncancellable(), p.getFuture(), endpoint); + p.loadRemoteEndpoint(endpoint); } else { const auto& ep = p.getEndpoint().token; serializer(ar, ep); @@ -478,7 +487,7 @@ class ReplyPromiseStream { value.sequence = queue->acknowledgements.sequence++; queue->acknowledgements.bytesSent += value.expectedSize(); FlowTransport::transport().sendUnreliable( - SerializeSource>>(value), getEndpoint(), false); + SerializeSource>>(value), getEndpoint(), false); } else { queue->send(std::forward(value)); } @@ -757,7 +766,7 @@ class RequestStream { template Future getReply(const X& value) const { // Ensure the same request isn't used multiple times - ASSERT(!getReplyPromise(value).getFuture().isReady()); + ASSERT(!getReplyPromise(value).isSet()); if (queue->isRemoteEndpoint()) { return sendCanceler(getReplyPromise(value), FlowTransport::transport().sendReliable(SerializeSource(value), getEndpoint()), @@ -802,7 +811,7 @@ class RequestStream { Reference peer = FlowTransport::transport().sendUnreliable(SerializeSource(value), getEndpoint(taskID), true); auto& p = getReplyPromise(value); - return waitValueOrSignal(p.getFuture(), disc, getEndpoint(taskID), p, peer); + return waitValueOrSignal(p.getFuture(), std::move(disc), getEndpoint(taskID), p, std::move(peer)); } send(value); auto& p = getReplyPromise(value); @@ -823,7 +832,7 @@ class RequestStream { Reference peer = FlowTransport::transport().sendUnreliable(SerializeSource(value), getEndpoint(), true); auto& p = getReplyPromise(value); - return waitValueOrSignal(p.getFuture(), disc, getEndpoint(), p, peer); + return waitValueOrSignal(p.getFuture(), std::move(disc), getEndpoint(), p, std::move(peer)); } else { send(value); auto& p = getReplyPromise(value); diff --git a/fdbrpc/include/fdbrpc/genericactors.h b/fdbrpc/include/fdbrpc/genericactors.h index f3da2254072..ea17bf1c074 100644 --- a/fdbrpc/include/fdbrpc/genericactors.h +++ b/fdbrpc/include/fdbrpc/genericactors.h @@ -312,9 +312,9 @@ Future incrementalBroadcastWithError(Future input, struct PeerHolder { Reference peer; - explicit PeerHolder(Reference peer) : peer(peer) { - if (peer) { - peer->outstandingReplies++; + explicit PeerHolder(Reference peer) : peer(std::move(peer)) { + if (this->peer) { + this->peer->outstandingReplies++; } } ~PeerHolder() { @@ -336,12 +336,13 @@ Future endStreamOnDisconnect(Uncancellable, ReplyPromiseStream stream, Endpoint endpoint, Reference peer = Reference()) { - PeerHolder holder = PeerHolder(peer); + PeerHolder holder(std::move(peer)); stream.setRequestStreamEndpoint(endpoint); Error err; try { - auto res = co_await race( - signal, peer.isValid() ? peer->disconnect.getFuture() : Never(), stream.getErrorFutureAndDelPromiseRef()); + auto res = co_await race(signal, + holder.peer.isValid() ? holder.peer->disconnect.getFuture() : Never(), + stream.getErrorFutureAndDelPromiseRef()); if (res.index() == 0) { stream.sendError(connection_failed()); } else if (res.index() == 1) { @@ -368,10 +369,11 @@ Future> waitValueOrSignal(Future value, Endpoint endpoint, ReplyPromise holdme = ReplyPromise(), Reference peer = Reference()) { - PeerHolder holder = PeerHolder(peer); + PeerHolder holder(std::move(peer)); while (true) { try { - auto res = co_await race(value, signal, peer.isValid() ? peer->disconnect.getFuture() : Never()); + auto res = + co_await race(value, signal, holder.peer.isValid() ? holder.peer->disconnect.getFuture() : Never()); if (res.index() == 0) { X x = std::get<0>(std::move(res)); diff --git a/fdbrpc/include/fdbrpc/grpc/AsyncGrpcClient.h b/fdbrpc/include/fdbrpc/grpc/AsyncGrpcClient.h index 140a19cf1d0..8ba07e79fb8 100644 --- a/fdbrpc/include/fdbrpc/grpc/AsyncGrpcClient.h +++ b/fdbrpc/include/fdbrpc/grpc/AsyncGrpcClient.h @@ -23,7 +23,6 @@ #define FDBRPC_FLOW_ASYNC_GRPC_CLIENT_H #include -#undef loop #include #include "flow/flow.h" @@ -47,7 +46,6 @@ class AsyncGrpcClient { public: using Rpc = typename ServiceType::Stub; - // Isn't necessary unless initialized mid-block using Flow actor compiler. AsyncGrpcClient() = default; AsyncGrpcClient(const std::string& endpoint, std::shared_ptr pool) diff --git a/fdbrpc/include/fdbrpc/networksender.h b/fdbrpc/include/fdbrpc/networksender.h index 9bd62d4a5d2..5d3c9988267 100644 --- a/fdbrpc/include/fdbrpc/networksender.h +++ b/fdbrpc/include/fdbrpc/networksender.h @@ -25,21 +25,44 @@ #include "fdbrpc/FlowTransport.h" #include "flow/flow.h" +template +struct NetworkSenderTable { + using type = EnsureTableRef; +}; + +template +struct NetworkSenderTable> { + using type = EnsureTable>; +}; + +template +using NetworkSenderTableT = typename NetworkSenderTable::type; + // Used by FlowTransport to serialize the response to a ReplyPromise across the network. template -Future networkSender(Uncancellable, Future input, Endpoint endpoint, ExplicitVoid = {}) { +coro::DetachedCoroutine networkSender(Uncancellable, Future input, const Endpoint* endpoint) { + // Error replies can also throw, so the reporting boundary must cover both send paths. try { - T value = co_await input; - FlowTransport::transport().sendUnreliable(SerializeSource>>(value), endpoint, false); - } catch (Error& err) { - // if (err.code() == error_code_broken_promise) return; - if (err.code() == error_code_never_reply) { - co_return Void(); + try { + co_await input; + const T& value = input.get(); + FlowTransport::transport().sendUnreliable( + SerializeSource>>(value), *endpoint, false); + } catch (Error& err) { + // if (err.code() == error_code_broken_promise) return; + if (err.code() == error_code_never_reply) { + co_return; + } + ASSERT(err.code() != error_code_actor_cancelled); + FlowTransport::transport().sendUnreliable( + SerializeSource>>(err), *endpoint, false); } - ASSERT(err.code() != error_code_actor_cancelled); - FlowTransport::transport().sendUnreliable(SerializeSource>>(err), endpoint, false); + } catch (const Error&) { + // There is no consumer for errors raised while sending a reply. + } catch (...) { + (void)unknown_error(); } - co_return Void(); + co_return; } #endif diff --git a/fdbrpc/sim2.cpp b/fdbrpc/sim2.cpp index 3ab41e5d796..951e021fb39 100644 --- a/fdbrpc/sim2.cpp +++ b/fdbrpc/sim2.cpp @@ -767,8 +767,8 @@ class SimpleFile : public IAsyncFile, public ReferenceCounted { throw io_error(); } - unsigned int read_bytes = 0; - if ((read_bytes = _read(self->h, data, (unsigned int)length)) == -1) { + unsigned int read_bytes = _read(self->h, data, (unsigned int)length); + if (read_bytes == -1) { TraceEvent(SevWarn, "SimpleFileIOError").detail("Location", 2); throw io_error(); } @@ -814,8 +814,8 @@ class SimpleFile : public IAsyncFile, public ReferenceCounted { throw io_error(); } - unsigned int write_bytes = 0; - if ((write_bytes = _write(self->h, (void*)data.begin(), data.size())) == -1) { + unsigned int write_bytes = _write(self->h, (void*)data.begin(), data.size()); + if (write_bytes == -1) { TraceEvent(SevWarn, "SimpleFileIOError").detail("Location", 4); throw io_error(); } @@ -1089,7 +1089,7 @@ class Sim2 final : public ISimulator, public INetworkConnections { return delay(getCurrentProcess()->rebooting ? 0 : .001, taskID) || checkShutdown(this, taskID); } setCurrentTask(taskID); - return Void(); + return readyYield; } bool check_yield(TaskPriority taskID) override { if (yielded) @@ -2402,6 +2402,7 @@ class Sim2 final : public ISimulator, public INetworkConnections { // Whether or not yield has returned true during the current iteration of the run loop bool yielded; int yield_limit; // how many more times yield may return false before next returning true + Future readyYield = Void(); bool printSimTime; private: diff --git a/fdbrpc/tests/fdbrpc_bench.cpp b/fdbrpc/tests/fdbrpc_bench.cpp index 0edd9121c94..382098f7b4a 100644 --- a/fdbrpc/tests/fdbrpc_bench.cpp +++ b/fdbrpc/tests/fdbrpc_bench.cpp @@ -29,6 +29,9 @@ namespace fdbrpc_bench { NetworkAddress serverAddress; +TaskPriority echoEndpointPriority = TaskPriority::DefaultEndpoint; + +constexpr int MAX_CLIENT_CONCURRENCY = 4096; enum FdbRpcBenchWellKnownEndpoints { WLTOKEN_ECHO_SERVER = WLTOKEN_FIRST_AVAILABLE, @@ -155,7 +158,10 @@ class EchoServer { StatCounter counter; public: - EchoServer() { interf.getInterface.makeWellKnownEndpoint(WLTOKEN_ECHO_SERVER, TaskPriority::DefaultEndpoint); } + EchoServer() { + interf.getInterface.makeWellKnownEndpoint(WLTOKEN_ECHO_SERVER, TaskPriority::DefaultEndpoint); + interf.echo.getEndpoint(echoEndpointPriority); + } Future run() { co_await race(serveGetInterfaceReqs(), serveEchoReqs(), printThroughput()); } }; @@ -224,7 +230,9 @@ int main(int argc, char* argv[]) { desc.add_options() ("help,h","show help message") ("mode,m", po::value(), "process mode [server/client]") - ("payload_size,s", po::value(), "size of payload sent by client (bytes)"); + ("payload_size,s", po::value(), "size of payload sent by client (bytes)") + ("endpoint_priority", po::value()->default_value("default"), "server echo endpoint priority [default/loadbalanced]") + ("concurrency,c", po::value()->default_value(1), "number of client actors [1/4096]"); // clang-format on po::variables_map vm; @@ -244,7 +252,13 @@ int main(int argc, char* argv[]) { } auto mode = vm["mode"].as(); - if ((mode != "client" && mode != "server") || (mode == "server" && vm.count("payload_size") > 0)) { + auto endpointPriority = vm["endpoint_priority"].as(); + auto concurrency = vm["concurrency"].as(); + if ((mode != "client" && mode != "server") || + (endpointPriority != "default" && endpointPriority != "loadbalanced") || concurrency < 1 || + concurrency > MAX_CLIENT_CONCURRENCY || (vm.count("payload_size") > 0 && vm["payload_size"].as() < 0) || + (mode == "server" && (vm.count("payload_size") > 0 || concurrency != 1)) || + (mode == "client" && endpointPriority != "default")) { std::cerr << errMsg << desc << std::endl; return -1; } @@ -252,11 +266,13 @@ int main(int argc, char* argv[]) { if (vm.count("payload_size") > 0) { payload_size_bytes = vm["payload_size"].as(); } + echoEndpointPriority = + endpointPriority == "loadbalanced" ? TaskPriority::LoadBalancedEndpoint : TaskPriority::DefaultEndpoint; bool isServer = (mode == "server"); std::vector()>> toRun; auto actor = actors.find(mode); - toRun.push_back(actor->second); + toRun.resize(isServer ? 1 : concurrency, actor->second); platformInit(); g_network = newNet2(TLSConfig(), false, true); diff --git a/fdbserver/CMakeLists.txt b/fdbserver/CMakeLists.txt index dbb0aa45d1b..13df6150244 100644 --- a/fdbserver/CMakeLists.txt +++ b/fdbserver/CMakeLists.txt @@ -87,7 +87,7 @@ if (WITH_SWIFT) ) # TODO: the TBD validation skip is because of swift_job_run_generic, though it seems weird why we need to do that? - target_compile_options(fdbserver_swift PRIVATE "$<$:SHELL:-Xcc -std=c++20 -Xfrontend -validate-tbd-against-ir=none -Xcc -DNO_INTELLISENSE -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/flow/include/headeroverlay.yaml -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/fdbclient/include/headeroverlay.yaml -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/fdbserver/include/headeroverlay.yaml>") + target_compile_options(fdbserver_swift PRIVATE "$<$:SHELL:-Xcc -std=c++20 -Xfrontend -validate-tbd-against-ir=none -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/flow/include/headeroverlay.yaml -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/fdbclient/include/headeroverlay.yaml -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/fdbserver/include/headeroverlay.yaml>") # Ensure that C++ code in fdbserver can import Swift using a compatibility header. include(SwiftToCXXInterop) @@ -100,7 +100,7 @@ if (WITH_SWIFT) "${CMAKE_CURRENT_SOURCE_DIR}/swift_fdbserver_cxx_swift_value_conformance.swift" FLAGS -Xcc -fmodules-cache-path=${CLANG_MODULE_CACHE_PATH} - -Xcc -std=c++20 -Xcc -DNO_INTELLISENSE -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/flow/include/headeroverlay.yaml -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/fdbclient/include/headeroverlay.yaml -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/fdbserver/include/headeroverlay.yaml + -Xcc -std=c++20 -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/flow/include/headeroverlay.yaml -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/fdbclient/include/headeroverlay.yaml -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/fdbserver/include/headeroverlay.yaml # Important: This is needed to avoid including header that depends on this generated header. -Xcc -DFDBSERVER_FORWARD_DECLARE_SWIFT_APIS -Xcc -DFOUNDATIONDB_FDBSERVER_STREAM_SUPPORT_H @@ -112,20 +112,18 @@ if (WITH_SWIFT) "${CMAKE_CURRENT_BINARY_DIR}/include/SwiftModules/FDBServer" FLAGS -Xcc -fmodules-cache-path=${CLANG_MODULE_CACHE_PATH} - -Xcc -std=c++20 -Xcc -DNO_INTELLISENSE -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/flow/include/headeroverlay.yaml -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/fdbclient/include/headeroverlay.yaml -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/fdbserver/include/headeroverlay.yaml + -Xcc -std=c++20 -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/flow/include/headeroverlay.yaml -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/fdbclient/include/headeroverlay.yaml -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/fdbserver/include/headeroverlay.yaml # Important: This is needed to avoid including the generated header while generating it. -DFDBSERVER_FORWARD_DECLARE_SWIFT_APIS -Xcc -DFDBSERVER_FORWARD_DECLARE_SWIFT_APIS ) - add_dependencies(fdbserver_swift_to_cxx_conformance_header flow_swift flow_swift_header flow_actors fdbclient_actors fdbrpc_actors fdbserver_actors fdboptions) + add_dependencies(fdbserver_swift_to_cxx_conformance_header flow_swift flow_swift_header fdboptions) add_dependencies(fdbserver_swift_header fdbserver_swift_to_cxx_conformance_header) add_dependencies(fdbserver_swift fdbserver_swift_header) add_dependencies(fdbserver_swift flow_swift) - add_dependencies(fdbserver_swift flow_actors) add_dependencies(fdbserver_swift fdbclient_swift) - add_dependencies(fdbserver_swift fdbserver_actors) # This does not work! (see rdar://99107402) # target_link_libraries(flow PRIVATE flow_swift) add_dependencies(fdbserver fdbserver_swift) diff --git a/fdbserver/SimulatedCluster.cpp b/fdbserver/SimulatedCluster.cpp index da6f46de8a0..0402d1fabb8 100644 --- a/fdbserver/SimulatedCluster.cpp +++ b/fdbserver/SimulatedCluster.cpp @@ -481,6 +481,8 @@ class TestConfig : public BasicTestConfig { Optional generateFearless, buggify, faultInjection; Optional config; Optional remoteConfig; + // Satellite redundancy mode override applied to all regions (e.g. "one_satellite_single") + Optional satelliteRedundancyMode; bool randomlyRenameZoneId = false; bool simHTTPServerEnabled = true; @@ -548,6 +550,7 @@ class TestConfig : public BasicTestConfig { .add("storageEngineType", &storageEngineType) .add("config", &config) .add("remoteConfig", &remoteConfig) + .add("satelliteRedundancyMode", &satelliteRedundancyMode) .add("buggify", &buggify) .add("faultInjection", &faultInjection) .add("StderrSeverity", &stderrSeverity) @@ -1896,7 +1899,10 @@ void SimulationConfig::setRegions(const TestConfig& testConfig) { bool needsRemote = generateFearless; if (generateFearless) { - if (datacenters > 4) { + std::string satelliteRedundancyModeStr; + if (testConfig.satelliteRedundancyMode.present()) { + satelliteRedundancyModeStr = testConfig.satelliteRedundancyMode.get(); + } else if (datacenters > 4) { // FIXME: we cannot use one satellite replication with more than one satellite per region because // canKillProcesses does not respect usable_dcs int satellite_replication_type = deterministicRandom()->randomInt(0, 3); @@ -1912,14 +1918,12 @@ void SimulationConfig::setRegions(const TestConfig& testConfig) { } case 1: { CODE_PROBE(true, "Simulated cluster using two satellite fast redundancy mode"); - primaryObj["satellite_redundancy_mode"] = "two_satellite_fast"; - remoteObj["satellite_redundancy_mode"] = "two_satellite_fast"; + satelliteRedundancyModeStr = "two_satellite_fast"; break; } case 2: { CODE_PROBE(true, "Simulated cluster using two satellite safe redundancy mode"); - primaryObj["satellite_redundancy_mode"] = "two_satellite_safe"; - remoteObj["satellite_redundancy_mode"] = "two_satellite_safe"; + satelliteRedundancyModeStr = "two_satellite_safe"; break; } default: @@ -1939,20 +1943,17 @@ void SimulationConfig::setRegions(const TestConfig& testConfig) { } case 2: { CODE_PROBE(true, "Simulated cluster using single satellite redundancy mode"); - primaryObj["satellite_redundancy_mode"] = "one_satellite_single"; - remoteObj["satellite_redundancy_mode"] = "one_satellite_single"; + satelliteRedundancyModeStr = "one_satellite_single"; break; } case 3: { CODE_PROBE(true, "Simulated cluster using double satellite redundancy mode"); - primaryObj["satellite_redundancy_mode"] = "one_satellite_double"; - remoteObj["satellite_redundancy_mode"] = "one_satellite_double"; + satelliteRedundancyModeStr = "one_satellite_double"; break; } case 4: { CODE_PROBE(true, "Simulated cluster using triple satellite redundancy mode"); - primaryObj["satellite_redundancy_mode"] = "one_satellite_triple"; - remoteObj["satellite_redundancy_mode"] = "one_satellite_triple"; + satelliteRedundancyModeStr = "one_satellite_triple"; break; } default: @@ -1960,6 +1961,11 @@ void SimulationConfig::setRegions(const TestConfig& testConfig) { } } + if (!satelliteRedundancyModeStr.empty()) { + primaryObj["satellite_redundancy_mode"] = satelliteRedundancyModeStr; + remoteObj["satellite_redundancy_mode"] = satelliteRedundancyModeStr; + } + // Calculate the maximum satellite_logs we can support based on available machines bool useNormalDCsAsSatellites = datacenters > 4 && testConfig.minimumRegions < 2 && deterministicRandom()->random01() < 0.3; diff --git a/fdbserver/backupworker/RangePartitionedBackupWorker.cpp b/fdbserver/backupworker/RangePartitionedBackupWorker.cpp index feb5b94b384..344ba33e1a6 100644 --- a/fdbserver/backupworker/RangePartitionedBackupWorker.cpp +++ b/fdbserver/backupworker/RangePartitionedBackupWorker.cpp @@ -239,6 +239,7 @@ struct RangePartitionedBackupData { Future waitAllBackupsReady() { std::vector> all; + all.reserve(backups.size()); for (auto& [uid, info] : backups) { all.push_back(info.waitBackupReady()); } @@ -909,6 +910,7 @@ Future saveMutationsToFile(RangePartitionedBackupData* self, Version lastV // Finish files // TODO akanksha: Add FileLevel checksum. std::vector> finished; + finished.reserve(activeFiles.size()); for (auto& lf : activeFiles) { finished.push_back(lf.file->finish()); } diff --git a/fdbserver/clustercontroller/ClusterController.cpp b/fdbserver/clustercontroller/ClusterController.cpp index c30246b1754..be057d4d28b 100644 --- a/fdbserver/clustercontroller/ClusterController.cpp +++ b/fdbserver/clustercontroller/ClusterController.cpp @@ -581,6 +581,7 @@ Future monitorAndRecruitLogRouters(ClusterControllerData* self) { Future> monitorCDCProxies(std::vector const& cdcProxies) { std::vector> failures; + failures.reserve(cdcProxies.size()); for (const auto& proxy : cdcProxies) { failures.push_back( waitFailureClient(proxy.waitFailure, diff --git a/fdbserver/clustercontroller/ClusterController.h b/fdbserver/clustercontroller/ClusterController.h index ac032aa54d2..a1950b7166d 100644 --- a/fdbserver/clustercontroller/ClusterController.h +++ b/fdbserver/clustercontroller/ClusterController.h @@ -3200,12 +3200,14 @@ class ClusterControllerData { // Sort degraded peers based on the number of workers complaining about it. std::vector> count2DegradedPeer; + count2DegradedPeer.reserve(degradedLinkDst2Src.size()); for (const auto& [degradedPeer, complainers] : degradedLinkDst2Src) { count2DegradedPeer.push_back({ complainers.size(), degradedPeer }); } std::sort(count2DegradedPeer.begin(), count2DegradedPeer.end(), deterministicDecendingOrder); std::vector> count2DisconnectedPeer; + count2DisconnectedPeer.reserve(disconnectedLinkDst2Src.size()); for (const auto& [disconnectedPeer, complainers] : disconnectedLinkDst2Src) { count2DisconnectedPeer.push_back({ complainers.size(), disconnectedPeer }); } diff --git a/fdbserver/clustercontroller/ClusterRecovery.cpp b/fdbserver/clustercontroller/ClusterRecovery.cpp index fc30414f4b7..ea7cc3883d3 100644 --- a/fdbserver/clustercontroller/ClusterRecovery.cpp +++ b/fdbserver/clustercontroller/ClusterRecovery.cpp @@ -506,14 +506,66 @@ Future rejoinRequestHandler(Reference self) { } } +// Snapshot of the remote-region stall state shared between trackTlogRecovery (which updates it on +// every core-state change) and remoteRegionStallEventRefresher (which re-emits the trackLatest event +// on a fixed cadence while the stall persists). +struct RemoteRegionStallState : ReferenceCounted { + bool remoteRegionLogsMissing = false; + bool allLogs = false; + int oldTLogDataSize = 0; + int usableRegions = 1; + double missingSince = 0; // now() at stall start; valid when remoteRegionLogsMissing + std::string eventName; // stall event name, resolved once by trackTlogRecovery +}; + +// Emits the remote-region stall trackLatest event from the given snapshot. Called on core-state +// changes (where the signal can also be cleared) and on the refresher's cadence. +static void traceRemoteRegionStall(Reference self, + const RemoteRegionStallState* state, + bool remoteRegionLogsMissing, + double stallSeconds) { + TraceEvent(state->eventName.c_str(), self->dbgid) + .detail("RemoteRegionLogsMissing", remoteRegionLogsMissing) + .detail("AllLogs", state->allLogs) + .detail("OldTLogDataSize", state->oldTLogDataSize) + .detail("UsableRegions", state->usableRegions) + .detail("StallSeconds", stallSeconds) + .trackLatest(self->clusterRecoveryRemoteRegionStallEventHolder->trackingKey); +} + +// Re-emits the remote-region stall trackLatest event on a fixed cadence while the stall persists, +// keeping StallSeconds advancing on an otherwise idle cluster. Started lazily by trackTlogRecovery +// on the first observed stall; returns once the shared state says the stall is over. +Future remoteRegionStallEventRefresher(Reference self, + Reference state) { + while (state->remoteRegionLogsMissing) { + co_await delay(SERVER_KNOBS->DEGRADED_MULTI_REGION_REFRESH_SECONDS); + // trackTlogRecovery may have cleared the stall while the delay was pending. + if (!state->remoteRegionLogsMissing) { + co_return; + } + traceRemoteRegionStall(self, state.getPtr(), true, std::max(0.0, now() - state->missingSince)); + } +} + // Keeps the coordinated state (cstate) updated as the set of recruited tlogs change through recovery. Future trackTlogRecovery(Reference self, Reference>> oldLogSystems, Future minRecoveryDuration) { + // Shared with the refresher, which runs as a local future while a stall is active: when this + // function returns (final update), the refresher is cancelled automatically. + Reference stallState = makeReference(); + stallState->eventName = + getRecoveryEventName(ClusterRecoveryEventType::CLUSTER_RECOVERY_REMOTE_REGION_STALL_EVENT_NAME); + Future refresher; Future rejoinRequests = Never(); DBRecoveryCount recoverCount = self->cstate.myDBState.recoveryCount + 1; DatabaseConfiguration configuration = self->configuration; // self-configuration can be changed by configurationMonitor so we need a copy + // Start of the current remote-region stall, if any. Kept across loop iterations so that + // re-emissions of the event on unrelated core-state changes do not reset the reported + // stall duration; it grows monotonically until the log set is complete. + Optional remoteLogsMissingSince; while (true) { DBCoreState newState; self->logSystem->toCoreState(newState); @@ -581,6 +633,45 @@ Future trackTlogRecovery(Reference self, .trackLatest(self->clusterRecoveryStateEventHolder->trackingKey); } + // "Degraded multi-region": with usableRegions > 1 the remote region's log set has not been + // recruited (allLogs == false). oldTLogData is deliberately not a discriminator: when the + // remote region is down, old generations cannot be purged (finalUpdate requires allLogs), + // so oldTLogData stays non-empty precisely in this stalled state. + // + // Gate on ACCEPTING_COMMITS: a missing remote log set is a transient recruiting artifact + // during normal recovery, so StallSeconds must not start accumulating before the cluster + // accepts commits, or a slow-but-healthy recovery trips degraded_multi_region. + bool remoteRegionLogsMissing = + configuration.usableRegions > 1 && !allLogs && self->recoveryState >= RecoveryState::ACCEPTING_COMMITS; + if (remoteRegionLogsMissing && !remoteLogsMissingSince.present()) { + remoteLogsMissingSince = now(); + } else if (!remoteRegionLogsMissing) { + remoteLogsMissingSince = Optional(); + } + // Publish the current snapshot for the refresher actor before starting it: the refresher reads + // remoteRegionLogsMissing synchronously on creation (before its first co_await), so publishing + // first lets it enter its loop immediately instead of deferring the first re-emission by a full + // loop iteration. + stallState->remoteRegionLogsMissing = remoteRegionLogsMissing; + stallState->allLogs = allLogs; + stallState->oldTLogDataSize = newState.oldTLogData.size(); + stallState->usableRegions = configuration.usableRegions; + if (remoteLogsMissingSince.present()) { + stallState->missingSince = remoteLogsMissingSince.get(); + } + if (remoteRegionLogsMissing) { + if (!refresher.isValid() || refresher.isReady()) { + refresher = remoteRegionStallEventRefresher(self, stallState); + } + } + // StallSeconds is carried in the event itself rather than derived from the event's + // emission Time by status: the event is re-emitted on every core-state change, and + // deriving the duration from the latest emission would reset the stall counter while + // the stall is still ongoing. + double remoteRegionStallSeconds = + remoteLogsMissingSince.present() ? std::max(0.0, now() - remoteLogsMissingSince.get()) : 0.0; + traceRemoteRegionStall(self, stallState.getPtr(), remoteRegionLogsMissing, remoteRegionStallSeconds); + self->registrationTrigger.trigger(); if (finalUpdate) { @@ -2100,6 +2191,8 @@ const std::string& getRecoveryEventName(ClusterRecoveryEventType type) { SERVER_KNOBS->CLUSTER_RECOVERY_EVENT_NAME_PREFIX + "RecoveryAvailable" }); recoveryEventNameMap.insert({ ClusterRecoveryEventType::CLUSTER_RECOVERY_METRICS_EVENT_NAME, SERVER_KNOBS->CLUSTER_RECOVERY_EVENT_NAME_PREFIX + "RecoveryMetrics" }); + recoveryEventNameMap.insert({ ClusterRecoveryEventType::CLUSTER_RECOVERY_REMOTE_REGION_STALL_EVENT_NAME, + SERVER_KNOBS->CLUSTER_RECOVERY_EVENT_NAME_PREFIX + "RecoveryRemoteRegionStall" }); } auto iter = recoveryEventNameMap.find(type); diff --git a/fdbserver/clustercontroller/ClusterRecovery.h b/fdbserver/clustercontroller/ClusterRecovery.h index ac6f4096619..c62f6aaca5e 100644 --- a/fdbserver/clustercontroller/ClusterRecovery.h +++ b/fdbserver/clustercontroller/ClusterRecovery.h @@ -55,6 +55,7 @@ enum ClusterRecoveryEventType { CLUSTER_RECOVERY_COMMIT_EVENT_NAME, CLUSTER_RECOVERY_AVAILABLE_EVENT_NAME, CLUSTER_RECOVERY_METRICS_EVENT_NAME, + CLUSTER_RECOVERY_REMOTE_REGION_STALL_EVENT_NAME, CLUSTER_RECOVERY_LAST // Always the last entry }; @@ -254,6 +255,7 @@ struct ClusterRecoveryData : NonCopyable, ReferenceCounted Reference clusterRecoveryGenerationsEventHolder; Reference clusterRecoveryDurationEventHolder; Reference clusterRecoveryAvailableEventHolder; + Reference clusterRecoveryRemoteRegionStallEventHolder; ClusterRecoveryData(ClusterControllerData* controllerData, Reference const> const& dbInfo, @@ -290,6 +292,8 @@ struct ClusterRecoveryData : NonCopyable, ReferenceCounted getRecoveryEventName(ClusterRecoveryEventType::CLUSTER_RECOVERY_DURATION_EVENT_NAME)); clusterRecoveryAvailableEventHolder = makeReference( getRecoveryEventName(ClusterRecoveryEventType::CLUSTER_RECOVERY_AVAILABLE_EVENT_NAME)); + clusterRecoveryRemoteRegionStallEventHolder = makeReference( + getRecoveryEventName(ClusterRecoveryEventType::CLUSTER_RECOVERY_REMOTE_REGION_STALL_EVENT_NAME)); logger = cc.traceCounters(getRecoveryEventName(ClusterRecoveryEventType::CLUSTER_RECOVERY_METRICS_EVENT_NAME), dbgid, diff --git a/fdbserver/clustercontroller/Status.cpp b/fdbserver/clustercontroller/Status.cpp index 1936a7c0c56..3c032a2d936 100644 --- a/fdbserver/clustercontroller/Status.cpp +++ b/fdbserver/clustercontroller/Status.cpp @@ -676,6 +676,46 @@ struct RolesInfo { if (commitBatchingWindowSize.size()) { obj["commit_batching_window_size"] = addLatencyStatistics(commitBatchingWindowSize); } + + TraceEventFields const& commitBatchTransactions = metrics.at("CommitBatchTransactions"); + if (commitBatchTransactions.size()) { + obj["commit_batch_transactions"] = addLatencyStatistics(commitBatchTransactions); + } + + TraceEventFields const& commitBatchBytes = metrics.at("CommitBatchBytes"); + if (commitBatchBytes.size()) { + obj["commit_batch_bytes"] = addLatencyStatistics(commitBatchBytes); + } + + TraceEventFields const& commitBatchingWaiting = metrics.at("CommitBatchingWaiting"); + if (commitBatchingWaiting.size()) { + obj["commit_batching_waiting"] = addLatencyStatistics(commitBatchingWaiting); + } + + TraceEventFields const& commitPreresolutionLatency = metrics.at("CommitPreresolutionLatency"); + if (commitPreresolutionLatency.size()) { + obj["commit_preresolution_latency"] = addLatencyStatistics(commitPreresolutionLatency); + } + + TraceEventFields const& commitResolutionLatency = metrics.at("CommitResolutionLatency"); + if (commitResolutionLatency.size()) { + obj["commit_resolution_latency"] = addLatencyStatistics(commitResolutionLatency); + } + + TraceEventFields const& commitPostresolutionLatency = metrics.at("CommitPostresolutionLatency"); + if (commitPostresolutionLatency.size()) { + obj["commit_postresolution_latency"] = addLatencyStatistics(commitPostresolutionLatency); + } + + TraceEventFields const& commitTLogLoggingLatency = metrics.at("CommitTLogLoggingLatency"); + if (commitTLogLoggingLatency.size()) { + obj["commit_tlog_logging_latency"] = addLatencyStatistics(commitTLogLoggingLatency); + } + + TraceEventFields const& commitReplyLatency = metrics.at("CommitReplyLatency"); + if (commitReplyLatency.size()) { + obj["commit_reply_latency"] = addLatencyStatistics(commitReplyLatency); + } } catch (Error& e) { if (e.code() != error_code_attribute_not_found) { throw e; @@ -1162,13 +1202,54 @@ static JsonBuilderObject clientStatusFetcher( return clientStatus; } +// Parse the recovery-provided remote-region-stall event into a boolean. The event is emitted by +// trackTlogRecovery when the remote region's log set could not be recruited (allLogs == false with +// usableRegions > 1). An absent or empty event (e.g. an older server that does not emit it yet) means the +// signal is not present, so the cluster is not flagged as degraded. +static bool parseRemoteRegionLogsMissing(const TraceEventFields& remoteRegionStallEvent) { + if (remoteRegionStallEvent.size() == 0) { + return false; + } + return remoteRegionStallEvent.getInt("RemoteRegionLogsMissing") != 0; +} + +// Parse the duration (in seconds) for which the remote-region-stall signal has been active. The recovery +// side carries the duration in the event itself ("StallSeconds") because the event is (re)emitted on every +// core-state change; deriving the duration from the latest emission time would reset the stall counter +// while the stall is still ongoing. An absent event, an event reporting that the logs are not missing, or +// any malformed value conservatively reports 0, which keeps the cluster from being flagged as degraded. +static double parseRemoteRegionStallSeconds(const TraceEventFields& remoteRegionStallEvent) { + try { + if (remoteRegionStallEvent.size() == 0) { + return 0.0; + } + if (remoteRegionStallEvent.getInt("RemoteRegionLogsMissing") == 0) { + return 0.0; + } + double stallSeconds = std::stod(remoteRegionStallEvent.getValue("StallSeconds")); + return stallSeconds > 0.0 ? stallSeconds : 0.0; + } catch (Error& e) { + if (e.code() == error_code_actor_cancelled) { + throw; + } + return 0.0; + } catch (std::exception&) { + // Malformed numeric fields fail safe to "no sustained stall". + return 0.0; + } +} + static AsyncResult recoveryStateStatusFetcher(Database cx, WorkerDetails ccWorker, WorkerDetails mWorker, int workerCount, std::set* incomplete_reasons, - int* statusCode) { + int* statusCode, + bool* remoteRegionLogsMissing, + double* remoteRegionStallSeconds) { JsonBuilderObject message; + *remoteRegionLogsMissing = false; + *remoteRegionStallSeconds = 0.0; Transaction tr(cx); try { Future mdActiveGensF = @@ -1183,10 +1264,24 @@ static AsyncResult recoveryStateStatusFetcher(Database cx, timeoutError(ccWorker.interf.eventLogRequest.getReply(EventLogRequest(StringRef( getRecoveryEventName(ClusterRecoveryEventType::CLUSTER_RECOVERY_AVAILABLE_EVENT_NAME)))), 1.0); + Future remoteRegionStallF = + timeoutError(ccWorker.interf.eventLogRequest.getReply(EventLogRequest(StringRef(getRecoveryEventName( + ClusterRecoveryEventType::CLUSTER_RECOVERY_REMOTE_REGION_STALL_EVENT_NAME)))), + 1.0); tr.setOption(FDBTransactionOptions::PRIORITY_SYSTEM_IMMEDIATE); Future> rvF = errorOr(timeoutError(tr.getReadVersion(), 1.0)); - co_await (success(mdActiveGensF) && success(mdF) && success(rvF) && success(mDBAvailableF)); + co_await (success(mdActiveGensF) && success(mdF) && success(rvF) && success(mDBAvailableF) && + success(remoteRegionStallF)); + + // Precisely report whether recovery is stalled because the remote region's log set could not be + // recruited (allLogs == false with usableRegions > 1). This replaces the coarser heuristic of matching + // on the accepting_commits recovery state, which would also flag a normal transient pass through + // accepting_commits as degraded. Additionally report how long the stall has been active, because the + // signal is also transiently true during a normal (non-initial) recovery of a healthy multi-region + // cluster until the remote epoch is recruited. + *remoteRegionLogsMissing = parseRemoteRegionLogsMissing(remoteRegionStallF.get()); + *remoteRegionStallSeconds = parseRemoteRegionStallSeconds(remoteRegionStallF.get()); const TraceEventFields& md = mdF.get(); int mStatusCode = md.getInt("StatusCode"); @@ -1765,6 +1860,7 @@ static JsonBuilderObject configurationFetcher(Optional co static AsyncResult dataStatusFetcher(WorkerDetails ddWorker, DatabaseConfiguration configuration, + bool degradedMultiRegion, int* minStorageReplicasRemaining) { JsonBuilderObject statusObjData; @@ -1791,7 +1887,9 @@ static AsyncResult dataStatusFetcher(WorkerDetails ddWorker, if (startingStats.size() && startingStats.getValue("State") != "Active") { JsonBuilderObject stateSectionObj; stateSectionObj["name"] = "initializing"; - stateSectionObj["description"] = "(Re)initializing automatic data distribution"; + stateSectionObj["description"] = degradedMultiRegion + ? "Degraded multiregional (Re)initializing automatic data distribution" + : "(Re)initializing automatic data distribution"; statusObjData["state"] = stateSectionObj; co_return statusObjData; } @@ -2002,7 +2100,12 @@ static Future>> getTLogsAndMetric static Future>> getCommitProxiesAndMetrics( Reference> db, std::unordered_map address_workers) { - std::vector eventNames{ "CommitLatencyMetrics", "CommitLatencyBands", "CommitBatchingWindowSize" }; + std::vector eventNames{ + "CommitLatencyMetrics", "CommitLatencyBands", "CommitBatchingWindowSize", + "CommitBatchTransactions", "CommitBatchBytes", "CommitBatchingWaiting", + "CommitPreresolutionLatency", "CommitResolutionLatency", "CommitPostresolutionLatency", + "CommitTLogLoggingLatency", "CommitReplyLatency" + }; std::vector> results = co_await getServerMetrics(db->get().client.commitProxies, address_workers, std::move(eventNames)); @@ -2695,6 +2798,7 @@ AsyncResult layerStatusFetcher(Database cx, // TODO: Also fetch other linked subtrees of meta keys std::vector> docFutures; + docFutures.reserve(jsonLayers.size()); for (int i = 0; i < jsonLayers.size(); ++i) { docFutures.push_back( tr.getRange(KeyRangeRef(jsonLayers[i].value, strinc(jsonLayers[i].value)), 1000)); @@ -3024,9 +3128,17 @@ static AsyncResult clusterGetStatusImpl(Reference s // construct status information for cluster subsections int statusCode = (int)RecoveryStatus::END; + bool remoteRegionLogsMissing = false; + double remoteRegionStallSeconds = 0.0; std::vector>> clusterSubsectionFetchers; - clusterSubsectionFetchers.push_back(errorOr(recoveryStateStatusFetcher( - cx, ccWorker, mWorker, workers.size(), &status_incomplete_reasons, &statusCode))); + clusterSubsectionFetchers.push_back(errorOr(recoveryStateStatusFetcher(cx, + ccWorker, + mWorker, + workers.size(), + &status_incomplete_reasons, + &statusCode, + &remoteRegionLogsMissing, + &remoteRegionStallSeconds))); clusterSubsectionFetchers.push_back(errorOr(timeoutError(getIdmpKeyStatus(cx), 5.0))); clusterSubsectionFetchers.push_back(errorOr(versionEpochStatusFetcher(cx, &status_incomplete_reasons))); @@ -3164,9 +3276,22 @@ static AsyncResult clusterGetStatusImpl(Reference s int fullyReplicatedRegions = -1; // NOTE: here we should start all the transaction before wait in order to overlay latency Future> primaryDCFO = getActivePrimaryDC(cx, &fullyReplicatedRegions, &messages); + // The cluster is "degraded multi-region" when recovery explicitly reports that it is stalled because + // the remote region's log set could not be recruited (allLogs == false with usableRegions > 1) after + // accepting commits, and the stall has persisted for a while. Committed data is still safe in the + // surviving region (and its satellite), so this is a degraded-but-not-data-losing state. Requiring + // the recovery state to be at (or past) accepting_commits avoids the transient phases before the + // cluster is actually accepting commits. The sustained-stall requirement avoids the transient window + // during a normal (non-initial) recovery of a healthy multi-region cluster, where the remote log set + // is legitimately absent at accepting_commits until the remote epoch finishes recruiting. + bool degradedMultiRegion = + configuration.present() && statusCode >= RecoveryStatus::accepting_commits && + configuration.get().usableRegions > 1 && remoteRegionLogsMissing && + remoteRegionStallSeconds >= SERVER_KNOBS->DEGRADED_MULTI_REGION_MIN_STALL_SECONDS; + statusObj["degraded_multi_region"] = degradedMultiRegion; std::vector> statusSectionFetchers; statusSectionFetchers.push_back( - dataStatusFetcher(ddWorker, configuration.get(), &minStorageReplicasRemaining)); + dataStatusFetcher(ddWorker, configuration.get(), degradedMultiRegion, &minStorageReplicasRemaining)); statusSectionFetchers.push_back(workloadStatusFetcher( db, workers, mWorker, rkWorker, &qos, &dataOverlay, &status_incomplete_reasons, storageServerFuture)); statusSectionFetchers.push_back(layerStatusFetcher(cx, &messages, &status_incomplete_reasons)); @@ -3503,10 +3628,15 @@ StatusReply clusterGetFaultToleranceStatus(const std::string& statusStr) { json_spirit::mValue mv = readJSONStrictly(statusStr); JSONDoc jsonDoc(mv); - std::string faultToleranceRelatedFields[] = { - "fault_tolerance", "data", "logs", "maintenance_zone", "maintenance_seconds_remaining", "qos", - "recovery_state", "messages" - }; + std::string faultToleranceRelatedFields[] = { "fault_tolerance", + "data", + "logs", + "maintenance_zone", + "maintenance_seconds_remaining", + "qos", + "recovery_state", + "messages", + "degraded_multi_region" }; JsonBuilderObject statusObj; for (std::string& field : faultToleranceRelatedFields) { @@ -3907,6 +4037,86 @@ TEST_CASE("/status/json/merging") { return Void(); } +// Verify the "degraded multi-region" signal parsing and computation, exercising the actual code paths used by +// recoveryStateStatusFetcher and clusterGetStatusImpl. A cluster is only considered degraded multi-region when +// recovery explicitly reports that the remote region's log set could not be recruited (allLogs == false with +// usableRegions > 1) and the stall has persisted long enough to rule out the transient window that occurs +// during a normal, non-initial recovery of a healthy multi-region cluster. +TEST_CASE("/fdbserver/clustercontroller/degradedMultiRegionComputation") { + // parseRemoteRegionLogsMissing: event carries the signal. + { + TraceEventFields remoteRegionStallEvent; + remoteRegionStallEvent.addField("RemoteRegionLogsMissing", "1"); + ASSERT(parseRemoteRegionLogsMissing(remoteRegionStallEvent)); + } + // parseRemoteRegionLogsMissing: event explicitly says logs are not missing (normal recovery progressing). + { + TraceEventFields remoteRegionStallEvent; + remoteRegionStallEvent.addField("RemoteRegionLogsMissing", "0"); + ASSERT(!parseRemoteRegionLogsMissing(remoteRegionStallEvent)); + } + // parseRemoteRegionLogsMissing: empty event (older server that does not emit the signal) means false. + { + TraceEventFields remoteRegionStallEvent; + ASSERT(!parseRemoteRegionLogsMissing(remoteRegionStallEvent)); + } + + // Real scenario from production logs: with 2 usable regions, the first region is down, recovery is stalled + // and the remote log set has not been recruited (AllLogs=0), while old log generations remain (OldTLogDataSize=2). + // This is exactly the "degraded but not data-losing" state the signal is meant to surface. + { + TraceEventFields remoteRegionStallEvent; + remoteRegionStallEvent.addField("RemoteRegionLogsMissing", "1"); + remoteRegionStallEvent.addField("AllLogs", "0"); + remoteRegionStallEvent.addField("OldTLogDataSize", "2"); + remoteRegionStallEvent.addField("UsableRegions", "2"); + ASSERT(parseRemoteRegionLogsMissing(remoteRegionStallEvent)); + } + + // parseRemoteRegionStallSeconds: logs not missing -> no active stall, regardless of the carried duration. + { + TraceEventFields remoteRegionStallEvent; + remoteRegionStallEvent.addField("StallSeconds", "3600.0"); + remoteRegionStallEvent.addField("RemoteRegionLogsMissing", "0"); + ASSERT_EQ(parseRemoteRegionStallSeconds(remoteRegionStallEvent), 0.0); + } + // parseRemoteRegionStallSeconds: empty event -> no active stall. + { + TraceEventFields remoteRegionStallEvent; + ASSERT_EQ(parseRemoteRegionStallSeconds(remoteRegionStallEvent), 0.0); + } + // parseRemoteRegionStallSeconds: sustained stall carries the monotonic duration reported by recovery. + { + const double threshold = SERVER_KNOBS->DEGRADED_MULTI_REGION_MIN_STALL_SECONDS; + TraceEventFields remoteRegionStallEvent; + remoteRegionStallEvent.addField("StallSeconds", format("%.6f", threshold + 5.0)); + remoteRegionStallEvent.addField("RemoteRegionLogsMissing", "1"); + ASSERT_GE(parseRemoteRegionStallSeconds(remoteRegionStallEvent), threshold); + } + { + const double threshold = SERVER_KNOBS->DEGRADED_MULTI_REGION_MIN_STALL_SECONDS; + TraceEventFields remoteRegionStallEvent; + remoteRegionStallEvent.addField("StallSeconds", "0.1"); + remoteRegionStallEvent.addField("RemoteRegionLogsMissing", "1"); + ASSERT_LT(parseRemoteRegionStallSeconds(remoteRegionStallEvent), threshold); + } + // parseRemoteRegionStallSeconds: malformed or negative StallSeconds fails safe to 0 (not degraded). + { + TraceEventFields remoteRegionStallEvent; + remoteRegionStallEvent.addField("StallSeconds", "garbage"); + remoteRegionStallEvent.addField("RemoteRegionLogsMissing", "1"); + ASSERT_EQ(parseRemoteRegionStallSeconds(remoteRegionStallEvent), 0.0); + } + { + TraceEventFields remoteRegionStallEvent; + remoteRegionStallEvent.addField("StallSeconds", "-5.0"); + remoteRegionStallEvent.addField("RemoteRegionLogsMissing", "1"); + ASSERT_EQ(parseRemoteRegionStallSeconds(remoteRegionStallEvent), 0.0); + } + + co_return; +} + // Test that clusterGetStatus returns partial results within the specified deadline // even when the database is unavailable and transactions hang forever. TEST_CASE("/fdbserver/clustercontroller/clusterGetStatusTimeout") { diff --git a/fdbserver/commitproxy/CommitProxyServer.cpp b/fdbserver/commitproxy/CommitProxyServer.cpp index 879fc224635..848a7670e1e 100644 --- a/fdbserver/commitproxy/CommitProxyServer.cpp +++ b/fdbserver/commitproxy/CommitProxyServer.cpp @@ -505,6 +505,7 @@ struct CommitBatchContext { const int currentBatchMemBytesCount; double startTime; + double timerStartTime; // The current stage of batch commit std::string_view stage = UNSET; @@ -746,7 +747,7 @@ CommitBatchContext::CommitBatchContext(ProxyCommitData* const pProxyCommitData_, const int currentBatchMemBytesCount) : pProxyCommitData(pProxyCommitData_), trs(std::move(*const_cast*>(trs_))), currentBatchMemBytesCount(currentBatchMemBytesCount), startTime(g_network->now()), - localBatchNumber(++pProxyCommitData->localCommitBatchesStarted), + timerStartTime(g_network->timer()), localBatchNumber(++pProxyCommitData->localCommitBatchesStarted), toCommit(pProxyCommitData->logSystem, pProxyCommitData->localTLogCount), span("MP:commitBatch"_loc), committed(trs.size()), lastShardMove(invalidVersion) { @@ -920,6 +921,7 @@ Future preresolutionProcessing(CommitBatchContext* self) { r->value().emplace_back(versionReply.resolverChangesVersion, it.dest); } + pProxyCommitData->stats.commitPreresolutionLatency.addMeasurement(g_network->timer_monotonic() - startTime); //TraceEvent("ProxyGotVer", pProxyContext->dbgid).detail("Commit", commitVersion).detail("Prev", prevVersion); if (debugID.present()) { @@ -1013,7 +1015,10 @@ Future getResolution(CommitBatchContext* self) { self->resolution = std::move(resolutionResp); } - self->pProxyCommitData->stats.resolutionDist->sampleSeconds(g_network->timer_monotonic() - resolutionStart); + double resolutionDuration = g_network->timer_monotonic() - resolutionStart; + + self->pProxyCommitData->stats.commitResolutionLatency.addMeasurement(resolutionDuration); + self->pProxyCommitData->stats.resolutionDist->sampleSeconds(resolutionDuration); if (self->debugIDs.present()) { Optional debugID = self->getDebugID(); g_traceBatch.addEvent("CommitDebug", @@ -1875,7 +1880,10 @@ Future postResolution(CommitBatchContext* self) { } } - pProxyCommitData->stats.processingMutationDist->sampleSeconds(g_network->timer_monotonic() - postResolutionQueuing); + double postResolutionEnd = g_network->timer_monotonic(); + + pProxyCommitData->stats.commitPostresolutionLatency.addMeasurement(postResolutionEnd - postResolutionStart); + pProxyCommitData->stats.processingMutationDist->sampleSeconds(postResolutionEnd - postResolutionQueuing); } Future transactionLogging(CommitBatchContext* self) { @@ -1915,7 +1923,11 @@ Future transactionLogging(CommitBatchContext* self) { pProxyCommitData->txsPopVersions.emplace_back(self->commitVersion, self->msg.popTo); } pProxyCommitData->logSystemConsumer->popTxs(self->msg.popTo); - pProxyCommitData->stats.tlogLoggingDist->sampleSeconds(g_network->timer_monotonic() - tLoggingStart); + + double tLoggingDuration = g_network->timer_monotonic() - tLoggingStart; + + pProxyCommitData->stats.commitTLogLoggingLatency.addMeasurement(tLoggingDuration); + pProxyCommitData->stats.tlogLoggingDist->sampleSeconds(tLoggingDuration); } Future reply(CommitBatchContext* self) { @@ -2066,6 +2078,7 @@ Future reply(CommitBatchContext* self) { // TODO: filter if pipelined with large commit const double duration = endTime - tr.requestTime(); pProxyCommitData->stats.commitLatencySample.addMeasurement(duration); + pProxyCommitData->stats.commitBatchingWaiting.addMeasurement(self->timerStartTime - tr.requestTime()); if (pProxyCommitData->latencyBandConfig.present()) { bool filter = self->maxTransactionBytes > pProxyCommitData->latencyBandConfig.get().commitConfig.maxCommitBytes.orDefault( @@ -2126,7 +2139,10 @@ Future reply(CommitBatchContext* self) { pProxyCommitData->commitBatchesMemBytesCount -= self->currentBatchMemBytesCount; ASSERT_ABORT(pProxyCommitData->commitBatchesMemBytesCount >= 0); co_await self->releaseFuture; - pProxyCommitData->stats.replyCommitDist->sampleSeconds(g_network->timer_monotonic() - replyStart); + double replyDuration = g_network->timer_monotonic() - replyStart; + + pProxyCommitData->stats.commitReplyLatency.addMeasurement(replyDuration); + pProxyCommitData->stats.replyCommitDist->sampleSeconds(replyDuration); } // Commit one batch of transactions trs @@ -2144,6 +2160,8 @@ Future commitBatchImpl(CommitBatchContext* pContext) { pContext->pProxyCommitData->lastVersionTime = pContext->startTime; ++pContext->pProxyCommitData->stats.commitBatchIn; pContext->setupTraceBatch(); + pContext->pProxyCommitData->stats.commitBatchBytes.addMeasurement(pContext->currentBatchMemBytesCount); + pContext->pProxyCommitData->stats.commitBatchTransactions.addMeasurement(pContext->trs.size()); /////// Phase 1: Pre-resolution processing (CPU bound except waiting for a version # which is separately pipelined /// and *should* be available by now (unless empty commit); ordered; currently atomic but could yield) diff --git a/fdbserver/commitproxy/ProxyCommitData.h b/fdbserver/commitproxy/ProxyCommitData.h index 96e0e8c16c5..51611553599 100644 --- a/fdbserver/commitproxy/ProxyCommitData.h +++ b/fdbserver/commitproxy/ProxyCommitData.h @@ -90,6 +90,19 @@ struct ProxyStats { LatencySample commitBatchingWindowSize; + // Number of transactions in the batch + LatencySample commitBatchTransactions; + // Summary length of transactions in the batch + LatencySample commitBatchBytes; + // what time transactions were waiting in the batch before they + // started processing with commitBatch() + LatencySample commitBatchingWaiting; + LatencySample commitPreresolutionLatency; + LatencySample commitResolutionLatency; + LatencySample commitPostresolutionLatency; + LatencySample commitTLogLoggingLatency; + LatencySample commitReplyLatency; + LatencySample computeLatency; Future logger; @@ -184,6 +197,38 @@ struct ProxyStats { id, SERVER_KNOBS->LATENCY_METRICS_LOGGING_INTERVAL, SERVER_KNOBS->LATENCY_SKETCH_ACCURACY), + commitBatchTransactions("CommitBatchTransactions", + id, + SERVER_KNOBS->LATENCY_METRICS_LOGGING_INTERVAL, + SERVER_KNOBS->LATENCY_SKETCH_ACCURACY), + commitBatchBytes("CommitBatchBytes", + id, + SERVER_KNOBS->LATENCY_METRICS_LOGGING_INTERVAL, + SERVER_KNOBS->LATENCY_SKETCH_ACCURACY), + commitBatchingWaiting("CommitBatchingWaiting", + id, + SERVER_KNOBS->LATENCY_METRICS_LOGGING_INTERVAL, + SERVER_KNOBS->LATENCY_SKETCH_ACCURACY), + commitPreresolutionLatency("CommitPreresolutionLatency", + id, + SERVER_KNOBS->LATENCY_METRICS_LOGGING_INTERVAL, + SERVER_KNOBS->LATENCY_SKETCH_ACCURACY), + commitResolutionLatency("CommitResolutionLatency", + id, + SERVER_KNOBS->LATENCY_METRICS_LOGGING_INTERVAL, + SERVER_KNOBS->LATENCY_SKETCH_ACCURACY), + commitPostresolutionLatency("CommitPostresolutionLatency", + id, + SERVER_KNOBS->LATENCY_METRICS_LOGGING_INTERVAL, + SERVER_KNOBS->LATENCY_SKETCH_ACCURACY), + commitTLogLoggingLatency("CommitTLogLoggingLatency", + id, + SERVER_KNOBS->LATENCY_METRICS_LOGGING_INTERVAL, + SERVER_KNOBS->LATENCY_SKETCH_ACCURACY), + commitReplyLatency("CommitReplyLatency", + id, + SERVER_KNOBS->LATENCY_METRICS_LOGGING_INTERVAL, + SERVER_KNOBS->LATENCY_SKETCH_ACCURACY), computeLatency("ComputeLatency", id, SERVER_KNOBS->LATENCY_METRICS_LOGGING_INTERVAL, diff --git a/fdbserver/core/MoveKeys.cpp b/fdbserver/core/MoveKeys.cpp index 3367154f286..e03a26fb654 100644 --- a/fdbserver/core/MoveKeys.cpp +++ b/fdbserver/core/MoveKeys.cpp @@ -207,6 +207,7 @@ Future deleteCheckpoints(Transaction* tr, std::set checkpointIds, UID } TraceEvent(SevDebug, "DataMoveDeleteCheckpoints", dataMoveId).detail("Checkpoints", describe(checkpointIds)); std::vector>> checkpointEntries; + checkpointEntries.reserve(checkpointIds.size()); for (const UID& id : checkpointIds) { checkpointEntries.push_back(tr->get(checkpointKeyFor(id))); } @@ -884,6 +885,7 @@ Future> addReadWriteDestinations(KeyRangeRef shard, // Returns storage servers selected from 'candidates', who is serving a read-write copy of 'range'. Future> pickReadWriteServers(Transaction* tr, std::vector candidates, KeyRangeRef range) { std::vector>> serverListEntries; + serverListEntries.reserve(candidates.size()); for (const UID id : candidates) { serverListEntries.push_back(tr->get(serverListKeyFor(id))); @@ -892,6 +894,7 @@ Future> pickReadWriteServers(Transaction* tr, std::vector std::vector> serverListValues = co_await getAll(serverListEntries); std::vector ssis; + ssis.reserve(serverListValues.size()); for (auto& v : serverListValues) { ssis.push_back(decodeServerListValue(v.get())); } @@ -3155,6 +3158,7 @@ Future> addStorageServer(Database cx, StorageServerInter std::vector>> localityExclusions; std::map localityData = server.locality.getAllData(); + localityExclusions.reserve(localityData.size()); for (const auto& l : localityData) { localityExclusions.push_back(tr->get(StringRef(encodeExcludedLocalityKey( LocalityData::ExcludeLocalityPrefix.toString() + l.first + ":" + l.second)))); @@ -3575,6 +3579,7 @@ Future removeKeysFromFailedServer(Database cx, } std::vector> actors; + actors.reserve(dest.size()); // Unassign the shard from the dest servers. for (const UID& id : dest) { @@ -3864,6 +3869,7 @@ Future cleanUpDataMoveCore(Database occ, } std::vector> actors; + actors.reserve(oldDests.size()); for (const auto& uid : oldDests) { actors.push_back(unassignServerKeys(&tr, uid, range, physicalShardMap[uid], dataMoveId)); } diff --git a/fdbserver/core/RocksDBCheckpointUtils.cpp b/fdbserver/core/RocksDBCheckpointUtils.cpp index fef727c659a..118936d5fca 100644 --- a/fdbserver/core/RocksDBCheckpointUtils.cpp +++ b/fdbserver/core/RocksDBCheckpointUtils.cpp @@ -571,6 +571,7 @@ rocksdb::Status RocksDBColumnFamilyReader::Reader::tryOpenForRead(const std::str const rocksdb::ColumnFamilyOptions cfOptions = getCFOptions(); std::vector descriptors; + descriptors.reserve(columnFamilies.size()); for (const std::string& name : columnFamilies) { descriptors.emplace_back(name, cfOptions); } @@ -633,6 +634,7 @@ rocksdb::Status RocksDBColumnFamilyReader::Reader::importCheckpoint(const std::s const rocksdb::ColumnFamilyOptions cfOptions = getCFOptions(); std::vector descriptors; + descriptors.reserve(columnFamilies.size()); for (const std::string& name : columnFamilies) { descriptors.emplace_back(name, cfOptions); } diff --git a/fdbserver/core/ServerKnobs.cpp b/fdbserver/core/ServerKnobs.cpp index c5ef7653b17..cca63437f2b 100644 --- a/fdbserver/core/ServerKnobs.cpp +++ b/fdbserver/core/ServerKnobs.cpp @@ -524,6 +524,8 @@ void ServerKnobs::initialize(Randomize randomize, ClientKnobs* clientKnobs, IsSi init( BULKLOAD_ASYNC_READ_WRITE_BLOCK_SIZE, 1024*1024 ); if (isSimulated) BULKLOAD_ASYNC_READ_WRITE_BLOCK_SIZE = deterministicRandom()->randomInt(1024, 10240); init( MANIFEST_COUNT_MAX_PER_BULKLOAD_TASK, 10 ); if (isSimulated) MANIFEST_COUNT_MAX_PER_BULKLOAD_TASK = deterministicRandom()->randomInt(1, 11); init( BULKLOAD_SIM_FAILURE_INJECTION, false ); if (isSimulated) BULKLOAD_SIM_FAILURE_INJECTION = true; + init( BULKLOAD_SIM_INJECT_DEST_TEAM_FAILURES, 0 ); + init( DD_BULKLOAD_MAX_RETRYABLE_REDISPATCH, 20 ); if( randomize && buggify() ) DD_BULKLOAD_MAX_RETRYABLE_REDISPATCH = deterministicRandom()->randomInt(1, 4); init( DD_BULKLOAD_POWER_OF_D_RATIO, 2.0 ); if (isSimulated) DD_BULKLOAD_POWER_OF_D_RATIO = deterministicRandom()->randomInt(1, 11); init( DD_BULKLOAD_TASK_SUBMISSION_INTERVAL_SEC, 0.01 ); @@ -872,6 +874,8 @@ void ServerKnobs::initialize(Randomize randomize, ClientKnobs* clientKnobs, IsSi bool shortRecoveryDuration = randomize && buggify(); init( ENFORCED_MIN_RECOVERY_DURATION, 0.085 ); if( shortRecoveryDuration ) ENFORCED_MIN_RECOVERY_DURATION = 0.01; init( REQUIRED_MIN_RECOVERY_DURATION, 0.080 ); if( shortRecoveryDuration ) REQUIRED_MIN_RECOVERY_DURATION = 0.01; + init( DEGRADED_MULTI_REGION_MIN_STALL_SECONDS, 30.0 ); if( isSimulated ) DEGRADED_MULTI_REGION_MIN_STALL_SECONDS = 15.0; + init( DEGRADED_MULTI_REGION_REFRESH_SECONDS, 1.0 ); init( ALWAYS_CAUSAL_READ_RISKY, false ); init( MAX_COMMIT_UPDATES, 2000 ); if( randomize && buggify() ) MAX_COMMIT_UPDATES = 1; init( MAX_PROXY_COMPUTE, 2.0 ); @@ -1165,7 +1169,7 @@ void ServerKnobs::initialize(Randomize randomize, ClientKnobs* clientKnobs, IsSi init( FETCH_KEYS_PARALLELISM, 2 ); init( FETCH_KEYS_LOWER_PRIORITY, 0 ); init( SERVE_FETCH_CHECKPOINT_PARALLELISM, 4 ); - init( SERVE_AUDIT_STORAGE_PARALLELISM, 1 ); + init( SERVE_AUDIT_STORAGE_PARALLELISM, 4 ); if ( isSimulated ) SERVE_AUDIT_STORAGE_PARALLELISM = deterministicRandom()->randomInt(1, SERVE_AUDIT_STORAGE_PARALLELISM+1); init( PERSIST_FINISH_AUDIT_COUNT, 10 ); if ( isSimulated ) PERSIST_FINISH_AUDIT_COUNT = deterministicRandom()->randomInt(1, PERSIST_FINISH_AUDIT_COUNT+1); init( AUDIT_RETRY_COUNT_MAX, 10000 ); if ( isSimulated ) AUDIT_RETRY_COUNT_MAX = 10; init( CONCURRENT_AUDIT_TASK_COUNT_MAX, 20 ); if ( isSimulated ) CONCURRENT_AUDIT_TASK_COUNT_MAX = deterministicRandom()->randomInt(1, CONCURRENT_AUDIT_TASK_COUNT_MAX+1); @@ -1175,6 +1179,36 @@ void ServerKnobs::initialize(Randomize randomize, ClientKnobs* clientKnobs, IsSi init( AUDIT_DATAMOVE_POST_CHECK_RETRY_COUNT_MAX, 50 ); init( AUDIT_STORAGE_RATE_PER_SERVER_MAX, 50e6 ); // per second init( AUDIT_RESTORE_BATCH_KEY_LIMIT, 100000 ); // 100K keys per batch (was hardcoded 10K) + // An audit divides its range into TASKS; each task is handled by one storage server, which walks it in + // BATCHES of reads. The knobs below bound those two units independently: + // AUDIT_TASK_MAX_BYTES -- how much keyspace one task covers + // AUDIT_RESTORE_BATCH_* -- how much one read inside a task fetches (validate_restore only) + + // Max bytes one comparison batch fetches from each side. This is an upper bound, not the size used: + // the actual budget moves between AUDIT_RESTORE_BATCH_BYTE_LIMIT_MIN and this value, halving whenever a + // read fails and growing back on success (see nextAuditBatchBytes). + // + // It adapts because the real constraint is a deadline, not a size: a batch reads at a pinned version + // that expires after MAX_READ_TRANSACTION_LIFE_VERSIONS, so what a server can fetch in one go depends + // on current load, and overshooting fails with transaction_too_old and is re-read from the start. + // + // Raise for fewer, larger round trips. Lower if audits provoke transaction_too_old, or to cut memory: + // a server holds ~4 x this x SERVE_AUDIT_STORAGE_PARALLELISM in flight. + init( AUDIT_RESTORE_BATCH_BYTE_LIMIT, 4e6 ); if( randomize && buggify() ) AUDIT_RESTORE_BATCH_BYTE_LIMIT = 80000; + // Smallest the adaptive batch above may shrink to, so a run of failed reads cannot leave the audit + // crawling through tiny batches. + init( AUDIT_RESTORE_BATCH_BYTE_LIMIT_MIN, 256e3 ); if( randomize && buggify() ) AUDIT_RESTORE_BATCH_BYTE_LIMIT_MIN = 1000; + // Max bytes of keyspace one audit task covers, for every audit type (ValidateHA, ValidateReplica, + // ValidateRestore). Tasks default to one keyServers shard, and a shard bigger than this is subdivided + // so that no single task dominates. + // + // This bounds the audit phase's wall-clock: a task is scanned start to finish by one storage server, + // so the phase cannot end before its largest task does, and making individual tasks faster cannot help. + // + // Lower for a shorter tail at the cost of more tasks. 0 disables subdivision. A target, not a hard + // bound -- splitStorageMetrics jitters piece sizes and the client merges a trailing piece under + // STORAGE_METRICS_UNFAIR_SPLIT_LIMIT back into its predecessor, so a task can reach ~1.4x this. + init( AUDIT_TASK_MAX_BYTES, 128e6 ); if( randomize && buggify() ) AUDIT_TASK_MAX_BYTES = deterministicRandom()->coinflip() ? 0 : 1e6; init( AUDIT_PROGRESS_PERSIST_BYTES_INTERVAL, 100000000 ); // 100MB - only persist progress after this many bytes init( ENABLE_AUDIT_VERBOSE_TRACE, false ); // Disabled in simulation: audit_storage locationmetadata already runs at controlled times in sim, diff --git a/fdbserver/core/include/fdbserver/core/Knobs.h b/fdbserver/core/include/fdbserver/core/Knobs.h index ca1c827cc73..f6d3fd8fa42 100644 --- a/fdbserver/core/include/fdbserver/core/Knobs.h +++ b/fdbserver/core/include/fdbserver/core/Knobs.h @@ -434,6 +434,16 @@ class SWIFT_CXX_IMMORTAL_SINGLETON_TYPE ServerKnobs : public KnobsImplisSimulated() || + bulkLoadInjectedDestTeamFailures >= SERVER_KNOBS->BULKLOAD_SIM_INJECT_DEST_TEAM_FAILURES) { + return false; + } + UID const taskId = rd.bulkLoadTask.get().coreState.getTaskId(); + if (!bulkLoadInjectionTargetTaskId.isValid()) { + bulkLoadInjectionTargetTaskId = taskId; + } else if (bulkLoadInjectionTargetTaskId != taskId) { + return false; + } + ++bulkLoadInjectedDestTeamFailures; + return true; +} + static bool shouldYieldDestinationFailureRetry(RelocateData const& retry, RelocateData const& queued) { bool isSplit = retry.reason == RelocateReason::SIZE_SPLIT || retry.reason == RelocateReason::WRITE_SPLIT; return isSplit && retry.keys != queued.keys && retry.keys.contains(queued.keys); @@ -308,6 +324,7 @@ class ParallelTCInfo final : public ReferenceCounted, public IDa Future updateStorageMetrics() override { std::vector> futures; + futures.reserve(teams.size()); for (auto& team : teams) { futures.push_back(team->updateStorageMetrics()); @@ -2041,8 +2058,10 @@ Future dataDistributionRelocator(DDQueue* self, .detail("Priority", rd.priority) .detail("DataMoveReason", static_cast(rd.dmReason)); if (rd.bulkLoadTask.get().completeAck.canBeSet()) { - // Unretriable error. So, we give up the task at this time. - rd.bulkLoadTask.get().completeAck.send(BulkLoadAck(/*unretryableError=*/true, rd.priority)); + // No team is disjoint from src. Terminal for this data move, but the task can + // still be narrowed: see BulkLoadAck::Outcome::Unplaceable. + rd.bulkLoadTask.get().completeAck.send( + BulkLoadAck(BulkLoadAck::Outcome::Unplaceable, rd.priority)); throw data_move_dest_team_not_found(); // This relocator should silently exit. Note that if this bulkload data move is // a team unhealthy data move, the bulkload engine will issue a new data move on @@ -2302,7 +2321,8 @@ Future dataDistributionRelocator(DDQueue* self, break; } } else if (res.index() == 1) { - if (!healthyDestinations.isHealthy()) { + if (!healthyDestinations.isHealthy() || + self->injectBulkLoadDestinationTeamFailure(doBulkLoading, rd)) { if (!signalledTransferComplete) { signalledTransferComplete = true; self->dataTransferComplete.send(rd); @@ -2316,8 +2336,13 @@ Future dataDistributionRelocator(DDQueue* self, .detail("Range", rd.keys) .detail("Dest", describe(destIds)); if (doBulkLoading && rd.bulkLoadTask.get().completeAck.canBeSet()) { + CODE_PROBE(true, "Bulkload data move lost its destination team"); + // Recoverable: the same task can succeed against a team chosen + // later. Terminal here would abandon the task, and its key-values + // exist only in the dump until an attempt ingests them, so the + // range would simply be missing from the restored database. rd.bulkLoadTask.get().completeAck.send( - BulkLoadAck(/*unretryableError=*/true, rd.priority)); + BulkLoadAck(BulkLoadAck::Outcome::Retryable, rd.priority)); } retryAfterDestinationTeamFailure = shouldRetryDestinationTeamFailure(doBulkLoading, rd); throw data_move_dest_team_not_found(); diff --git a/fdbserver/datadistributor/DDRelocationQueue.h b/fdbserver/datadistributor/DDRelocationQueue.h index 2ade05d8272..8bde515b443 100644 --- a/fdbserver/datadistributor/DDRelocationQueue.h +++ b/fdbserver/datadistributor/DDRelocationQueue.h @@ -370,6 +370,16 @@ class DDQueue : public IDDRelocationQueue, public ReferenceCounted { int getUnhealthyRelocationCount() const override; + // Simulation-only test hook, off by default (BULKLOAD_SIM_INJECT_DEST_TEAM_FAILURES). A team rarely + // goes unhealthy inside the window a bulkload move is in flight, so the retry and give-up paths need + // the failure injected. The budget is spent on a single task, because the give-up path bounds one + // task's restartCount and a budget spread across tasks never reaches it. Per-DDQueue rather than + // process-global: simulated processes share an address space, so file statics would make one budget + // serve every simulated data distributor in the run. + bool injectBulkLoadDestinationTeamFailure(bool doBulkLoading, const RelocateData& rd); + int bulkLoadInjectedDestTeamFailures = 0; + UID bulkLoadInjectionTargetTaskId; + void processRelocationComplete(const RelocateData& done); Future getSrcDestTeams(const int& teamCollectionIndex, diff --git a/fdbserver/datadistributor/DDTeamCollection.cpp b/fdbserver/datadistributor/DDTeamCollection.cpp index b29c13ff179..e0d4351fbf9 100644 --- a/fdbserver/datadistributor/DDTeamCollection.cpp +++ b/fdbserver/datadistributor/DDTeamCollection.cpp @@ -1015,6 +1015,9 @@ class DDTeamCollectionImpl { bool lastZeroHealthy = self->zeroHealthyTeams->get(); bool firstCheck = true; bool firstHealthChangeTrace = true; + bool lastProcessingUnhealthy = self->processingUnhealthy->get(); + bool retryDue = false; + Future retryTimer; std::unordered_set submittedShards; Future zeroServerLeftLogger; @@ -1083,14 +1086,23 @@ class DDTeamCollectionImpl { !healthy && self->shardsAffectedByTeamFailure->hasShards( ShardsAffectedByTeamFailure::Team(team->getServerIDs(), self->primary)); if (retryUnhealthyShards) { + if (!retryTimer.isValid()) { + retryTimer = delay(checkTeamDelay, TaskPriority::DataDistributionLow); + } else if (retryTimer.isReady()) { + retryDue = true; + retryTimer = delay(checkTeamDelay, TaskPriority::DataDistributionLow); + } change.push_back(self->processingUnhealthy->onChange()); change.push_back(self->pipelineFull->onChange()); - // Partial moves can leave a merged shard associated with this team without another health change. - change.push_back(delay(checkTeamDelay, TaskPriority::DataDistributionLow)); + } else { + retryTimer = Future(); + retryDue = false; } const bool healthyTeamBecameAvailable = lastZeroHealthy && !self->zeroHealthyTeams->get(); const bool processingUnhealthy = self->processingUnhealthy->get(); const bool pipelineFull = self->pipelineFull->get(); + const bool unhealthyDrained = lastProcessingUnhealthy && !processingUnhealthy; + lastProcessingUnhealthy = processingUnhealthy; bool recheck = !healthy && (lastReady != self->initialFailureReactionDelay.isReady() || healthyTeamBecameAvailable || containsFailed || retryUnhealthyShards); bool teamStateChanged = serversLeft != lastServersLeft || anyUndesired != lastAnyUndesired || @@ -1257,10 +1269,15 @@ class DDTeamCollectionImpl { std::vector shards = self->shardsAffectedByTeamFailure->getShardsFor( ShardsAffectedByTeamFailure::Team(team->getServerIDs(), self->primary)); if (teamStateChanged || !retryUnhealthyShards || healthyTeamBecameAvailable || - (!processingUnhealthy && !pipelineFull)) { + unhealthyDrained || (retryDue && !processingUnhealthy && !pipelineFull)) { // Undesired and explicitly failed relocations do not set processingUnhealthy. Keep - // retrying stranded ranges when the queue is idle and its pipeline can accept them. + // retrying stranded ranges at the polling interval when the queue can accept them. + // A transient pipeline-full edge must not clear the dedupe set and resubmit all ranges. submittedShards.clear(); + retryDue = false; + if (retryUnhealthyShards) { + retryTimer = delay(checkTeamDelay, TaskPriority::DataDistributionLow); + } } else { // An unchanged range may still be waiting behind the relocation pipeline gate. Only retry // newly mapped ranges while unhealthy relocations remain in the queue, and forget ranges @@ -1370,6 +1387,10 @@ class DDTeamCollectionImpl { } } + if (retryUnhealthyShards) { + // Partial moves can leave a merged shard associated with this team without another health change. + change.push_back(retryTimer); + } // Wait for any of the machines to change status co_await quorum(change, 1); co_await yield(); @@ -3049,7 +3070,7 @@ class DDTeamCollectionImpl { PromiseStream> addTSSInProgress; Future inProgressTSS = actorCollection(addTSSInProgress.getFuture(), &inProgressTSSCount, nullptr, nullptr, nullptr); - Reference tssState = makeReference(); + auto tssState = makeReference(); Future checkTss = self->initialFailureReactionDelay; bool pendingTSSCheck = false; @@ -3299,6 +3320,7 @@ class DDTeamCollectionImpl { static Future updateReplicasKey(DDTeamCollection* self, Optional dcId) { std::vector> serverUpdates; + serverUpdates.reserve(self->server_info.size()); for (auto& it : self->server_info) { serverUpdates.push_back(it.second->updated.getFuture()); @@ -3487,7 +3509,7 @@ class DDTeamCollectionImpl { TxnCounters* counters = updateStorageMetadataCounters(); KeyBackedObjectMap metadataMap( serverMetadataKeys.begin, IncludeVersion()); - Reference tr = makeReference(self->dbContext()); + auto tr = makeReference(self->dbContext()); bool isTss = server->getLastKnownInterface().isTss(); // Update server's storeType, especially when it was created @@ -7463,6 +7485,22 @@ class DDTeamCollectionUnitTest { ASSERT_EQ(relocations.pop().keys, mergedRange); ASSERT(!relocations.isReady()); + // A brief full/clear flap can be caused by a duplicate crossing the pipeline gate. It must not + // immediately clear submitted ranges and feed another duplicate back into the gate. + for (int i = 0; i < 3; ++i) { + collection->pipelineFull->set(true); + co_await delay(0.002); + collection->pipelineFull->set(false); + co_await delay(0.002); + ASSERT(!relocations.isReady()); + } + co_await delay(checkTeamDelay + 0.01); + ASSERT(relocations.isReady()); + ASSERT_EQ(relocations.pop().keys, mergedRange); + ASSERT(relocations.isReady()); + ASSERT_EQ(relocations.pop().keys, mergedRange); + ASSERT(!relocations.isReady()); + collection->processingUnhealthy->set(true); co_await delay(checkTeamDelay + 0.01); ASSERT(!relocations.isReady()); @@ -7623,7 +7661,7 @@ TEST_CASE("/DataDistribution/GetTeam/DeprioritizeWigglePausedTeam") { } TEST_CASE("/DataDistribution/StorageWiggler/NextIdWithMinAge") { - Reference wiggler = makeReference(nullptr); + auto wiggler = makeReference(nullptr); double startTime = now(); wiggler->addServer(UID(1, 0), StorageMetadataType(startTime - SERVER_KNOBS->DD_STORAGE_WIGGLE_MIN_SS_AGE_SEC + 5.0, @@ -7675,7 +7713,7 @@ TEST_CASE("/DataDistribution/StorageWiggler/NextIdWithMinAge") { TEST_CASE("/DataDistribution/StorageWiggler/NextIdWithTSS") { std::unique_ptr collection = DDTeamCollectionUnitTest::testMachineTeamCollection(1, makeReference(), 5); - Reference wiggler = makeReference(collection.get()); + auto wiggler = makeReference(collection.get()); std::cout << "Test when need TSS ... \n"; collection->configuration.usableRegions = 1; diff --git a/fdbserver/datadistributor/DDTxnProcessor.cpp b/fdbserver/datadistributor/DDTxnProcessor.cpp index 83c63fbe7e5..69c7edf55b1 100644 --- a/fdbserver/datadistributor/DDTxnProcessor.cpp +++ b/fdbserver/datadistributor/DDTxnProcessor.cpp @@ -151,6 +151,7 @@ class DDTxnProcessorImpl { decodeKeyServersValue(UIDtoTagMap, shards[i].value, src, dest, srcId, destId); std::vector>> serverListEntries; + serverListEntries.reserve(src.size()); for (int j = 0; j < src.size(); ++j) { serverListEntries.push_back(tr.get(serverListKeyFor(src[j]))); } diff --git a/fdbserver/datadistributor/DataDistribution.cpp b/fdbserver/datadistributor/DataDistribution.cpp index 61cade8ea07..f4eadf3ce2f 100644 --- a/fdbserver/datadistributor/DataDistribution.cpp +++ b/fdbserver/datadistributor/DataDistribution.cpp @@ -19,6 +19,8 @@ */ #include +#include +#include #include "fdbclient/Audit.h" #include "fdbclient/AuditUtils.h" @@ -445,6 +447,12 @@ struct DataDistributor : NonCopyable, ReferenceCounted { Optional bulkLoadJobManager; + // Bulkload tasks that have reached the terminal Error phase and have already been reported. + // scheduleBulkLoadTasks() rescans the task metadata every DD_BULKLOAD_SCHEDULE_MIN_INTERVAL_SEC and an + // Error-phase task is never erased, so without this the same failure is re-logged for the life of the + // cluster. Reset per DD generation, which is intended: a new DD should report the current state once. + std::unordered_set reportedErrorBulkLoadTasks; + bool bulkDumpEnabled = false; ParallelismLimitor bulkDumpParallelismLimitor; std::string folder; @@ -1102,6 +1110,147 @@ Future> triggerBulkLoadTask(Reference splitBulkLoadTask(Reference self, BulkLoadTaskState parent) { + std::vector manifests = parent.getManifests(); + if (manifests.size() < 2) { + // A single manifest is as narrow as a task gets, and a manifest can span an arbitrarily wide range, + // so this is reachable with a range covering the whole key space. Nothing here can place it: the + // cluster needs servers outside src, or the manifest needs to have been dumped more finely. + TraceEvent(SevWarnAlways, "DDBulkLoadTaskSplitDeclined", self->ddId) + .detail("Reason", "Task holds a single manifest and cannot be narrowed") + .detail("TaskRange", parent.getRange()) + .detail("TaskID", parent.getTaskId()); + co_return false; + } + std::sort(manifests.begin(), manifests.end(), [](BulkLoadManifest const& a, BulkLoadManifest const& b) { + return a.getBeginKey() < b.getBeginKey(); + }); + + // Split at a manifest boundary, because the job's manifests tile the key space: cutting the range at + // manifests[i].getBeginKey() puts manifests [0, i) wholly below the cut and [i, N) wholly at or above + // it, so neither child is missing data for its own range. + // + // Derive the children's ranges from the PARENT's range, not from their manifests' span. A task's range + // is its manifests' span intersected with the job range (see generateBulkLoadTaskRange), so a parent + // whose range was clipped is narrower than the data its manifests describe. Splitting on manifest + // min/max then yields children reaching outside the parent, handing them key space this task was never + // given. Splitting the parent's range makes the children tile it by construction. A task range narrower + // than its manifests is expected and handled: the storage server filters file content to the task range. + // + // For the same reason the cut must be chosen from the boundaries that fall strictly inside the parent's + // range. A task at a job-range edge can hold many manifests whose midpoint lies outside its clipped + // range; cutting there would leave one child empty, and declining would hand the caller the very + // unplaceable task this function exists to rescue. + std::vector insideBoundaries; + int const manifestCount = static_cast(manifests.size()); + for (int i = 0; i < manifestCount; i++) { + Key const candidate = manifests[i].getBeginKey(); + if (candidate > parent.getRange().begin && candidate < parent.getRange().end) { + insideBoundaries.push_back(i); + } + } + if (insideBoundaries.empty()) { + // Every manifest boundary is outside the parent's clipped range, so the range cannot be cut at one. + TraceEvent(SevWarnAlways, "DDBulkLoadTaskSplitDeclined", self->ddId) + .detail("Reason", "No manifest split point lies inside the task's range") + .detail("TaskRange", parent.getRange()) + .detail("TaskID", parent.getTaskId()) + .detail("ManifestCount", manifests.size()); + co_return false; + } + int const half = insideBoundaries[insideBoundaries.size() / 2]; + // The parent's range starts at or after its first manifest's begin key, so that key can never be + // strictly inside the range and index 0 is never a candidate. Both children therefore hold manifests, + // which BulkLoadTaskState requires. + ASSERT(half > 0 && half < manifestCount); + Key const boundary = manifests[half].getBeginKey(); + std::vector childRanges = { Standalone(KeyRangeRef(parent.getRange().begin, boundary)), + Standalone(KeyRangeRef(boundary, parent.getRange().end)) }; + + std::vector children; + for (int part = 0; part < 2; part++) { + int const from = part == 0 ? 0 : half; + int const to = part == 0 ? half : manifests.size(); + BulkLoadManifestSet set(to - from); + for (int i = from; i < to; i++) { + bool added = set.addManifest(manifests[i]); + ASSERT(added); + } + ASSERT(set.isValid()); + children.push_back(BulkLoadTaskState(parent.getJobId(), set, childRanges[part])); + } + // Tiling holds by construction now; these remain as cheap guards on that reasoning. + ASSERT(children[0].getRange().begin == parent.getRange().begin); + ASSERT(children[0].getRange().end == children[1].getRange().begin); + ASSERT(children[1].getRange().end == parent.getRange().end); + // The writes below must be issued in ascending key order, so keep the guard next to the reason. + // krmSetRange reads oldValue at Snapshot::True on a plain Transaction, which has no read-your-writes, + // so each call is blind to the previous one's mutations and only their order makes the result correct. + // Each call emits clear(range); set(begin, value); set(end, oldValue). Ascending, the second call's + // clear() erases the boundary value the first call parked there before rewriting it, leaving + // begin->child0, boundary->child1. Descending, the second call's trailing set(boundary, oldValue) + // lands last and republishes the parent over the second child's range -- precisely the state the + // tiling comment above says cannot exist. + ASSERT(children[0].getRange().begin < children[1].getRange().begin); + + Database cx = self->txnProcessor->context(); + Transaction tr(cx); + while (true) { + Error err; + try { + tr.setOption(FDBTransactionOptions::LOCK_AWARE); + tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); + co_await checkMoveKeysLock(&tr, self->context->lock, self->context->ddEnabledState.get()); + // Confirm the parent is still ours and has not moved on before replacing it. + co_await getBulkLoadTask( + &tr, parent.getRange(), parent.getTaskId(), { BulkLoadPhase::Triggered, BulkLoadPhase::Running }); + for (const auto& child : children) { + co_await krmSetRange(&tr, bulkLoadTaskPrefix, child.getRange(), bulkLoadTaskStateValue(child)); + } + co_await tr.commit(); + TraceEvent(SevWarnAlways, "DDBulkLoadTaskSplit", self->ddId) + .detail("CommitVersion", tr.getCommittedVersion()) + .detail("TaskRange", parent.getRange()) + .detail("TaskID", parent.getTaskId()) + .detail("ManifestCount", manifests.size()) + .detail("FirstRange", children[0].getRange()) + .detail("FirstTaskID", children[0].getTaskId()) + .detail("SecondRange", children[1].getRange()) + .detail("SecondTaskID", children[1].getTaskId()); + break; + } catch (Error& e) { + err = e; + } + if (err.code() == error_code_bulkload_task_outdated) { + // Someone else already moved the parent on; it is no longer ours to split. + co_return false; + } + co_await tr.onError(err); + } + // Ordered after the commit: publishTask refuses an already-published taskId, so the stale parent + // would block its own children, but dropping it before the commit is durable would strand the range + // if the commit failed. + self->bulkLoadTaskCollection->eraseTask(parent); + co_return true; +} + // TODO(BulkLoad): add reason to persist Future failBulkLoadTask(Reference self, KeyRange taskRange, @@ -1249,12 +1398,73 @@ Future doBulkLoadTask(Reference self, KeyRange range, UID // re-dispatch on the next scan. throw timed_out(); } - if (ack.unretryableError) { + // restartCount advances on every re-trigger from any cause -- DD reinit, the backstop above, a + // supplanting trigger -- and is never reset, so this bounds a task's total thrash rather than its + // destination team failures alone. A task that has already burned the budget on unrelated churn + // therefore gets no retry for its first genuine team failure. That is deliberate for now: giving + // up is safe because a job that does not end Complete fails its restore instead of reporting + // completion, so the outcome is a loud restore failure rather than the silent range loss this + // replaced. + int const maxRetryableRedispatch = SERVER_KNOBS->DD_BULKLOAD_MAX_RETRYABLE_REDISPATCH; + bool const retriesExhausted = ack.outcome == BulkLoadAck::Outcome::Retryable && + triggeredBulkLoadTask.restartCount >= maxRetryableRedispatch; + if (ack.outcome == BulkLoadAck::Outcome::Retryable && !retriesExhausted) { + CODE_PROBE(true, "Bulkload task re-dispatched after a recoverable data move failure"); + // Drop this task from the collection before exiting. publishTask refuses a task whose taskId + // is already published, deliberately, to stop a task being triggered twice -- so leaving the + // entry behind makes every later re-dispatch fail as bulkload_task_outdated and the task + // spins between scheduleBulkLoadTasks and publishTask without ever moving data. Persisted + // task state is untouched; only the in-memory publication goes, which is what lets the next + // scan trigger this task again. + self->bulkLoadTaskCollection->eraseTask(triggeredBulkLoadTask); + TraceEvent(SevWarn, "DDBulkLoadTaskDoTask", self->ddId) + .detail("Phase", "See retryable error") + .detail("CancelledDataMovePriority", ack.dataMovePriority) + .detail("Range", range) + .detail("TaskID", taskId) + .detail("RestartCount", triggeredBulkLoadTask.restartCount) + .detail("MaxRetryableRedispatch", maxRetryableRedispatch) + .detail("Duration", now() - beginTime); + throw data_move_dest_team_not_found(); + } + + // A range with no disjoint destination team cannot be placed by re-attempting it: src is recomputed + // from the same range every time. Narrow it instead. Note this arrives with restartCount still 0 -- + // the relocator gives bulkload no retries for this condition -- so it is reached without spending + // any of the re-dispatch budget above. + bool taskSplit = false; + if (ack.outcome == BulkLoadAck::Outcome::Unplaceable || retriesExhausted) { + taskSplit = co_await splitBulkLoadTask(self, triggeredBulkLoadTask); + } + if (taskSplit) { + CODE_PROBE(true, "Bulkload task split because its range could not be placed"); TraceEvent(SevWarnAlways, "DDBulkLoadTaskDoTask", self->ddId) - .detail("Phase", "See unretryable error") + .detail("Phase", "Range could not be placed; task split") + .detail("Outcome", BulkLoadAck::toString(ack.outcome)) + .detail("RetriesExhausted", retriesExhausted) .detail("CancelledDataMovePriority", ack.dataMovePriority) .detail("Range", range) .detail("TaskID", taskId) + .detail("RestartCount", triggeredBulkLoadTask.restartCount) + .detail("Duration", now() - beginTime); + self->bulkLoadEngineParallelismLimitor.decrementTaskCounter(); + co_return; + } + + // An Unplaceable task that could not be narrowed is terminal. Note splitBulkLoadTask also declines + // when the parent is no longer ours, which is not a terminal range: failBulkLoadTask below re-reads + // the task, hits bulkload_task_outdated again, and exits without marking anything Error. + if (ack.outcome == BulkLoadAck::Outcome::Terminal || ack.outcome == BulkLoadAck::Outcome::Unplaceable || + retriesExhausted) { + if (retriesExhausted) { + CODE_PROBE(true, "Bulkload task marked Error after exhausting recoverable retries"); + } + TraceEvent(SevWarnAlways, "DDBulkLoadTaskDoTask", self->ddId) + .detail("Phase", retriesExhausted ? "Retryable error budget exhausted" : "See unretryable error") + .detail("CancelledDataMovePriority", ack.dataMovePriority) + .detail("Range", range) + .detail("TaskID", taskId) + .detail("RestartCount", triggeredBulkLoadTask.restartCount) .detail("Duration", now() - beginTime); try { // Mark this task failed in system metadata @@ -1413,9 +1623,15 @@ Future scheduleBulkLoadTasks(Reference self) { // We do one metadata erase at a time to aviod unnecessary transaction conflicts co_await eraseBulkLoadTask(self, bulkLoadTaskState.getRange(), bulkLoadTaskState.getTaskId()); } else if (bulkLoadTaskState.phase == BulkLoadPhase::Error) { - TraceEvent(SevWarnAlways, "DDBulkLoadTaskUnretriableError", self->ddId) - .detail("Range", bulkLoadTaskState.getRange()) - .detail("TaskID", bulkLoadTaskState.getTaskId()); + // Error is terminal and the metadata is deliberately left in place for the operator, so + // report each failed task once rather than on every rescan: the scan interval is seconds + // and nothing ever clears the entry, so re-logging continues for the cluster's life. + if (self->reportedErrorBulkLoadTasks.insert(bulkLoadTaskState.getTaskId()).second) { + TraceEvent(SevWarnAlways, "DDBulkLoadTaskUnretriableError", self->ddId) + .detail("Range", bulkLoadTaskState.getRange()) + .detail("TaskID", bulkLoadTaskState.getTaskId()) + .detail("TotalErrorTasksReported", self->reportedErrorBulkLoadTasks.size()); + } } else { ASSERT(bulkLoadTaskState.phase == BulkLoadPhase::Complete); } @@ -1512,12 +1728,23 @@ Future getBulkLoadJob(Transaction* tr, UID jobId, KeyRange job } } +// Outcome of looking up the task that owns a range. +struct BulkLoadJobTaskLookup { + // The task owning exactly this range, if one does. + Optional task; + // The range is owned by several narrower tasks, because a task covering it was split. This is neither + // "a task owns this range" nor "this range is unclaimed", and conflating it with either is a bug: the + // executor must not create a task over the range, which would overwrite the narrower ones, and a + // monitor watching the old wide range has nothing left to watch. + bool rangeSplit = false; +}; + // Find task metadata for a bulk load job with jobId and input range -Future> bulkLoadJobFindTask(Reference self, - KeyRange range, - UID jobId, - KeyRange jobRange, - UID logId) { +Future bulkLoadJobFindTask(Reference self, + KeyRange range, + UID jobId, + KeyRange jobRange, + UID logId) { BulkLoadTaskState bulkLoadTaskState; Database cx = self->txnProcessor->context(); Transaction tr(cx); @@ -1530,10 +1757,20 @@ Future> bulkLoadJobFindTask(Reference= 2 && !result[0].value.empty()); + if (result.size() > 2) { + // More than one entry covers the range, so a task over it was split into narrower tasks. + TraceEvent(SevWarnAlways, "DDBulkLoadJobExecutorFindRangeSplit", logId) + .detail("InputRange", range) + .detail("InputJobID", jobId) + .detail("EntryCount", result.size() - 1); + BulkLoadJobTaskLookup lookup; + lookup.rangeSplit = true; + co_return lookup; + } bulkLoadTaskState = decodeBulkLoadTaskState(result[0].value); if (!bulkLoadTaskState.isValid()) { - co_return Optional(); + co_return BulkLoadJobTaskLookup(); } KeyRange currentRange = Standalone(KeyRangeRef(result[0].key, result[1].key)); ASSERT(result[0].key != result[1].key); @@ -1553,7 +1790,9 @@ Future> bulkLoadJobFindTask(Reference bulkLoadJobSubmitTask(Reference self, Future bulkLoadJobWaitUntilTaskCompleteOrError(Reference self, UID jobId, + KeyRange jobRange, BulkLoadTaskState bulkLoadTask) { ASSERT(bulkLoadTask.isValid()); Database cx = self->txnProcessor->context(); @@ -1629,6 +1869,25 @@ Future bulkLoadJobWaitUntilTaskCompleteOrError(Reference hasErr = true; } if (hasErr) { + if (err.code() == error_code_bulkload_task_outdated) { + // The task being watched may have been replaced by narrower tasks covering the same range. + // getBulkLoadTask reports that as outdated, because the range no longer maps to a single + // task, but it is not a failure: the replacements carry the same data and are monitored in + // their own right. Treating it as one fails the whole job, and the restart cancels every + // other monitor, so one split would strand the entire load. + BulkLoadJobTaskLookup lookup = + co_await bulkLoadJobFindTask(self, bulkLoadTask.getRange(), jobId, jobRange, self->ddId); + bool replaced = lookup.rangeSplit || !lookup.task.present() || + lookup.task.get().getTaskId() != bulkLoadTask.getTaskId(); + if (replaced) { + TraceEvent(SevWarnAlways, "DDBulkLoadJobExecutorStopMonitoringReplacedTask", self->ddId) + .detail("InputJobID", jobId) + .detail("TaskRange", bulkLoadTask.getRange()) + .detail("TaskID", bulkLoadTask.getTaskId()) + .detail("RangeSplit", lookup.rangeSplit); + co_return; + } + } co_await tr.onError(err); } co_await delay(SERVER_KNOBS->DD_BULKLOAD_JOB_MONITOR_PERIOD_SEC); @@ -1668,10 +1927,11 @@ Future bulkLoadJobNewTask(Reference self, // Step 2: Check if the task has been created // We define the task range as the range between the min begin key and the max end key of all manifests - Optional bulkLoadTask_ = - co_await bulkLoadJobFindTask(self, taskRange, jobId, jobRange, self->ddId); - if (bulkLoadTask_.present()) { + BulkLoadJobTaskLookup lookup = co_await bulkLoadJobFindTask(self, taskRange, jobId, jobRange, self->ddId); + if (lookup.task.present() || lookup.rangeSplit) { // The task was not existing in the metadata but existing now. So, we need not create the task. + // A split range is likewise already covered, by narrower tasks; creating one here would + // overwrite them with a single task as wide as the range that could not be placed. co_return; } @@ -1735,14 +1995,15 @@ Future bulkLoadJobMonitorTask(Reference self, self->bulkLoadParallelismLimitor.incrementTaskCounter(); try { // Step 1: Check if the task has been created - Optional bulkLoadTask_ = - co_await bulkLoadJobFindTask(self, taskRange, jobId, jobRange, self->ddId); - if (!bulkLoadTask_.present()) { + BulkLoadJobTaskLookup lookup = co_await bulkLoadJobFindTask(self, taskRange, jobId, jobRange, self->ddId); + if (!lookup.task.present()) { // The task was existing in the metadata but now disappear. So, we need not monitor the task. + // The same applies to a range that was split: this monitor was watching one task over the whole + // range, and the narrower tasks that replaced it get their own monitors from the next scan. self->bulkLoadParallelismLimitor.decrementTaskCounter(); co_return; } - bulkLoadTask = bulkLoadTask_.get(); + bulkLoadTask = lookup.task.get(); TraceEvent(bulkLoadVerboseEventSev(), "DDBulkLoadJobExecutorTask", self->ddId) .detail("Phase", "Task found") .detail("JobID", jobId) @@ -1757,7 +2018,7 @@ Future bulkLoadJobMonitorTask(Reference self, } // Step 2: Monitor the bulkload completion - co_await bulkLoadJobWaitUntilTaskCompleteOrError(self, jobId, bulkLoadTask); + co_await bulkLoadJobWaitUntilTaskCompleteOrError(self, jobId, jobRange, bulkLoadTask); TraceEvent(bulkLoadPerfEventSev(), "DDBulkLoadJobExecutorTask", self->ddId) .detail("Phase", "Found task complete") .detail("JobID", jobId) @@ -1766,6 +2027,11 @@ Future bulkLoadJobMonitorTask(Reference self, self->bulkLoadParallelismLimitor.decrementTaskCounter(); } catch (Error& e) { if (e.code() == error_code_actor_cancelled) { + // Release the slot before unwinding. Cancellation is how the job manager tears monitors down + // when it restarts, so skipping the decrement permanently shrinks the parallelism budget: after + // enough restarts scheduleBulkLoadJob blocks forever waiting for a slot no live monitor holds, + // and the job stops progressing with no error reported anywhere. + self->bulkLoadParallelismLimitor.decrementTaskCounter(); throw e; } TraceEvent(SevWarn, "DDBulkLoadJobExecutorTaskMonitorError", self->ddId) @@ -4524,6 +4790,71 @@ Future> getStorageType(std::vector> boundAuditTaskRange(Reference txnProcessor, + UID ddId, + KeyRange shardRange) { + if (SERVER_KNOBS->AUDIT_TASK_MAX_BYTES <= 0 || shardRange.empty()) { + co_return std::vector{ shardRange }; + } + + Optional splitError; + try { + StorageMetrics splitMetrics; + splitMetrics.bytes = SERVER_KNOBS->AUDIT_TASK_MAX_BYTES; + // Split on size only. Setting the other dimensions to infinity is REQUIRED, not merely tidy: + // getSplitKey() does ASSERT(limits > 0), so leaving them at 0 would assert-fail in the storage + // server. Infinity makes getSplitKey's `limits < infinity / 2` test false, which is how a + // dimension is opted out of. Auditing is read-only, so size is the only dimension of interest. + splitMetrics.bytesWrittenPerKSecond = splitMetrics.infinity; + splitMetrics.iosPerKSecond = splitMetrics.infinity; + splitMetrics.bytesReadPerKSecond = splitMetrics.infinity; + + // minSplitBytes must not exceed half the target. The storage server stops splitting once + // `remaining.bytes < 2 * minSplitBytes`, so passing the target itself leaves every shard below + // TWICE the target whole. + // + // Anything below target/2 behaves the same, since getSplitKey() only splits while + // remaining > target/2 and that binds first; half the target just avoids generating split points + // the client will discard. + const int minSplitBytes = + std::max(1, + static_cast(std::min(std::numeric_limits::max(), + SERVER_KNOBS->AUDIT_TASK_MAX_BYTES / 2))); + Standalone> splitPoints = + co_await txnProcessor->splitStorageMetrics(shardRange, splitMetrics, StorageMetrics(), minSplitBytes); + // splitStorageMetrics() returns shardRange.begin and shardRange.end as its first and last points, + // in order, so consecutive pairs tile the range. Same idiom as DDShardTracker's executeShardSplit(). + std::vector taskRanges; + for (int i = 0; i + 1 < splitPoints.size(); ++i) { + taskRanges.push_back(KeyRangeRef(splitPoints[i], splitPoints[i + 1])); + } + ASSERT(!taskRanges.empty()); + if (taskRanges.size() > 1) { + TraceEvent(SevInfo, "DDAuditTaskRangeSubdivided", ddId) + .suppressFor(30.0) + .detail("ShardRange", shardRange) + .detail("NumTaskRanges", taskRanges.size()) + .detail("MaxBytesPerTask", SERVER_KNOBS->AUDIT_TASK_MAX_BYTES); + } + co_return taskRanges; + } catch (Error& e) { + if (e.code() == error_code_actor_cancelled) { + throw e; + } + splitError = e; + } + // A metrics read must never fail an audit: fall back to the unsplit shard, which is exactly the + // behaviour before this cap existed. Worst case we keep the straggler we were trying to avoid. + TraceEvent(SevWarn, "DDAuditTaskRangeSplitFailed", ddId) + .errorUnsuppressed(splitError.get()) + .detail("ShardRange", shardRange); + co_return std::vector{ shardRange }; +} + // Partition the input range into multiple subranges according to the range ownership, and // schedule ha/replica/restore audit tasks of each subrange on the server which owns the subrange // Automatically retry until complete or timed out @@ -4543,6 +4874,9 @@ Future scheduleAuditOnRange(Reference self, Key currentRangeToScheduleBegin = rangeToSchedule.begin; KeyRange currentRangeToSchedule; int64_t issueDoAuditCount = 0; + // Counts skipped task ranges, not distinct shards: the engine-type skip has always been per audit + // state range, and a shard subdivided by AUDIT_TASK_MAX_BYTES contributes one count per piece. Kept + // under its existing name so the SkippedShardsCountInThisSchedule trace field stays greppable. int64_t numSkippedShards = 0; try { @@ -4560,12 +4894,32 @@ Future scheduleAuditOnRange(Reference self, .detail("RangeLocationsBackKey", rangeLocations.back().range.end); } - // Divide the audit job in to tasks according to KeyServers system mapping + // Divide the audit job in to tasks according to KeyServers system mapping, then subdivide + // any shard larger than AUDIT_TASK_MAX_BYTES. A task is scanned sequentially by one storage + // server and the audit phase ends only when its slowest task does, so an uncapped fat shard + // bounds the whole phase from below. int assignedRangeTasks = 0; - int rangeLocationIndex = 0; - for (; rangeLocationIndex < rangeLocations.size(); ++rangeLocationIndex) { + // Split lazily, one shard at a time, so the first task dispatches immediately: splitting + // every shard up front would put thousands of sequential metrics reads ahead of any audit + // work. Entries are (task range, index of the shard it came from) -- the shard index is + // needed because the servers to audit come from that rangeLocations entry. + std::vector> taskRangesToSchedule; + int nextShardToSplit = 0; + for (int taskIndex = 0;; ++taskIndex) { + while (taskIndex >= taskRangesToSchedule.size() && nextShardToSplit < rangeLocations.size()) { + std::vector boundedRanges = co_await boundAuditTaskRange( + self->txnProcessor, self->ddId, rangeLocations[nextShardToSplit].range); + for (KeyRange const& boundedRange : boundedRanges) { + taskRangesToSchedule.emplace_back(boundedRange, nextShardToSplit); + } + ++nextShardToSplit; + } + if (taskIndex >= taskRangesToSchedule.size()) { + break; // every shard split and every resulting task scheduled + } // For each task, check the progress, and create task request for the unfinished range - KeyRange taskRange = rangeLocations[rangeLocationIndex].range; + KeyRange taskRange = taskRangesToSchedule[taskIndex].first; + const int rangeLocationIndex = taskRangesToSchedule[taskIndex].second; if (SERVER_KNOBS->ENABLE_AUDIT_VERBOSE_TRACE) { TraceEvent(SevInfo, "DDScheduleAuditOnCurrentRangeTask", self->ddId) .detail("AuditID", audit->coreState.id) @@ -4790,7 +5144,15 @@ Future scheduleAuditOnRange(Reference self, TraceEvent(SevInfo, "DDScheduleAuditOnCurrentRangeTaskAssigned", self->ddId); } ++assignedRangeTasks; - co_await delay(0.1); + // Throttle once per keyServers shard, NOT once per subdivided piece: per piece this + // multiplies by the split factor, adding a term linear in the piece count to the very + // wall-clock the cap exists to shorten. Pieces are appended a whole shard at a time, so + // the last present entry is always its shard's last piece. + const bool lastPieceOfShard = taskIndex + 1 >= taskRangesToSchedule.size() || + taskRangesToSchedule[taskIndex + 1].second != rangeLocationIndex; + if (lastPieceOfShard) { + co_await delay(0.1); + } } // Proceed to the next range if getSourceServerInterfacesForRange is partially read currentRangeToScheduleBegin = rangeLocations.back().range.end; @@ -5343,6 +5705,123 @@ inline int getRandomShardCount() { } // namespace data_distribution_test +namespace { + +// Drives boundAuditTaskRange() without a cluster: returns the configured split points, or fails the +// metrics read when asked to. +class AuditSplitTestTxnProcessor final : public DDTxnProcessor { +public: + explicit AuditSplitTestTxnProcessor(std::vector points, Optional failWith = {}) + : points(std::move(points)), failWith(failWith) {} + + int getSplitCalls() const { return splitCalls; } + int64_t getLastMinSplitBytes() const { return lastMinSplitBytes; } + + Future>> splitStorageMetrics(KeyRange const& keys, + StorageMetrics const&, + StorageMetrics const&, + Optional const& minSplitBytes) const override { + ++splitCalls; + lastMinSplitBytes = minSplitBytes.present() ? minSplitBytes.get() : -1; + if (failWith.present()) { + return failWith.get(); + } + Standalone> splitKeys; + for (std::string const& p : points) { + splitKeys.push_back_deep(splitKeys.arena(), StringRef(p)); + } + return splitKeys; + } + +private: + std::vector points; + Optional failWith; + mutable int splitCalls = 0; + mutable int64_t lastMinSplitBytes = 0; +}; + +// Restores AUDIT_TASK_MAX_BYTES on any exit path. Unit tests share a process, so a knob left at a +// test value by a failed ASSERT silently reconfigures every test that runs afterwards, turning one +// failure into a cascade that no longer points at its cause. +class AuditTaskMaxBytesGuard { +public: + AuditTaskMaxBytesGuard() : saved(SERVER_KNOBS->AUDIT_TASK_MAX_BYTES) {} + ~AuditTaskMaxBytesGuard() { set(saved); } + AuditTaskMaxBytesGuard(AuditTaskMaxBytesGuard const&) = delete; + AuditTaskMaxBytesGuard& operator=(AuditTaskMaxBytesGuard const&) = delete; + + void set(int64_t value) { const_cast(SERVER_KNOBS)->AUDIT_TASK_MAX_BYTES = value; } + +private: + int64_t saved; +}; + +} // namespace + +TEST_CASE("/DataDistribution/Audit/BoundAuditTaskRange") { + KeyRange shard = KeyRangeRef("a"_sr, "z"_sr); + AuditTaskMaxBytesGuard maxBytes; + + // A fat shard is subdivided at the reported split points. + { + maxBytes.set(1000); + auto processor = makeReference(std::vector{ "a", "m", "z" }); + std::vector ranges = co_await boundAuditTaskRange(processor, UID(), shard); + ASSERT_EQ(ranges.size(), 2); + ASSERT(ranges[0] == KeyRangeRef("a"_sr, "m"_sr)); + ASSERT(ranges[1] == KeyRangeRef("m"_sr, "z"_sr)); + ASSERT_EQ(processor->getSplitCalls(), 1); + + // REGRESSION: minSplitBytes above target/2 makes the storage server leave shards below + // 2 * minSplitBytes whole. Silent when wrong -- the only symptom is a slow tail that looks like + // ordinary variance. + ASSERT(processor->getLastMinSplitBytes() > 0); + ASSERT(processor->getLastMinSplitBytes() * 2 <= SERVER_KNOBS->AUDIT_TASK_MAX_BYTES); + } + + // A shard smaller than the cap reports no interior points and stays whole. + { + maxBytes.set(1000); + auto processor = makeReference(std::vector{ "a", "z" }); + std::vector ranges = co_await boundAuditTaskRange(processor, UID(), shard); + ASSERT_EQ(ranges.size(), 1); + ASSERT(ranges[0] == shard); + } + + // Knob disabled: no metrics read at all, and the shard is used as-is. + { + maxBytes.set(0); + auto processor = makeReference(std::vector{ "a", "m", "z" }); + std::vector ranges = co_await boundAuditTaskRange(processor, UID(), shard); + ASSERT_EQ(ranges.size(), 1); + ASSERT(ranges[0] == shard); + ASSERT_EQ(processor->getSplitCalls(), 0); + } + + // A failed metrics read must degrade to the unsplit shard rather than fail the audit: an audit that + // dies because a size estimate was unavailable is strictly worse than one straggler task. + { + maxBytes.set(1000); + auto processor = + makeReference(std::vector{}, Optional(timed_out())); + std::vector ranges = co_await boundAuditTaskRange(processor, UID(), shard); + ASSERT_EQ(ranges.size(), 1); + ASSERT(ranges[0] == shard); + } + + // A cap small enough that target/2 would round to zero must still send a usable minSplitBytes: a + // zero or negative value would fall back to MIN_SHARD_BYTES on the server, silently ignoring the cap. + { + maxBytes.set(1); + auto processor = makeReference(std::vector{ "a", "m", "z" }); + std::vector ranges = co_await boundAuditTaskRange(processor, UID(), shard); + ASSERT_EQ(ranges.size(), 2); + ASSERT(processor->getLastMinSplitBytes() >= 1); + } + + co_return; +} + TEST_CASE("/DataDistribution/Initialization/DcIds") { RegionInfo configuredPrimary; configuredPrimary.dcId = "primary"_sr; diff --git a/fdbserver/datadistributor/DataDistribution.h b/fdbserver/datadistributor/DataDistribution.h index 0d326d9d462..268e90d92c3 100644 --- a/fdbserver/datadistributor/DataDistribution.h +++ b/fdbserver/datadistributor/DataDistribution.h @@ -522,16 +522,56 @@ struct DDBulkLoadTaskBusyMap { std::unordered_map busyMap; // }; -// Used to piggyback the data move priority when an unretryable error happens to the task datamove. +// Used to piggyback the data move priority when a bulkload task's data move ends. // If the priority indicates the data move is a team unhealthy related data move, the bulkload engine // system trigger a new data move when terminate the error task. struct BulkLoadAck { - bool unretryableError = false; + // How the data move ended, from the bulkload engine's point of view. Exactly one outcome applies, so + // the engine cannot mistake one for another. + enum class Outcome { + // The move completed, or the task was superseded with nothing to report. + Completed = 0, + // Failed on a condition a later attempt can clear -- e.g. the destination team was momentarily + // unhealthy. The task must stay eligible for re-dispatch rather than be marked Error. + Retryable, + // No destination team could be found for long enough that re-attempting the same range is not + // worth it (see the BestTeamStuck path in dataDistributionRelocator). The engine's response is to + // narrow the range, and to mark Error only once the task is too narrow to split. + // + // Narrowing is a remedy for one of the reasons a team cannot be found: a destination team must be + // disjoint from src, src is the union of the owners of every shard the range spans, and so a range + // spanning enough of the fleet has no legal destination however often it is attempted. The + // relocator cannot distinguish that from the other reasons getTeamForBulkLoad returns nothing -- + // every team unhealthy, or none with disk headroom -- which narrowing does not address. Splitting + // on those is wasted work rather than harmful: the children tile the parent and carry the same + // manifests, so nothing is lost, and the split count is bounded by the manifest count. + Unplaceable, + // Failed in a way neither a later attempt nor a narrower range can clear, so the task is marked + // Error and its data is never ingested. Reserve this for genuinely terminal conditions: a bulkload + // task's data lives only in the dump until some attempt succeeds, so calling a recoverable failure + // terminal drops that data with no other copy in the cluster. + Terminal, + }; + + Outcome outcome = Outcome::Completed; int dataMovePriority = -1; BulkLoadAck() = default; - BulkLoadAck(bool unretryableError, int dataMovePriority) - : unretryableError(unretryableError), dataMovePriority(dataMovePriority) {} + BulkLoadAck(Outcome outcome, int dataMovePriority) : outcome(outcome), dataMovePriority(dataMovePriority) {} + + static std::string toString(Outcome outcome) { + switch (outcome) { + case Outcome::Completed: + return "Completed"; + case Outcome::Retryable: + return "Retryable"; + case Outcome::Unplaceable: + return "Unplaceable"; + case Outcome::Terminal: + return "Terminal"; + } + UNREACHABLE(); + } }; struct DDBulkLoadEngineTask { diff --git a/fdbserver/fdbserver.cpp b/fdbserver/fdbserver.cpp index 6211c3e655f..2edceb6ea41 100644 --- a/fdbserver/fdbserver.cpp +++ b/fdbserver/fdbserver.cpp @@ -243,9 +243,7 @@ CSimpleOpt::SOption g_rgOptions[] = { // clang-format on -extern void dsltest(); extern void pingtest(); -extern void copyTest(); extern void versionedMapTest(); extern void createTemplateDatabase(); @@ -940,7 +938,6 @@ enum class ServerRole { ConsistencyCheck, ConsistencyCheckUrgent, CreateTemplateDatabase, - DSLTest, FDBD, KVFileGenerateIOLogChecksums, KVFileIntegrityCheck, @@ -1226,8 +1223,6 @@ struct CLIOptions { role = ServerRole::SkipListTest; else if (!strcmp(sRole, "search")) role = ServerRole::SearchMutations; - else if (!strcmp(sRole, "dsltest")) - role = ServerRole::DSLTest; else if (!strcmp(sRole, "versionedmaptest")) role = ServerRole::VersionedMapTest; else if (!strcmp(sRole, "createtemplatedb")) @@ -1936,11 +1931,6 @@ int main(int argc, char* argv[]) { flushAndExit(FDB_EXIT_SUCCESS); } - if (role == ServerRole::DSLTest) { - dsltest(); - flushAndExit(FDB_EXIT_SUCCESS); - } - if (role == ServerRole::VersionedMapTest) { versionedMapTest(); flushAndExit(FDB_EXIT_SUCCESS); diff --git a/fdbserver/kvstore/KeyValueStoreRocksDB.cpp b/fdbserver/kvstore/KeyValueStoreRocksDB.cpp index f1e1e5e884a..048cb6c66cc 100644 --- a/fdbserver/kvstore/KeyValueStoreRocksDB.cpp +++ b/fdbserver/kvstore/KeyValueStoreRocksDB.cpp @@ -1432,6 +1432,7 @@ struct RocksDBKeyValueStore : IKeyValueStore { rocksdb::ColumnFamilyOptions cfOptions = sharedState->getCfOptions(); std::vector descriptors; + descriptors.reserve(columnFamilies.size()); for (const std::string& name : columnFamilies) { descriptors.push_back(rocksdb::ColumnFamilyDescriptor{ name, cfOptions }); } @@ -1648,6 +1649,7 @@ struct RocksDBKeyValueStore : IKeyValueStore { std::set columnFamilies{ "default" }; columnFamilies.insert(SERVER_KNOBS->DEFAULT_FDB_ROCKSDB_COLUMN_FAMILY); std::vector descriptors; + descriptors.reserve(columnFamilies.size()); for (const std::string name : columnFamilies) { descriptors.push_back(rocksdb::ColumnFamilyDescriptor{ name, sharedState->getCfOptions() }); } diff --git a/fdbserver/kvstore/KeyValueStoreShardedRocksDB.cpp b/fdbserver/kvstore/KeyValueStoreShardedRocksDB.cpp index ca9d88fb419..721e4983e98 100644 --- a/fdbserver/kvstore/KeyValueStoreShardedRocksDB.cpp +++ b/fdbserver/kvstore/KeyValueStoreShardedRocksDB.cpp @@ -1936,6 +1936,7 @@ class ShardManager { std::vector getColumnFamilies() { std::vector res; + res.reserve(columnFamilyMap.size()); for (auto& [id, cf] : columnFamilyMap) { res.push_back(cf); } @@ -4939,7 +4940,7 @@ TEST_CASE("noSim/determinism/checkpoint_metadata/serde3") { } TEST_CASE("noSim/determinism/checkpoint_metadata/serde4") { - checkpointMetadataSerdeTest("some_\nrocksdb_checkpoint_\nmetadata_123\0"); + checkpointMetadataSerdeTest("some_\nrocksdb_checkpoint_\nmetadata_123\0"_sr.toString()); return Void(); } diff --git a/fdbserver/kvstore/VersionedBTree.cpp b/fdbserver/kvstore/VersionedBTree.cpp index 5814c3f0b8c..961967ce692 100644 --- a/fdbserver/kvstore/VersionedBTree.cpp +++ b/fdbserver/kvstore/VersionedBTree.cpp @@ -56,6 +56,7 @@ #include #include #include +#include #include #include @@ -1898,7 +1899,7 @@ class ObjectCache : NonCopyable { }; template -Future forwardError(Future f, Promise target, ExplicitVoid = {}) { +Future forwardErrorSlow(Future f, Promise target, ExplicitVoid = {}) { try { T x = co_await f; co_return x; @@ -1913,6 +1914,57 @@ Future forwardError(Future f, Promise target, ExplicitVoid = {}) { } } +template +Future forwardError(Future f, Promise target) { + if (!f.isReady()) { + return forwardErrorSlow(std::move(f), std::move(target)); + } + if (f.isError()) { + Error error = f.getError(); + if (error.code() != error_code_actor_cancelled && target.canBeSet()) { + target.sendError(error); + } + } + return f; +} + +TEST_CASE("/fdbserver/kvstore/forwardError") { + { + Promise target; + Future result = forwardError(Future(42), target); + ASSERT(result.isReady() && !result.isError()); + ASSERT_EQ(result.get(), 42); + ASSERT(target.canBeSet()); + } + { + Promise target; + Future error = target.getFuture(); + Future result = forwardError(Future(io_error()), target); + ASSERT(result.isError()); + ASSERT(error.isError()); + ASSERT_EQ(error.getError().code(), error_code_io_error); + } + { + Promise source; + Promise target; + Future error = target.getFuture(); + Future result = forwardError(source.getFuture(), target); + ASSERT(!result.isReady()); + source.sendError(io_error()); + ASSERT(result.isError()); + ASSERT(error.isError()); + ASSERT_EQ(error.getError().code(), error_code_io_error); + } + { + Promise source; + Promise target; + Future result = forwardError(source.getFuture(), target); + result.cancel(); + ASSERT(target.canBeSet()); + } + return Void(); +} + constexpr int initialVersion = invalidVersion; class DWALPagerSnapshot; @@ -1932,9 +1984,54 @@ class DWALPager final : public IPager2 { using VersionToPageMapT = std::map; using PageToVersionedMapT = std::unordered_map; struct PageCacheEntry { + template + using PageFuture = + std::conditional_t>, Future>>; + Future> readFuture; + Future> readOnlyFuture; Future writeFuture; + static Future> makeReadOnlyFutureSlow(Future> future) { + Reference page = co_await future; + co_return Reference(std::move(page)); + } + + template + static PageFuture makePageFuture(Future> future) { + if constexpr (!ReadOnly) { + return future; + } else { + if (!future.isReady()) { + return makeReadOnlyFutureSlow(std::move(future)); + } + if (future.isError()) { + return Future>(future.getError()); + } + return Future>(Reference(future.get())); + } + } + + template + PageFuture getReadFuture() { + if constexpr (!ReadOnly) { + return readFuture; + } else { + if (!readFuture.isReady()) { + return makeReadOnlyFutureSlow(readFuture); + } + if (!readOnlyFuture.isValid()) { + readOnlyFuture = makePageFuture(readFuture); + } + return readOnlyFuture; + } + } + + void setReadFuture(Future> future) { + readOnlyFuture = Future>(); + readFuture = std::move(future); + } + bool initialized() const { return readFuture.isValid(); } bool reading() const { return !readFuture.isReady(); } @@ -1953,6 +2050,7 @@ class DWALPager final : public IPager2 { // Read and write futures are safe to cancel so just cancel them and return Future cancel() { writeFuture.cancel(); + readOnlyFuture.cancel(); readFuture.cancel(); return Void(); } @@ -2669,7 +2767,7 @@ class DWALPager final : public IPager2 { } // Always update the page contents immediately regardless of what happened above. - cacheEntry.readFuture = data; + cacheEntry.setReadFuture(data); } Future atomicUpdatePage(PagerEventReasons reason, @@ -2891,6 +2989,7 @@ class DWALPager final : public IPager2 { // TODO: Could a dispatched read try to write to page after it has been destroyed if this actor is cancelled? int blockSize = self->physicalPageSize; std::vector> reads; + reads.reserve(pageIDs.size()); for (int i = 0; i < pageIDs.size(); ++i) { reads.push_back( readPhysicalBlock(self, page, i * blockSize, blockSize, ((int64_t)pageIDs[i]) * blockSize, priority)); @@ -2945,6 +3044,16 @@ class DWALPager final : public IPager2 { int priority, bool cacheable, bool noHit) override { + return readPageImpl(reason, level, pageID, priority, cacheable, noHit); + } + + template + PageCacheEntry::PageFuture readPageImpl(PagerEventReasons reason, + unsigned int level, + PhysicalPageID pageID, + int priority, + bool cacheable, + bool noHit) { // Use cached page if present, without triggering a cache hit. // Otherwise, read the page and return it but don't add it to the cache debug_printf("DWALPager(%s) op=read %s reason=%s noHit=%d\n", @@ -2960,11 +3069,12 @@ class DWALPager final : public IPager2 { if (pCacheEntry != nullptr) { ++g_redwoodMetrics.metric.pagerProbeHit; debug_printf("DWALPager(%s) op=readUncachedHit %s\n", filename.c_str(), toString(pageID).c_str()); - return pCacheEntry->readFuture; + return pCacheEntry->getReadFuture(); } ++g_redwoodMetrics.metric.pagerProbeMiss; debug_printf("DWALPager(%s) op=readUncachedMiss %s\n", filename.c_str(), toString(pageID).c_str()); - return forwardError(readPhysicalPage(this, pageID, priority, false, reason), errorPromise); + return PageCacheEntry::makePageFuture( + forwardError(readPhysicalPage(this, pageID, priority, false, reason), errorPromise)); } PageCacheEntry& cacheEntry = pageCache.get(pageID, physicalPageSize, noHit); debug_printf("DWALPager(%s) op=read %s cached=%d reading=%d writing=%d noHit=%d\n", @@ -2976,7 +3086,8 @@ class DWALPager final : public IPager2 { noHit); if (!cacheEntry.initialized()) { debug_printf("DWALPager(%s) issuing actual read of %s\n", filename.c_str(), toString(pageID).c_str()); - cacheEntry.readFuture = forwardError(readPhysicalPage(this, pageID, priority, false, reason), errorPromise); + cacheEntry.setReadFuture( + forwardError(readPhysicalPage(this, pageID, priority, false, reason), errorPromise)); cacheEntry.writeFuture = Void(); ++g_redwoodMetrics.metric.pagerCacheMiss; @@ -2985,7 +3096,7 @@ class DWALPager final : public IPager2 { ++g_redwoodMetrics.metric.pagerCacheHit; eventReasons.addEventReason(PagerEvents::CacheHit, reason); } - return cacheEntry.readFuture; + return cacheEntry.getReadFuture(); } Future> readMultiPage(PagerEventReasons reason, @@ -2994,6 +3105,16 @@ class DWALPager final : public IPager2 { int priority, bool cacheable, bool noHit) override { + return readMultiPageImpl(reason, level, pageIDs, priority, cacheable, noHit); + } + + template + PageCacheEntry::PageFuture readMultiPageImpl(PagerEventReasons reason, + unsigned int level, + VectorRef pageIDs, + int priority, + bool cacheable, + bool noHit) { // Use cached page if present, without triggering a cache hit. // Otherwise, read the page and return it but don't add it to the cache debug_printf("DWALPager(%s) op=read %s reason=%s noHit=%d\n", @@ -3009,11 +3130,12 @@ class DWALPager final : public IPager2 { if (pCacheEntry != nullptr) { ++g_redwoodMetrics.metric.pagerProbeHit; debug_printf("DWALPager(%s) op=readUncachedHit %s\n", filename.c_str(), toString(pageIDs).c_str()); - return pCacheEntry->readFuture; + return pCacheEntry->getReadFuture(); } ++g_redwoodMetrics.metric.pagerProbeMiss; debug_printf("DWALPager(%s) op=readUncachedMiss %s\n", filename.c_str(), toString(pageIDs).c_str()); - return forwardError(readPhysicalMultiPage(this, pageIDs, priority, reason), errorPromise); + return PageCacheEntry::makePageFuture( + forwardError(readPhysicalMultiPage(this, pageIDs, priority, reason), errorPromise)); } PageCacheEntry& cacheEntry = pageCache.get(pageIDs.front(), pageIDs.size() * physicalPageSize, noHit); @@ -3026,7 +3148,8 @@ class DWALPager final : public IPager2 { noHit); if (!cacheEntry.initialized()) { debug_printf("DWALPager(%s) issuing actual read of %s\n", filename.c_str(), toString(pageIDs).c_str()); - cacheEntry.readFuture = forwardError(readPhysicalMultiPage(this, pageIDs, priority, reason), errorPromise); + cacheEntry.setReadFuture( + forwardError(readPhysicalMultiPage(this, pageIDs, priority, reason), errorPromise)); cacheEntry.writeFuture = Void(); ++g_redwoodMetrics.metric.pagerCacheMiss; @@ -3035,7 +3158,7 @@ class DWALPager final : public IPager2 { ++g_redwoodMetrics.metric.pagerCacheHit; eventReasons.addEventReason(PagerEvents::CacheHit, reason); } - return cacheEntry.readFuture; + return cacheEntry.getReadFuture(); } PhysicalPageID getPhysicalPageID(LogicalPageID pageID, Version v) { @@ -3068,15 +3191,24 @@ class DWALPager final : public IPager2 { return (PhysicalPageID)pageID; } - Future> readPageAtVersion(PagerEventReasons reason, - unsigned int level, - LogicalPageID logicalID, - int priority, - Version v, - bool cacheable, - bool noHit) { + Future> readPageAtVersion(PagerEventReasons reason, + unsigned int level, + LogicalPageID logicalID, + int priority, + Version v, + bool cacheable, + bool noHit) { PhysicalPageID physicalID = getPhysicalPageID(logicalID, v); - return readPage(reason, level, physicalID, priority, cacheable, noHit); + return readPageImpl(reason, level, physicalID, priority, cacheable, noHit); + } + + Future> readMultiPageAtSnapshot(PagerEventReasons reason, + unsigned int level, + VectorRef pageIDs, + int priority, + bool cacheable, + bool noHit) { + return readMultiPageImpl(reason, level, pageIDs, priority, cacheable, noHit); } void releaseExtentReadLock() override { concurrentExtentReads->release(); } @@ -3193,8 +3325,8 @@ class DWALPager final : public IPager2 { PageCacheEntry& cacheEntry = extentCache.get(pageID, 1); if (!cacheEntry.initialized()) { cacheEntry.writeFuture = Void(); - cacheEntry.readFuture = - forwardError(readPhysicalExtent(this, (PhysicalPageID)pageID, readSize), errorPromise); + cacheEntry.setReadFuture( + forwardError(readPhysicalExtent(this, (PhysicalPageID)pageID, readSize), errorPromise)); debug_printf("DWALPager(%s) Set the cacheEntry readFuture for page: %s\n", filename.c_str(), toString(pageID).c_str()); @@ -3941,9 +4073,7 @@ class DWALPagerSnapshot : public IPagerSnapshot, public ReferenceCounted page = - co_await pager->readPageAtVersion(reason, level, pageID, priority, version, cacheable, noHit); - co_return Reference(std::move(page)); + return pager->readPageAtVersion(reason, level, pageID, priority, version, cacheable, noHit); } Future> getMultiPhysicalPage(PagerEventReasons reason, @@ -3953,8 +4083,7 @@ class DWALPagerSnapshot : public IPagerSnapshot, public ReferenceCounted page = co_await pager->readMultiPage(reason, level, pageIDs, priority, cacheable, noHit); - co_return Reference(std::move(page)); + return pager->readMultiPageAtSnapshot(reason, level, pageIDs, priority, cacheable, noHit); } Key getMetaKey() const override { return metaKey; } @@ -5902,6 +6031,22 @@ class VersionedBTree { co_return records; } + static void recordPageRead(const Reference& page, int pageReadExt) { + const BTreePage* btPage = (const BTreePage*)page->data(); + auto& metrics = g_redwoodMetrics.level(btPage->height).metrics; + metrics.pageRead += 1; + metrics.pageReadExt += pageReadExt; + } + + static Future> finishReadPage(Future> pageFuture, + BTreeNodeLink id, + Version snapshotVersion) { + Reference page = co_await pageFuture; + debug_printf("readPage() op=readComplete %s @%" PRId64 " \n", toString(id).c_str(), snapshotVersion); + recordPageRead(page, id.size() - 1); + co_return page; + } + static Future> readPage(VersionedBTree* self, PagerEventReasons reason, unsigned int level, @@ -5916,24 +6061,24 @@ class VersionedBTree { toString(id).c_str(), snapshot->getVersion()); - Reference page; + Future> pageFuture; if (id.size() == 1) { - Reference p = - co_await snapshot->getPhysicalPage(reason, level, id.front(), priority, cacheable, false); - page = std::move(p); + pageFuture = snapshot->getPhysicalPage(reason, level, id.front(), priority, cacheable, false); } else { ASSERT(!id.empty()); - Reference p = - co_await snapshot->getMultiPhysicalPage(reason, level, id, priority, cacheable, false); - page = std::move(p); + pageFuture = snapshot->getMultiPhysicalPage(reason, level, id, priority, cacheable, false); } - debug_printf("readPage() op=readComplete %s @%" PRId64 " \n", toString(id).c_str(), snapshot->getVersion()); - const BTreePage* btPage = (const BTreePage*)page->data(); - auto& metrics = g_redwoodMetrics.level(btPage->height).metrics; - metrics.pageRead += 1; - metrics.pageReadExt += (id.size() - 1); - co_return page; + if (pageFuture.isReady()) { + if (!pageFuture.isError()) { + debug_printf( + "readPage() op=readComplete %s @%" PRId64 " \n", toString(id).c_str(), snapshot->getVersion()); + recordPageRead(pageFuture.get(), id.size() - 1); + } + return pageFuture; + } + + return finishReadPage(std::move(pageFuture), id, snapshot->getVersion()); } // Get cursor into a BTree node, creating decode cache from boundaries if needed @@ -7338,45 +7483,80 @@ class VersionedBTree { PathEntry& back() { return path.back(); } void popPath() { path.pop_back(); } - Future pushPage(const BTreePage::BinaryTree::Cursor& link) { - debug_printf("pushPage(link=%s)\n", link.get().toString(false).c_str()); - BTreePage::BinaryTree::Cursor linkCopy = link; - Reference p = - co_await readPage(btree, - reason, - path.back().btPage()->height - 1, - pager.getPtr(), - linkCopy.get().getChildPage(), - ioMaxPriority, - false, - !options.present() || options.get().cacheResult || path.back().btPage()->height != 2); - BTreePage::BinaryTree::Cursor cursor = btree->getCursor(p.getPtr(), linkCopy); + void appendChildPage(const BTreePage::BinaryTree::Cursor& link, Reference page) { + BTreePage::BinaryTree::Cursor cursor = btree->getCursor(page.getPtr(), link); #if REDWOOD_DEBUG - path.push_back({ p, cursor, linkCopy.get().getChildPage() }); + path.push_back({ std::move(page), cursor, link.get().getChildPage() }); #else - path.push_back({ p, cursor }); + path.push_back({ std::move(page), cursor }); #endif if (btree->m_pBoundaryVerifier != nullptr) { - ASSERT(btree->m_pBoundaryVerifier->verify(linkCopy.get().getChildPage().front(), + ASSERT(btree->m_pBoundaryVerifier->verify(link.get().getChildPage().front(), pager->getVersion(), - linkCopy.get().key, - linkCopy.next().getOrUpperBound().key, + link.get().key, + link.next().getOrUpperBound().key, cursor)); } } - Future pushPage(BTreeNodeLinkRef id) { - debug_printf("pushPage(root=%s)\n", ::toString(id).c_str()); - Reference p = co_await readPage( - btree, reason, btree->m_header.height, pager.getPtr(), id, ioMaxPriority, false, true); + Future pushChildPageSlow(Future> pageFuture, + BTreePage::BinaryTree::Cursor link) { + Reference page = co_await pageFuture; + appendChildPage(link, std::move(page)); + } + + Future pushPage(const BTreePage::BinaryTree::Cursor& link) { + debug_printf("pushPage(link=%s)\n", link.get().toString(false).c_str()); + BTreePage::BinaryTree::Cursor linkCopy = link; + Future> pageFuture = + readPage(btree, + reason, + path.back().btPage()->height - 1, + pager.getPtr(), + linkCopy.get().getChildPage(), + ioMaxPriority, + false, + !options.present() || options.get().cacheResult || path.back().btPage()->height != 2); + if (!pageFuture.isReady()) { + return pushChildPageSlow(std::move(pageFuture), linkCopy); + } + if (pageFuture.isError()) { + return pageFuture.getError(); + } + appendChildPage(linkCopy, pageFuture.get()); + return Void(); + } + + void appendRootPage(Reference page, BTreeNodeLinkRef id) { #if REDWOOD_DEBUG - path.push_back({ p, btree->getCursor(p.getPtr(), dbBegin, dbEnd), id }); + auto cursor = btree->getCursor(page.getPtr(), dbBegin, dbEnd); + path.push_back({ std::move(page), cursor, id }); #else - path.push_back({ p, btree->getCursor(p.getPtr(), dbBegin, dbEnd) }); + auto cursor = btree->getCursor(page.getPtr(), dbBegin, dbEnd); + path.push_back({ std::move(page), cursor }); #endif } + Future pushRootPageSlow(Future> pageFuture, BTreeNodeLink id) { + Reference page = co_await pageFuture; + appendRootPage(std::move(page), id); + } + + Future pushPage(BTreeNodeLinkRef id) { + debug_printf("pushPage(root=%s)\n", ::toString(id).c_str()); + Future> pageFuture = + readPage(btree, reason, btree->m_header.height, pager.getPtr(), id, ioMaxPriority, false, true); + if (!pageFuture.isReady()) { + return pushRootPageSlow(std::move(pageFuture), id); + } + if (pageFuture.isError()) { + return pageFuture.getError(); + } + appendRootPage(pageFuture.get(), id); + return Void(); + } + // Initialize or reinitialize cursor Future init(VersionedBTree* btree_in, PagerEventReasons reason_in, @@ -7386,7 +7566,7 @@ class VersionedBTree { btree = btree_in; reason = reason_in; options = options_in; - pager = pager_in; + pager = std::move(pager_in); path.clear(); path.reserve(6); valid = false; @@ -7449,6 +7629,45 @@ class VersionedBTree { Future seekGTE(RedwoodRecordRef query) { return seekGTE_impl(this, query); } + Future seekExactKeySlow(Key key, Future pageFuture) { + co_await pageFuture; + co_await seekExactKeyFromCurrent(key); + } + + Future seekExactKeyFromCurrent(KeyRef key) { + RedwoodRecordRef query(key); + RedwoodRecordRef internalPageQuery = query.withMaxPageID(); + + while (true) { + auto& entry = path.back(); + if (entry.btPage()->isLeaf()) { + valid = entry.cursor.seekGreaterThanOrEqual(query) && entry.cursor.get().key == query.key; + return Void(); + } + + if (!entry.cursor.seekLessThan(internalPageQuery) || !entry.cursor.get().value.present()) { + valid = false; + return Void(); + } + + Future pageFuture = pushPage(entry.cursor); + if (!pageFuture.isReady()) { + return seekExactKeySlow(Key(key), std::move(pageFuture)); + } + if (pageFuture.isError()) { + return pageFuture.getError(); + } + } + } + + Future seekExactKey(KeyRef key) { + if (path.empty()) { + return Void(); + } + path.resize(1); + return seekExactKeyFromCurrent(key); + } + // Start fetching sibling nodes in the forward or backward direction, stopping after recordLimit or byteLimit void prefetch(KeyRef rangeEnd, bool directionForward, int recordLimit, int byteLimit) { // Prefetch scans level 2 so if there are less than 2 nodes in the path there is no level 2 @@ -7598,7 +7817,7 @@ class VersionedBTree { root = *snapshot->extra.getPtr(); } - return cursor->init(this, reason, options, snapshot, root); + return cursor->init(this, reason, options, std::move(snapshot), root); } }; @@ -7883,8 +8102,8 @@ class KeyValueStoreRedwood : public IKeyValueStore { &cur, self->m_tree->getLastCommittedVersion(), PagerEventReasons::PointRead, options); ++g_redwoodMetrics.metric.opGet; - co_await cur.seekGTE(key); - if (cur.isValid() && cur.get().key == key) { + co_await cur.seekExactKey(key); + if (cur.isValid()) { // Return a Value whose arena depends on the source page arena Value v; v.arena().dependsOn(cur.back().page->getArena()); @@ -9769,6 +9988,86 @@ Future commitAndReportLoadProgress(VersionedBTree* btree, } // namespace +static Future> neverCompletingPageRead(Reference page) { + co_await Future(Never()); + co_return page; +} + +TEST_CASE("/redwood/correctness/unit/readOnlyPageFuture") { + const StringRef expectedPageContents = "retained page contents"_sr; + Future> survivingRead; + { + DWALPager::PageCacheEntry entry; + Reference firstPage = makeReference(4096, 4096); + ArenaPage* firstPagePtr = firstPage.getPtr(); + entry.setReadFuture(firstPage); + + Future> firstRead = entry.getReadFuture(); + ASSERT(firstRead.isReady() && !firstRead.isError()); + ASSERT(firstRead.get().getPtr() == firstPagePtr); + ASSERT(firstRead == entry.getReadFuture()); + ASSERT(entry.getReadFuture().get().getPtr() == firstPagePtr); + + Reference updatedPage = makeReference(4096, 4096); + updatedPage->init(EncodingType::XXHash64, PageType::BTreeNode, 1); + memcpy(updatedPage->mutateData(), expectedPageContents.begin(), expectedPageContents.size()); + ArenaPage* updatedPagePtr = updatedPage.getPtr(); + entry.setReadFuture(updatedPage); + Future> updatedRead = entry.getReadFuture(); + ASSERT(updatedRead.isReady() && updatedRead.get().getPtr() == updatedPagePtr); + ASSERT(firstRead.get().getPtr() == firstPagePtr); + + Promise> pendingPromise; + entry.setReadFuture(pendingPromise.getFuture()); + Future> pendingRead = entry.getReadFuture(); + Future> secondPendingRead = entry.getReadFuture(); + ASSERT(!pendingRead.isReady()); + ASSERT(!secondPendingRead.isReady()); + pendingRead.cancel(); + ASSERT(pendingRead.isReady() && pendingRead.isError()); + ASSERT_EQ(pendingRead.getError().code(), error_code_actor_cancelled); + ASSERT(!secondPendingRead.isReady()); + Reference pendingPage = makeReference(4096, 4096); + ArenaPage* pendingPagePtr = pendingPage.getPtr(); + pendingPromise.send(pendingPage); + ASSERT(secondPendingRead.isReady() && !secondPendingRead.isError()); + ASSERT(secondPendingRead.get().getPtr() == pendingPagePtr); + Future> cachedRead = entry.getReadFuture(); + ASSERT(cachedRead.isReady() && !cachedRead.isError()); + ASSERT(cachedRead.get().getPtr() == pendingPagePtr); + ASSERT(cachedRead == entry.getReadFuture()); + + Promise> errorPromise; + entry.setReadFuture(errorPromise.getFuture()); + Future> pendingError = entry.getReadFuture(); + ASSERT(!pendingError.isReady()); + errorPromise.sendError(io_error()); + ASSERT(pendingError.isReady() && pendingError.isError()); + ASSERT(pendingError.getError().code() == error_code_io_error); + + entry.setReadFuture(io_error()); + Future> readyError = entry.getReadFuture(); + ASSERT(readyError.isReady() && readyError.isError()); + ASSERT(readyError.getError().code() == error_code_io_error); + ASSERT(readyError == entry.getReadFuture()); + + entry.setReadFuture(neverCompletingPageRead(makeReference(4096, 4096))); + Future> cancelledRead = entry.getReadFuture(); + ASSERT(!cancelledRead.isReady()); + entry.cancel(); + ASSERT(cancelledRead.isReady() && cancelledRead.isError()); + ASSERT(cancelledRead.getError().code() == error_code_actor_cancelled); + + entry.setReadFuture(updatedPage); + survivingRead = entry.getReadFuture(); + updatedPage.clear(); + } + ASSERT(survivingRead.isReady() && !survivingRead.isError()); + ASSERT_EQ(survivingRead.get()->dataAsStringRef().substr(0, expectedPageContents.size()), expectedPageContents); + + return Void(); +} + TEST_CASE("/redwood/correctness/btreeCloseWithQueuedCommits") { g_redwoodMetricsActor = Void(); g_redwoodMetrics.clear(); @@ -9865,6 +10164,162 @@ TEST_CASE("Lredwood/correctness/forwardErrorReentrantCancellation") { return Void(); } +Future checkExactKey(VersionedBTree* btree, std::string key, Optional expectedValue) { + VersionedBTree::BTreeCursor cursor; + co_await btree->initBTreeCursor(&cursor, btree->getLastCommittedVersion(), PagerEventReasons::PointRead); + co_await cursor.seekExactKey(StringRef(key)); + ASSERT_EQ(cursor.isValid(), expectedValue.present()); + if (expectedValue.present()) { + ASSERT_EQ(cursor.get().key, StringRef(key)); + ASSERT(cursor.get().value.present()); + ASSERT_EQ(cursor.get().value.get(), StringRef(expectedValue.get())); + ASSERT(!cursor.inRoot()); + } + co_return; +} + +Future checkExactKVStoreKey(IKeyValueStore* kvStore, std::string key, Optional expectedValue) { + Optional value = co_await kvStore->readValue(StringRef(key)); + ASSERT_EQ(value.present(), expectedValue.present()); + if (expectedValue.present()) { + ASSERT_EQ(value.get(), StringRef(expectedValue.get())); + } + co_return; +} + +TEST_CASE("Lredwood/correctness/seekExactKey") { + g_redwoodMetricsActor = Void(); + g_redwoodMetrics.clear(); + + std::string file = "unittest_btree-seek-exact.redwood-v1"; + deleteFile(file); + + auto* btree = new VersionedBTree(new DWALPager(250, + SERVER_KNOBS->REDWOOD_DEFAULT_EXTENT_SIZE, + file, + FLOW_KNOBS->PAGE_CACHE_4K, + 0, + SERVER_KNOBS->REDWOOD_EXTENT_CONCURRENT_READS, + false), + file, + UID(), + {}); + co_await btree->init(); + + for (int i = 0; i < 80; ++i) { + std::string key = format("key-%04d-%s", i * 2, std::string(64, 'k').c_str()); + std::string value = i == 30 ? std::string() : format("value-%04d-%s", i, std::string(64, 'v').c_str()); + btree->set(KeyValueRef(StringRef(key), StringRef(value))); + } + Version version = btree->getLastCommittedVersion(); + co_await btree->commit(++version); + + int checkIndex = 0; + for (; checkIndex < 80; ++checkIndex) { + co_await checkExactKey(btree, + format("key-%04d-%s", checkIndex * 2, std::string(64, 'k').c_str()), + checkIndex == 30 ? std::string() + : format("value-%04d-%s", checkIndex, std::string(64, 'v').c_str())); + co_await checkExactKey( + btree, format("key-%04d-%s", checkIndex * 2 + 1, std::string(64, 'k').c_str()), Optional()); + } + co_await checkExactKey(btree, "before-all-keys", Optional()); + co_await checkExactKey(btree, "z-after-all-keys", Optional()); + + Key clearedKey = StringRef(format("key-%04d-%s", 80, std::string(64, 'k').c_str())); + btree->clear(singleKeyRange(clearedKey)); + co_await btree->commit(++version); + co_await checkExactKey(btree, clearedKey.toString(), Optional()); + + Future closed = btree->onClosed(); + btree->dispose(); + co_await closed; + co_await delay(0); + ASSERT(DWALPager::PageCacheT::Evictor::getEvictor()->empty()); + + std::string kvFile = "unittest_kvstore-seek-exact.redwood-v1"; + deleteFile(kvFile); + IKeyValueStore* kvStore = keyValueStoreRedwoodV1(kvFile, UID(), {}, 65536); + co_await kvStore->init(); + for (int i = 0; i < 160; ++i) { + std::string key = format("present-%04d-%s", i * 2, std::string(64, 'k').c_str()); + std::string value = i == 30 ? std::string() : format("value-%04d-%s", i, std::string(512, 'v').c_str()); + kvStore->set(KeyValueRef(StringRef(key), StringRef(value))); + } + Key largeKey = "zz-large-value"_sr; + Value largeValue = StringRef(std::string(32768, 'v')); + kvStore->set(KeyValueRef(largeKey, largeValue)); + co_await kvStore->commit(); + + int kvIndex = 0; + for (; kvIndex < 160; ++kvIndex) { + co_await checkExactKVStoreKey(kvStore, + format("present-%04d-%s", kvIndex * 2, std::string(64, 'k').c_str()), + kvIndex == 30 ? std::string() + : format("value-%04d-%s", kvIndex, std::string(512, 'v').c_str())); + co_await checkExactKVStoreKey( + kvStore, format("present-%04d-%s", kvIndex * 2 + 1, std::string(64, 'k').c_str()), Optional()); + } + co_await checkExactKVStoreKey(kvStore, "before-all-keys", Optional()); + co_await checkExactKVStoreKey(kvStore, "z-after-all-keys", Optional()); + co_await checkExactKVStoreKey(kvStore, largeKey.toString(), largeValue.toString()); + + Key clearBegin = StringRef(format("present-%04d-%s", 80, std::string(64, 'k').c_str())); + Key clearEnd = StringRef(format("present-%04d-%s", 240, std::string(64, 'k').c_str())); + kvStore->clear(KeyRangeRef(clearBegin, clearEnd)); + co_await kvStore->commit(); + int clearedIndex = 40; + for (; clearedIndex < 120; ++clearedIndex) { + co_await checkExactKVStoreKey(kvStore, + format("present-%04d-%s", clearedIndex * 2, std::string(64, 'k').c_str()), + Optional()); + } + co_await checkExactKVStoreKey(kvStore, + format("present-%04d-%s", 78, std::string(64, 'k').c_str()), + format("value-%04d-%s", 39, std::string(512, 'v').c_str())); + co_await checkExactKVStoreKey(kvStore, + format("present-%04d-%s", 240, std::string(64, 'k').c_str()), + format("value-%04d-%s", 120, std::string(512, 'v').c_str())); + + kvStore->set(KeyValueRef(clearBegin, "reinserted"_sr)); + co_await kvStore->commit(); + co_await checkExactKVStoreKey(kvStore, clearBegin.toString(), std::string("reinserted")); + + Key retainedKey = StringRef(format("present-%04d-%s", 20, std::string(64, 'k').c_str())); + Key retainedTailKey = StringRef(format("present-%04d-%s", 300, std::string(64, 'k').c_str())); + Key emptyKey = StringRef(format("present-%04d-%s", 60, std::string(64, 'k').c_str())); + Optional retainedValue = co_await kvStore->readValue(retainedKey); + Optional retainedTailValue = co_await kvStore->readValue(retainedTailKey); + Optional emptyValue = co_await kvStore->readValue(emptyKey); + ASSERT_EQ(retainedValue, Optional(StringRef(format("value-%04d-%s", 10, std::string(512, 'v').c_str())))); + ASSERT_EQ(retainedTailValue, + Optional(StringRef(format("value-%04d-%s", 150, std::string(512, 'v').c_str())))); + ASSERT_EQ(emptyValue, Optional(""_sr)); + + Future kvClosed = kvStore->onClosed(); + kvStore->close(); + co_await kvClosed; + ASSERT_EQ(retainedValue, Optional(StringRef(format("value-%04d-%s", 10, std::string(512, 'v').c_str())))); + ASSERT_EQ(retainedTailValue, + Optional(StringRef(format("value-%04d-%s", 150, std::string(512, 'v').c_str())))); + ASSERT_EQ(emptyValue, Optional(""_sr)); + + // Keep the large leaf, its ancestors, and pager metadata cached even with buggified page sizes. + // The earlier reads exercise eviction with a small cache; this phase requires a cache hit. + kvStore = keyValueStoreRedwoodV1(kvFile, UID(), {}, FLOW_KNOBS->PAGE_CACHE_4K); + co_await kvStore->init(); + unsigned int cacheHitsBefore = g_redwoodMetrics.metric.pagerCacheHit; + co_await checkExactKVStoreKey(kvStore, largeKey.toString(), largeValue.toString()); + co_await checkExactKVStoreKey(kvStore, largeKey.toString(), largeValue.toString()); + ASSERT(g_redwoodMetrics.metric.pagerCacheHit > cacheHitsBefore); + + Future reopenedClosed = kvStore->onClosed(); + kvStore->dispose(); + co_await reopenedClosed; + + co_return; +} + TEST_CASE("Lredwood/correctness/btree") { g_redwoodMetricsActor = Void(); // Prevent trace event metrics from starting g_redwoodMetrics.clear(); @@ -10405,7 +10860,7 @@ TEST_CASE(":/redwood/performance/extentQueue") { if (reload) { pager = new DWALPager( pageSize, extentSize, fileName, cacheSizeBytes, remapCleanupWindowBytes, concurrentExtentReads, false); - co_await success(pager->init()); + co_await pager->init(); LogicalPageID extID = pager->newLastExtentID(); m_extentQueue.create(pager, extID, "ExtentQueue", pager->newLastQueueID(), true); @@ -10455,7 +10910,7 @@ TEST_CASE(":/redwood/performance/extentQueue") { printf("Reopening pager file from disk.\n"); pager = new DWALPager( pageSize, extentSize, fileName, cacheSizeBytes, remapCleanupWindowBytes, concurrentExtentReads, false); - co_await success(pager->init()); + co_await pager->init(); printf("Starting ExtentQueue FastPath Recovery from Disk.\n"); diff --git a/fdbserver/kvstore/VersionedBTreeDebug.h b/fdbserver/kvstore/VersionedBTreeDebug.h index 75afe9a0914..1240a2ed936 100644 --- a/fdbserver/kvstore/VersionedBTreeDebug.h +++ b/fdbserver/kvstore/VersionedBTreeDebug.h @@ -49,7 +49,6 @@ bool enableRedwoodDebug(); #define debug_print_always(str) debug_printf_always("%s\n", str.c_str()) #define debug_printf_noop(...) -#if defined(NO_INTELLISENSE) #if REDWOOD_DEBUG // debug_print() only outputs the line if the enableRedwoodDebug() pass. @@ -62,13 +61,9 @@ bool enableRedwoodDebug(); // Completely compile out debug statements if REDWOOD_DEBUG is off. #define debug_printf debug_printf_noop #endif -#else -// To get error-checking on debug_printf statements in IDE -#define debug_printf printf -#endif #define BEACON debug_printf_always("HERE\n") #define TRACE \ debug_printf_always("%s: %s line %d %s\n", __FUNCTION__, __FILE__, __LINE__, platform::get_backtrace().c_str()); -#endif // FDBSERVER_VERSIONEDBTREEDEBUG_H \ No newline at end of file +#endif // FDBSERVER_VERSIONEDBTREEDEBUG_H diff --git a/fdbserver/logsystem/LogSystem.cpp b/fdbserver/logsystem/LogSystem.cpp index 43ae3e14122..2d2ebdea049 100644 --- a/fdbserver/logsystem/LogSystem.cpp +++ b/fdbserver/logsystem/LogSystem.cpp @@ -877,6 +877,7 @@ Version LogSystem::getKnownCommittedVersion() { Future LogSystem::onKnownCommittedVersionChange() { std::vector> result; + result.reserve(lockResults.size()); for (auto& it : lockResults) { result.push_back(LogSystem::getDurableVersionChanged(it)); } diff --git a/fdbserver/logsystem/LogSystemPeekCursor.cpp b/fdbserver/logsystem/LogSystemPeekCursor.cpp index cfa294e8734..2866bc353b8 100644 --- a/fdbserver/logsystem/LogSystemPeekCursor.cpp +++ b/fdbserver/logsystem/LogSystemPeekCursor.cpp @@ -824,6 +824,7 @@ MergedPeekCursor::MergedPeekCursor(std::vector> cons Reference MergedPeekCursor::cloneNoMore() { std::vector> cursors; + cursors.reserve(serverCursors.size()); for (const auto& it : serverCursors) { cursors.push_back(it->cloneServerNoMore()); } diff --git a/fdbserver/sequencer/ResolutionBalancer.cpp b/fdbserver/sequencer/ResolutionBalancer.cpp index 14b4e017c1c..15e6ec5f310 100644 --- a/fdbserver/sequencer/ResolutionBalancer.cpp +++ b/fdbserver/sequencer/ResolutionBalancer.cpp @@ -121,6 +121,7 @@ Future ResolutionBalancer::resolutionBalancing() { while (!resolverChanges.get().empty()) co_await resolverChanges.onChange(); std::vector> futures; + futures.reserve(resolvers.size()); for (auto& p : resolvers) futures.push_back( brokenPromiseToNever(p.metrics.getReply(ResolutionMetricsRequest(), TaskPriority::ResolutionMetrics))); diff --git a/fdbserver/storageserver/ReadLatencySamples.cpp b/fdbserver/storageserver/ReadLatencySamples.cpp index 80d51415b7a..ba39e47889b 100644 --- a/fdbserver/storageserver/ReadLatencySamples.cpp +++ b/fdbserver/storageserver/ReadLatencySamples.cpp @@ -58,3 +58,11 @@ void ReadLatencySamples::sample(double latency, SampleType sampleType, Optional< perType[readType.get()].samples[sampleType]->addMeasurement(latency); } } + +void ReadLatencySamples::samplePair(double latency, SampleType first, SampleType second, Optional readType) { + aggregate.samples[first]->addMeasurementPair(latency, *aggregate.samples[second]); + if (readType.present()) { + auto& samples = perType[readType.get()].samples; + samples[first]->addMeasurementPair(latency, *samples[second]); + } +} diff --git a/fdbserver/storageserver/ReadLatencySamples.h b/fdbserver/storageserver/ReadLatencySamples.h index ed9a1049de4..a15076bebd8 100644 --- a/fdbserver/storageserver/ReadLatencySamples.h +++ b/fdbserver/storageserver/ReadLatencySamples.h @@ -54,4 +54,5 @@ class ReadLatencySamples { explicit ReadLatencySamples(UID serverId); void sample(double latency, SampleType, Optional = {}); + void samplePair(double latency, SampleType first, SampleType second, Optional = {}); }; diff --git a/fdbserver/storageserver/storageserver.cpp b/fdbserver/storageserver/storageserver.cpp index b42ca1918de..5c7af2ea86a 100644 --- a/fdbserver/storageserver/storageserver.cpp +++ b/fdbserver/storageserver/storageserver.cpp @@ -19,6 +19,7 @@ */ #include +#include #include #include #include @@ -65,12 +66,14 @@ #include "fdbserver/core/WaitFailure.h" #include "flow/ActorCollection.h" #include "flow/Arena.h" +#include "flow/Deque.h" #include "flow/Error.h" #include "flow/Hash3.h" #include "flow/Histogram.h" #include "flow/PriorityMultiLock.h" #include "flow/IRandom.h" #include "flow/IndexedSet.h" +#include "flow/ScopeExit.h" #include "flow/SystemMonitor.h" #include "flow/Trace.h" #include "fdbclient/Tracing.h" @@ -1346,6 +1349,19 @@ struct StorageServer : public IStorageMetricsService { } } counters; + class GetValueQuery { + public: + GetValueQuery(GetValueRequest request, Counters& counters) + : request(std::move(request)), count(counters.allQueries, counters.finishedQueries) {} + GetValueQuery(GetValueQuery&& other) : request(std::move(other.request)) { count = std::move(other.count); } + + GetValueRequest& getRequest() { return request; } + + private: + GetValueRequest request; + CountedSection count; + }; + ActorCollection actors; CoalescedKeyRangeMap> byteSampleClears; @@ -1390,8 +1406,26 @@ struct StorageServer : public IStorageMetricsService { return ssLock->lock(readPriorityRanks[readType]); } + Optional tryGetReadLock(const Optional& options) { + int readType = (int)(options.present() ? options.get().type : ReadType::NORMAL); + readType = std::clamp(readType, 0, readPriorityRanks.size() - 1); + return ssLock->tryLock(readPriorityRanks[readType]); + } + FlowLock serveAuditStorageParallelismLock; + // Serializes ValidateStorageServerShard audits against each other. They drive + // shardAssignmentHistory / trackShardAssignmentMinVersion, single-instance state on this server + // (startTrackShardAssignment() ASSERTs tracking is off), so two concurrent ssshard audits corrupt + // each other's view of which ranges moved. Exclusion must not depend on + // SERVE_AUDIT_STORAGE_PARALLELISM. Held only by ssshard audits; other types stay parallel. + FlowLock ssShardAuditExclusionLock; + + // Shared by every concurrent audit task on this server. It must NOT be per-task: the knob is named + // AUDIT_STORAGE_RATE_PER_SERVER_MAX, and a per-task limiter would multiply the allowance by + // SERVE_AUDIT_STORAGE_PARALLELISM, against a server that is also serving reads. + Reference auditStorageRateLimiter; + FlowLock serveBulkDumpParallelismLock; int64_t instanceID; @@ -1413,6 +1447,10 @@ struct StorageServer : public IStorageMetricsService { Version lastDurableVersionEBrake; int maxQueryQueue; + Deque pendingDefaultGetValues; + Deque pendingLowPriorityGetValues; + bool defaultGetValueDispatchScheduled = false; + bool lowPriorityGetValueDispatchScheduled = false; int getAndResetMaxQueryQueueSize() { int val = maxQueryQueue; maxQueryQueue = 0; @@ -1483,7 +1521,8 @@ struct StorageServer : public IStorageMetricsService { serveFetchCheckpointParallelismLock(SERVER_KNOBS->SERVE_FETCH_CHECKPOINT_PARALLELISM), ssLock(makeReference(SERVER_KNOBS->STORAGE_SERVER_READ_CONCURRENCY, SERVER_KNOBS->STORAGESERVER_READ_PRIORITIES)), - serveAuditStorageParallelismLock(SERVER_KNOBS->SERVE_AUDIT_STORAGE_PARALLELISM), + serveAuditStorageParallelismLock(SERVER_KNOBS->SERVE_AUDIT_STORAGE_PARALLELISM), ssShardAuditExclusionLock(1), + auditStorageRateLimiter(makeReference(SERVER_KNOBS->AUDIT_STORAGE_RATE_PER_SERVER_MAX, 1)), serveBulkDumpParallelismLock(SERVER_KNOBS->SS_SERVE_BULKDUMP_PARALLELISM), instanceID(deterministicRandom()->randomUniqueID().first()), shuttingDown(false), behind(false), versionBehind(false), debug_inApplyUpdate(false), debug_lastValidateTime(0), lastBytesInputEBrake(0), @@ -1615,19 +1654,24 @@ struct StorageServer : public IStorageMetricsService { // penalty used by loadBalance() to balance requests among SSes. We prefer SS with less write queue size. double getPenalty() const override { + double localRate = currentRate(); return std::max(std::max(1.0, (queueSize() - (SERVER_KNOBS->TARGET_BYTES_PER_STORAGE_SERVER - 2.0 * SERVER_KNOBS->SPRING_BYTES_STORAGE_SERVER)) / SERVER_KNOBS->SPRING_BYTES_STORAGE_SERVER), - (currentRate() < 1e-6 ? 1e6 : 1.0 / currentRate())); + (localRate < 1e-6 ? 1e6 : 1.0 / localRate)); } // Normally the storage server prefers to serve read requests over making mutations // durable to disk. However, when the storage server falls to far behind on // making mutations durable, this function will change the priority to prefer writes. + bool useLowPriorityRead() const { + return (version.get() - durableVersion.get() > SERVER_KNOBS->LOW_PRIORITY_DURABILITY_LAG) || + (queueSize() > SERVER_KNOBS->LOW_PRIORITY_STORAGE_QUEUE_BYTES); + } + Future getQueryDelay() { - if ((version.get() - durableVersion.get() > SERVER_KNOBS->LOW_PRIORITY_DURABILITY_LAG) || - (queueSize() > SERVER_KNOBS->LOW_PRIORITY_STORAGE_QUEUE_BYTES)) { + if (useLowPriorityRead()) { ++counters.lowPriorityQueries; return delay(0, TaskPriority::LowPriorityRead); } @@ -2007,6 +2051,47 @@ Version getLatestCommitVersion(VersionVector& ssLatestCommitVersions, Tag& tag) return commitVersion; } +// Returns only successful, immediately readable versions so the common read path can avoid allocating a +// Future. +Optional tryGetReadyReadVersion(Version currentVersion, + Version oldestVersion, + Version commitVersion, + Version readVersion) { + ASSERT(commitVersion == invalidVersion || commitVersion < readVersion); + + if (readVersion == latestVersion) { + readVersion = std::max(Version(1), currentVersion); + } + if (readVersion < oldestVersion || readVersion <= 0) { + return {}; + } + if (commitVersion == invalidVersion) { + return readVersion <= currentVersion ? Optional(readVersion) : Optional(); + } + if (commitVersion < oldestVersion) { + return currentVersion < readVersion ? currentVersion : readVersion; + } + return commitVersion <= currentVersion ? Optional(commitVersion) : Optional(); +} + +TEST_CASE("/fdbserver/storageserver/tryGetReadyReadVersion") { + const Version currentVersion = 100; + const Version oldestVersion = 90; + + ASSERT(tryGetReadyReadVersion(currentVersion, oldestVersion, invalidVersion, 95) == Optional(95)); + ASSERT(tryGetReadyReadVersion(currentVersion, oldestVersion, invalidVersion, latestVersion) == + Optional(currentVersion)); + ASSERT(!tryGetReadyReadVersion(currentVersion, oldestVersion, invalidVersion, 89).present()); + ASSERT(!tryGetReadyReadVersion(currentVersion, oldestVersion, invalidVersion, 0).present()); + ASSERT(!tryGetReadyReadVersion(currentVersion, oldestVersion, invalidVersion, 101).present()); + + ASSERT(tryGetReadyReadVersion(currentVersion, oldestVersion, 80, 99) == Optional(99)); + ASSERT(tryGetReadyReadVersion(currentVersion, oldestVersion, 80, 110) == Optional(currentVersion)); + ASSERT(tryGetReadyReadVersion(currentVersion, oldestVersion, 95, 110) == Optional(95)); + ASSERT(!tryGetReadyReadVersion(currentVersion, oldestVersion, 105, 110).present()); + return Void(); +} + Future waitForVersion(StorageServer* data, Version version, SpanContext spanContext) { if (version == latestVersion) { version = std::max(Version(1), data->version.get()); @@ -2133,30 +2218,186 @@ std::shared_ptr StorageServer::getMoveInShard(const UID& dataMoveId return shard; } -Future getValueQ(StorageServer* data, GetValueRequest req) { +void beginGetValueQ(StorageServer* data, const GetValueRequest& req) { + ++data->counters.getValueQueries; + if (req.key.startsWith(systemKeys.begin)) { + ++data->counters.systemKeyQueries; + } + data->maxQueryQueue = std::max( + data->maxQueryQueue, data->counters.allQueries.getValue() - data->counters.finishedQueries.getValue()); +} + +template +int getValueReadPath(Iterator&& item, KeyRef key) { + if (item && item->isValue() && item.key() == key) { + return 1; + } + if (item && item->isClearTo() && item->getEndKey() > key) { + return 0; + } + return 2; +} + +TEST_CASE("/fdbserver/storageserver/getValueReadPath") { + struct Entry { + bool value; + bool clear; + KeyRef end; + bool isValue() const { return value; } + bool isClearTo() const { return clear; } + KeyRef getEndKey() const { return end; } + }; + struct Iterator { + const Entry* entry; + KeyRef itemKey; + explicit operator bool() const { return entry != nullptr; } + const Entry* operator->() const { return entry; } + KeyRef key() const { return itemKey; } + }; + + const Entry value{ true, false, StringRef() }; + const Entry coveringClear{ false, true, "z"_sr }; + const Entry expiredClear{ false, true, "m"_sr }; + ASSERT_EQ(getValueReadPath(Iterator{ nullptr, StringRef() }, "m"_sr), 2); + ASSERT_EQ(getValueReadPath(Iterator{ &value, "m"_sr }, "m"_sr), 1); + ASSERT_EQ(getValueReadPath(Iterator{ &value, "a"_sr }, "m"_sr), 2); + ASSERT_EQ(getValueReadPath(Iterator{ &coveringClear, "a"_sr }, "m"_sr), 0); + ASSERT_EQ(getValueReadPath(Iterator{ &expiredClear, "a"_sr }, "m"_sr), 2); + return Void(); +} + +int64_t replyGetValueQ(StorageServer* data, GetValueRequest& req, Version version, int path, Optional& value) { + DEBUG_MUTATION("ShardGetValue", + version, + MutationRef(MutationRef::DebugKey, req.key, value.present() ? value.get() : ""_sr), + data->thisServerID); + DEBUG_MUTATION("ShardGetPath", + version, + MutationRef(MutationRef::DebugKey, + req.key, + path == 0 ? "0"_sr + : path == 1 ? "1"_sr + : "2"_sr), + data->thisServerID); + + int64_t resultSize = 0; + if (value.present()) { + ++data->counters.rowsQueried; + resultSize = value.get().size(); + data->counters.bytesQueried += resultSize; + } else { + ++data->counters.emptyQueries; + } + + if (SERVER_KNOBS->READ_SAMPLING_ENABLED) { + int64_t bytesReadPerKSecond = + value.present() ? std::max((int64_t)(req.key.size() + value.get().size()), SERVER_KNOBS->EMPTY_READ_PENALTY) + : SERVER_KNOBS->EMPTY_READ_PENALTY; + data->metrics.notifyBytesReadPerKSecond(req.key, bytesReadPerKSecond); + } + + if (req.options.present() && req.options.get().debugID.present()) { + g_traceBatch.addEvent("GetValueDebug", + req.options.get().debugID.get().first(), + "getValueQ.AfterRead", + req.spanContext.traceID, + req.spanContext.spanID); //.detail("TaskID", g_network->getCurrentTask()); + } + + GetValueReply reply(std::move(value), /*cached=*/false); + reply.penalty = data->getPenalty(); + req.reply.send(std::move(reply)); + return resultSize; +} + +void finishGetValueQ(StorageServer* data, const GetValueRequest& req, int64_t resultSize) { + // Key size is not included in "BytesQueried", but still contributes to cost. + data->transactionTagCounter.addRequest(req.tags, req.key.size() + resultSize); + + double duration = g_network->timer() - req.requestTime(); + Optional readType = trackedReadType(req); + data->counters.readLatencySamples.samplePair( + duration, ReadLatencySamples::READ, ReadLatencySamples::READ_VALUE, readType); + if (data->latencyBandConfig.present()) { + int maxReadBytes = + data->latencyBandConfig.get().readConfig.maxReadBytes.orDefault(std::numeric_limits::max()); + data->counters.readLatencyBands.addMeasurement(duration, 1, Filtered(resultSize > maxReadBytes)); + } +} + +int64_t finishGetValueQStorageRead(StorageServer* data, + GetValueRequest& req, + Version version, + uint64_t changeCounter, + Optional value) { + data->counters.kvGetBytes += value.expectedSize(); + if (version < data->storageVersion()) { + CODE_PROBE(true, "transaction_too_old after readValue"); + throw transaction_too_old(); + } + data->checkChangeCounter(changeCounter, req.key); + return replyGetValueQ(data, req, version, 2, value); +} + +Future getValueQStorageRead(StorageServer* data, + StorageServer::GetValueQuery query, + Span span, + PriorityMultiLock::Releaser readLock, + Version version, + uint64_t changeCounter, + Future> valueFuture) { + Span activeSpan(std::move(span)); + StorageServer::GetValueQuery activeQuery(std::move(query)); + GetValueRequest& req = activeQuery.getRequest(); + int64_t resultSize = 0; + try { + PriorityMultiLock::Releaser heldReadLock(std::move(readLock)); + Optional value = co_await valueFuture; + resultSize = finishGetValueQStorageRead(data, req, version, changeCounter, std::move(value)); + } catch (Error& e) { + if (!canReplyWith(e)) + throw; + data->sendErrorWithPenalty(req.reply, e, data->getPenalty()); + } + + finishGetValueQ(data, req, resultSize); +} + +Future getValueQImpl(StorageServer* data, + StorageServer::GetValueQuery query, + Span span, + bool externallyDispatched, + Optional immediateReadLock = {}) { + // A completed Future can outlive the operation, so keep the active span in body scope. + Span activeSpan(std::move(span)); + StorageServer::GetValueQuery activeQuery(std::move(query)); + GetValueRequest& req = activeQuery.getRequest(); int64_t resultSize = 0; - Span span("SS:getValue"_loc, req.spanContext); - CountedSection cs(data->counters.allQueries, data->counters.finishedQueries); // Temporarily disabled -- this path is hit a lot // getCurrentLineage()->modify(&TransactionLineage::txID) = req.spanContext.first(); try { - ++data->counters.getValueQueries; - if (req.key.startsWith(systemKeys.begin)) { - ++data->counters.systemKeyQueries; + // Keep the immediate holder scoped like the queued holder so reentrant lock wakeups retain their ordering. + Optional heldReadLock = std::move(immediateReadLock); + if (!externallyDispatched) { + beginGetValueQ(data, req); } - data->maxQueryQueue = std::max( - data->maxQueryQueue, data->counters.allQueries.getValue() - data->counters.finishedQueries.getValue()); // Active load balancing runs at a very high priority (to obtain accurate queue lengths) // so we need to downgrade here - co_await data->getQueryDelay(); - PriorityMultiLock::Lock readLock = co_await data->getReadLock(req.options); + if (!externallyDispatched) { + co_await data->getQueryDelay(); + } + Optional queuedReadLock; + if (!heldReadLock.present()) { + queuedReadLock = co_await data->getReadLock(req.options); + } // Track time from requestTime through now as read queueing wait time double queueWaitEnd = g_network->timer(); + Optional readType = trackedReadType(req); data->counters.readLatencySamples.sample( - queueWaitEnd - req.requestTime(), ReadLatencySamples::READ_QUEUE_WAIT, trackedReadType(req)); + queueWaitEnd - req.requestTime(), ReadLatencySamples::READ_QUEUE_WAIT, readType); if (req.options.present() && req.options.get().debugID.present()) { g_traceBatch.addEvent("GetValueDebug", @@ -2170,7 +2411,7 @@ Future getValueQ(StorageServer* data, GetValueRequest req) { Version commitVersion = getLatestCommitVersion(req.ssLatestCommitVersions, data->tag); Version version = co_await waitForVersion(data, commitVersion, req.version, req.spanContext); data->counters.readLatencySamples.sample( - g_network->timer() - queueWaitEnd, ReadLatencySamples::READ_VERSION_WAIT, trackedReadType(req)); + g_network->timer() - queueWaitEnd, ReadLatencySamples::READ_VERSION_WAIT, readType); if (req.options.present() && req.options.get().debugID.present()) { g_traceBatch.addEvent("GetValueDebug", @@ -2187,13 +2428,11 @@ Future getValueQ(StorageServer* data, GetValueRequest req) { throw wrong_shard_server(); } - int path = 0; - auto i = data->data().at(version).lastLessOrEqual(req.key); - if (i && i->isValue() && i.key() == req.key) { + auto i = StorageServer::VersionedData::lastLessOrEqualAt(data->data().getRoot(version), version, req.key); + int path = getValueReadPath(i, req.key); + if (path == 1) { v = (Value)i->getValue(); - path = 1; - } else if (!i || !i->isClearTo() || i->getEndKey() <= req.key) { - path = 2; + } else if (path == 2) { Optional vv = co_await data->storage.readValue(req.key, req.options); data->counters.kvGetBytes += vv.expectedSize(); // Validate that while we were reading the data we didn't lose the version or shard @@ -2202,74 +2441,87 @@ Future getValueQ(StorageServer* data, GetValueRequest req) { throw transaction_too_old(); } data->checkChangeCounter(changeCounter, req.key); - v = vv; - } - - DEBUG_MUTATION("ShardGetValue", - version, - MutationRef(MutationRef::DebugKey, req.key, v.present() ? v.get() : ""_sr), - data->thisServerID); - DEBUG_MUTATION("ShardGetPath", - version, - MutationRef(MutationRef::DebugKey, - req.key, - path == 0 ? "0"_sr - : path == 1 ? "1"_sr - : "2"_sr), - data->thisServerID); - - /* - StorageMetrics m; - m.bytesWrittenPerKSecond = req.key.size() + (v.present() ? v.get().size() : 0); - m.iosPerKSecond = 1; - data->metrics.notify(req.key, m); - */ - - if (v.present()) { - ++data->counters.rowsQueried; - resultSize = v.get().size(); - data->counters.bytesQueried += resultSize; - } else { - ++data->counters.emptyQueries; - } - - if (SERVER_KNOBS->READ_SAMPLING_ENABLED) { - // If the read yields no value, randomly sample the empty read. - int64_t bytesReadPerKSecond = - v.present() ? std::max((int64_t)(req.key.size() + v.get().size()), SERVER_KNOBS->EMPTY_READ_PENALTY) - : SERVER_KNOBS->EMPTY_READ_PENALTY; - data->metrics.notifyBytesReadPerKSecond(req.key, bytesReadPerKSecond); + v = std::move(vv); } - if (req.options.present() && req.options.get().debugID.present()) { - g_traceBatch.addEvent("GetValueDebug", - req.options.get().debugID.get().first(), - "getValueQ.AfterRead", - req.spanContext.traceID, - req.spanContext.spanID); //.detail("TaskID", g_network->getCurrentTask()); - } - - GetValueReply reply(v, /*cached=*/false); - reply.penalty = data->getPenalty(); - req.reply.send(reply); + resultSize = replyGetValueQ(data, req, version, path, v); } catch (Error& e) { if (!canReplyWith(e)) throw; data->sendErrorWithPenalty(req.reply, e, data->getPenalty()); } - // Key size is not included in "BytesQueried", but still contributes to cost, - // so it must be accounted for here. - data->transactionTagCounter.addRequest(req.tags, req.key.size() + resultSize); + finishGetValueQ(data, req, resultSize); +} - double duration = g_network->timer() - req.requestTime(); - data->counters.readLatencySamples.sample(duration, ReadLatencySamples::READ, trackedReadType(req)); - data->counters.readLatencySamples.sample(duration, ReadLatencySamples::READ_VALUE, trackedReadType(req)); - if (data->latencyBandConfig.present()) { - int maxReadBytes = - data->latencyBandConfig.get().readConfig.maxReadBytes.orDefault(std::numeric_limits::max()); - data->counters.readLatencyBands.addMeasurement(duration, 1, Filtered(resultSize > maxReadBytes)); +Future getValueQ(StorageServer* data, GetValueRequest req) { + Span span("SS:getValue"_loc, req.spanContext); + return getValueQImpl(data, StorageServer::GetValueQuery(std::move(req), data->counters), std::move(span), false); +} + +Optional> getValueQDispatched(StorageServer* data, StorageServer::GetValueQuery query) { + GetValueRequest& req = query.getRequest(); + int64_t resultSize = 0; + try { + Optional readLock = data->tryGetReadLock(req.options); + if (!readLock.present()) { + Span span("SS:getValue"_loc, req.spanContext); + return getValueQImpl(data, std::move(query), std::move(span), true); + } + + double queueWaitEnd = g_network->timer(); + Version commitVersion = getLatestCommitVersion(req.ssLatestCommitVersions, data->tag); + Optional readyVersion = + tryGetReadyReadVersion(data->version.get(), data->oldestVersion.get(), commitVersion, req.version); + double versionWaitEnd = g_network->timer(); + if (!readyVersion.present()) { + Span span("SS:getValue"_loc, req.spanContext); + return getValueQImpl(data, std::move(query), std::move(span), true, std::move(readLock)); + } + + Version version = readyVersion.get(); + Optional readType = trackedReadType(req); + data->counters.readLatencySamples.sample( + queueWaitEnd - req.requestTime(), ReadLatencySamples::READ_QUEUE_WAIT, readType); + data->counters.readLatencySamples.sample( + versionWaitEnd - queueWaitEnd, ReadLatencySamples::READ_VERSION_WAIT, readType); + + uint64_t changeCounter = data->shardChangeCounter; + if (!data->shards[req.key]->isReadable()) { + throw wrong_shard_server(); + } + + auto i = StorageServer::VersionedData::lastLessOrEqualAt(data->data().getRoot(version), version, req.key); + int path = getValueReadPath(i, req.key); + if (path == 2) { + Future> valueFuture = data->storage.readValue(req.key, req.options); + if (!valueFuture.isReady()) { + Span span("SS:getValue"_loc, req.spanContext); + return getValueQStorageRead(data, + std::move(query), + std::move(span), + std::move(readLock.get()), + version, + changeCounter, + std::move(valueFuture)); + } + resultSize = finishGetValueQStorageRead(data, req, version, changeCounter, valueFuture.get()); + } else { + Optional value; + if (path == 1) { + value = (Value)i->getValue(); + } + resultSize = replyGetValueQ(data, req, version, path, value); + } + } catch (Error& e) { + if (!canReplyWith(e)) { + return Future(e); + } + data->sendErrorWithPenalty(req.reply, e, data->getPenalty()); } + + finishGetValueQ(data, req, resultSize); + return {}; } // Pessimistic estimate of the overhead bytes used by each watch. Watch key @@ -3567,8 +3819,8 @@ Future getKeyValuesQ(StorageServer* data, GetKeyValuesRequest req) data->transactionTagCounter.addRequest(req.tags, resultSize); double duration = g_network->timer() - req.requestTime(); - data->counters.readLatencySamples.sample(duration, ReadLatencySamples::READ, trackedReadType(req)); - data->counters.readLatencySamples.sample(duration, ReadLatencySamples::READ_RANGE, trackedReadType(req)); + data->counters.readLatencySamples.samplePair( + duration, ReadLatencySamples::READ, ReadLatencySamples::READ_RANGE, trackedReadType(req)); if (data->latencyBandConfig.present()) { int maxReadBytes = data->latencyBandConfig.get().readConfig.maxReadBytes.orDefault(std::numeric_limits::max()); @@ -3716,21 +3968,34 @@ AuditGetShardInfoRes getThisServerShardInfo(StorageServer* data, KeyRange range) // Check consistency between StorageServer->shardInfo and ServerKeys system key space Future auditStorageServerShardQ(StorageServer* data, AuditStorageRequest req) { ASSERT(req.getType() == AuditType::ValidateStorageServerShard); + // trackShardAssignment is only correct when at most one auditStorageServerShardQ runs at a time, so + // take the exclusion lock before the shared permit -- waiting on a peer ssshard audit must not sit on + // a permit the other audit types are competing for. The check below is the tripwire for this + // invariant and deliberately fails simulation, so reintroducing concurrency shows up loudly. + co_await data->ssShardAuditExclusionLock.take(TaskPriority::DefaultYield); + FlowLock::Releaser ssShardHolder(data->ssShardAuditExclusionLock); co_await data->serveAuditStorageParallelismLock.take(TaskPriority::DefaultYield); - // The trackShardAssignment is correct when at most 1 auditStorageServerShardQ runs - // at a time. Currently, this is guaranteed by setting serveAuditStorageParallelismLock == 1 - // If serveAuditStorageParallelismLock > 1, we need to check trackShardAssignmentMinVersion - // to make sure no onging auditStorageServerShardQ is running + // Constructed BEFORE the tripwire below: the early co_return there would otherwise leak a permit, + // and this lock is shared by every audit type, so enough leaks exhaust it permanently and every + // later audit on this server blocks in take() forever. + FlowLock::Releaser holder(data->serveAuditStorageParallelismLock); if (data->trackShardAssignmentMinVersion != invalidVersion) { - // Another auditStorageServerShardQ is running + // Another auditStorageServerShardQ is running, or a previous one was cancelled before reaching + // its stopTrackShardAssignment() -- that call is plain code at the end of the function, not a + // destructor, so cancellation skips it and leaves tracking on. req.reply.sendError(audit_storage_cancelled()); TraceEvent(g_network->isSimulated() ? SevError : SevWarnAlways, "ExistStorageServerShardAuditExit") // unexpected .detail("NewAuditId", req.id) - .detail("NewAuditType", req.getType()); + .detail("NewAuditType", req.getType()) + .detail("TrackShardAssignmentMinVersion", data->trackShardAssignmentMinVersion); co_return; } - FlowLock::Releaser holder(data->serveAuditStorageParallelismLock); + // The stopTrackShardAssignment() calls below are plain statements, so a cancelled audit destroys the + // coroutine frame without running any of them and strands trackShardAssignmentMinVersion set, which + // trips the check above for every later ssshard audit on this server. Idempotent, so it coexists with + // the early stops that close the tracking window before the audit finishes. + ScopeExit stopTrackingOnExit([data]() { data->stopTrackShardAssignment(); }); TraceEvent(SevInfo, "SSAuditStorageSsShardBegin", data->thisServerID) .detail("AuditId", req.id) .detail("AuditRange", req.range); @@ -3764,7 +4029,8 @@ Future auditStorageServerShardQ(StorageServer* data, AuditStorageRequest r int retryCount = 0; int64_t cumulatedValidatedLocalShardsNum = 0; int64_t cumulatedValidatedServerKeysNum = 0; - Reference rateLimiter = makeReference(SERVER_KNOBS->AUDIT_STORAGE_RATE_PER_SERVER_MAX, 1); + // Server-wide, not per-task: see StorageServer::auditStorageRateLimiter. + Reference rateLimiter = data->auditStorageRateLimiter; int64_t remoteReadBytes = 0; double startTime = now(); double lastRateLimiterWaitTime = 0; @@ -4179,6 +4445,44 @@ Future auditStorageServerShardQ(StorageServer* data, AuditStorageRequest r // Helper: Read both source and restored data for a given range // +// Per-batch validate_restore traces, which at scale run to millions of events per audit, so they are +// gated behind the existing ENABLE_AUDIT_VERBOSE_TRACE knob. The per-range events +// (SSAuditRestoreBegin/Complete, and any error) stay unconditional. +// +// TODO(Audit): move to fdbclient/Audit.h as auditVerboseEventSev(), beside the audit types, matching +// bulkLoadVerboseEventSev() in BulkLoading.h and s3VerboseEventSev() in S3Client.h. DataDistribution.cpp +// gates this same knob with six `if (SERVER_KNOBS->ENABLE_AUDIT_VERBOSE_TRACE)` blocks, so the subsystem +// has two idioms for one knob. Deferred because consolidating flips those six sites from emitting nothing +// to emitting SevDebug, which is a trace-volume change and wants measuring on its own. +static inline Severity auditRestoreVerboseEventSev() { + return SERVER_KNOBS->ENABLE_AUDIT_VERBOSE_TRACE ? SevInfo : SevDebug; +} + +// Adapt the byte budget of one validate_restore comparison batch (AIMD). See +// AUDIT_RESTORE_BATCH_BYTE_LIMIT for why a fixed size does not work. +// +// The increase is a quarter of the ceiling, coarse on purpose: a failure is cheap and self-correcting +// (one re-read), whereas sitting below the workable size costs throughput on every remaining batch. +// +// Never returns zero or below, because GetRangeLimits overloads the sign of `bytes`: -1 means +// BYTE_LIMIT_UNLIMITED and removes the bound, below -1 fails isValid(), and 0 costs a round trip per +// key. Not hang protection -- minRows=1 means even a zero budget still returns a key. +static int nextAuditBatchBytes(int currentBytes, bool succeeded, int floorBytes, int ceilingBytes) { + // Tolerate a misconfigured floor above the ceiling rather than producing an empty window. + floorBytes = std::max(1, floorBytes); + ceilingBytes = std::max(floorBytes, ceilingBytes); + currentBytes = std::clamp(currentBytes, floorBytes, ceilingBytes); + if (succeeded) { + const int increment = std::max(1, ceilingBytes / 4); + // Guard the add: currentBytes + increment can overflow int for a large ceiling. + if (currentBytes > ceilingBytes - increment) { + return ceilingBytes; + } + return currentBytes + increment; + } + return std::max(floorBytes, currentBytes / 2); +} + // Restored data is stored at validateRestoreLogKeys (\xff\x02/rlog/) in system key space. // NOTE: We read the ENTIRE restored keyspace (not just rangeToRead with prefix), // because restored keys are stored with their original names under the prefix. @@ -4194,7 +4498,7 @@ static Future> fetchSourceAndRes Key restoredEnd = rangeToRead.end.withPrefix(validateRestoreLogKeys.begin); KeyRange restoredRange = KeyRangeRef(restoredBegin, restoredEnd); - TraceEvent("SSAuditRestoreFetch", data->thisServerID) + TraceEvent(auditRestoreVerboseEventSev(), "SSAuditRestoreFetch", data->thisServerID) .detail("RangeToRead", rangeToRead) .detail("RestoredRange", restoredRange) .detail("Version", version) @@ -4210,8 +4514,18 @@ static Future> fetchSourceAndRes Transaction tr(data->cx); tr.setVersion(version); tr.setOption(FDBTransactionOptions::READ_LOCK_AWARE); + // ACCESS_SYSTEM_KEYS is needed for the restored range (it lives under \xff\x02/rlog/) and is set up + // front so both reads can be issued together. + tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); + // Issue both reads concurrently. They are independent ranges on the same transaction at the same + // explicit read version, so there is no ordering dependency between them; awaiting them in sequence + // paid two serial round trips per batch. GetRangeLimits sourceLimits(limit, limitBytes); - RangeResult sourceData = co_await tr.getRange(rangeToRead, sourceLimits, Snapshot::False, Reverse::False); + GetRangeLimits restoredLimits(limit, limitBytes); + Future sourceFuture = tr.getRange(rangeToRead, sourceLimits, Snapshot::False, Reverse::False); + Future restoredFuture = + tr.getRange(restoredRange, restoredLimits, Snapshot::False, Reverse::False); + RangeResult sourceData = co_await sourceFuture; // Convert source to GetKeyValuesReply format GetKeyValuesReply sourceReply; @@ -4220,10 +4534,7 @@ static Future> fetchSourceAndRes sourceReply.version = version; sourceResult = sourceReply; - // Read restored data with the same limits as source to ensure comparable results - tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); - GetRangeLimits restoredLimits(limit, limitBytes); - RangeResult restoredData = co_await tr.getRange(restoredRange, restoredLimits, Snapshot::False, Reverse::False); + RangeResult restoredData = co_await restoredFuture; // Convert restored to GetKeyValuesReply format GetKeyValuesReply restoredReply; @@ -4254,7 +4565,7 @@ static Future> fetchSourceAndRes } // Log what we fetched - TraceEvent("SSAuditRestoreFetchResult", data->thisServerID) + TraceEvent(auditRestoreVerboseEventSev(), "SSAuditRestoreFetchResult", data->thisServerID) .detail("SourceKeys", sourceResult.get().data.size()) .detail("RestoredKeys", restoredResult.get().data.size()) .detail("SourceBytes", sourceResult.get().data.expectedSize()) @@ -4280,7 +4591,7 @@ std::vector compareSourceAndRestoredData(UID thisServerID, int sourceIdx = 0; int restoredIdx = 0; - TraceEvent("SSAuditRestoreCompare", thisServerID) + TraceEvent(auditRestoreVerboseEventSev(), "SSAuditRestoreCompare", thisServerID) .detail("AuditID", auditID) .detail("SourceKeys", sourceReply.data.size()) .detail("RestoredKeys", restoredReply.data.size()) @@ -4289,17 +4600,17 @@ std::vector compareSourceAndRestoredData(UID thisServerID, // Log first few keys from both sets for debugging if (!sourceReply.data.empty()) { - TraceEvent("SSAuditRestoreCompareSourceKeys", thisServerID) + TraceEvent(auditRestoreVerboseEventSev(), "SSAuditRestoreCompareSourceKeys", thisServerID) .detail("FirstSourceKey", sourceReply.data[0].key) .detail("LastSourceKey", sourceReply.data[sourceReply.data.size() - 1].key); } if (!restoredReply.data.empty()) { - TraceEvent("SSAuditRestoreCompareRestoredKeys", thisServerID) + TraceEvent(auditRestoreVerboseEventSev(), "SSAuditRestoreCompareRestoredKeys", thisServerID) .detail("FirstRestoredKey", restoredReply.data[0].key) .detail("LastRestoredKey", restoredReply.data[restoredReply.data.size() - 1].key); } - TraceEvent("SSAuditRestoreCompareStart", thisServerID) + TraceEvent(auditRestoreVerboseEventSev(), "SSAuditRestoreCompareStart", thisServerID) .detail("SourceSize", sourceReply.data.size()) .detail("RestoredSize", restoredReply.data.size()) .detail("SourceMore", sourceReply.more) @@ -4403,7 +4714,7 @@ std::vector compareSourceAndRestoredData(UID thisServerID, errors.push_back(error); } - TraceEvent("SSAuditRestoreCompareEnd", thisServerID) + TraceEvent(auditRestoreVerboseEventSev(), "SSAuditRestoreCompareEnd", thisServerID) .detail("SourceIdx", sourceIdx) .detail("RestoredIdx", restoredIdx) .detail("SourceSize", sourceReply.data.size()) @@ -4441,7 +4752,20 @@ Future auditRestoreQ(StorageServer* data, AuditStorageRequest req) { Key rangeToReadBegin = req.range.begin; KeyRange claimRange; int limit = SERVER_KNOBS->AUDIT_RESTORE_BATCH_KEY_LIMIT; // Use knob instead of hardcoded 10K - int limitBytes = CLIENT_KNOBS->REPLY_BYTE_LIMIT; + // Ceiling of an adaptive budget, not a fixed batch size. Deliberately not + // CLIENT_KNOBS->REPLY_BYTE_LIMIT, which is what silently defeated the key limit above; see that knob + // and nextAuditBatchBytes(). + const int limitBytesCeiling = SERVER_KNOBS->AUDIT_RESTORE_BATCH_BYTE_LIMIT; + const int limitBytesFloor = std::min(limitBytesCeiling, SERVER_KNOBS->AUDIT_RESTORE_BATCH_BYTE_LIMIT_MIN); + // Seeded through nextAuditBatchBytes() so a nonsensical knob pair cannot put the FIRST batch outside + // the window. Returns the ceiling unchanged for any sane configuration. + int limitBytes = nextAuditBatchBytes(limitBytesCeiling, true, limitBytesFloor, limitBytesCeiling); + int consecutiveRetries = 0; + int64_t totalRetryableErrors = 0; + // Jittered geometric backoff on the house knobs (DEFAULT_BACKOFF -> DEFAULT_MAX_BACKOFF at + // BACKOFF_GROWTH_RATE), reset on success so an audit that hits an occasional retry does not ratchet + // to the maximum and stay there. + Backoff retryBackoff; int64_t readBytes = 0; Error nonRetryableError; int64_t numValidatedKeys = 0; @@ -4449,12 +4773,23 @@ Future auditRestoreQ(StorageServer* data, AuditStorageRequest req) { int64_t lastPersistBytes = 0; // Track when we last persisted progress bool complete = false; double startTime = now(); - Reference rateLimiter = makeReference(SERVER_KNOBS->AUDIT_STORAGE_RATE_PER_SERVER_MAX, 1); + // Time spent blocked on the audit rate limiter, which is now shared per server rather than per task. + double rateLimiterTotalWaitTime = 0; + // Server-wide, not per-task: see StorageServer::auditStorageRateLimiter. + Reference rateLimiter = data->auditStorageRateLimiter; { Optional err; try { while (true) { + // Snapshot every piece of per-attempt state: a retryable failure re-reads this same batch + // from the same start key. `complete` matters most -- a stale true would end the audit + // early and report Complete over a range never fully read; only the persist block's + // !complete guard makes that unreachable today. + const int64_t batchStartValidatedKeys = numValidatedKeys; + const int64_t batchStartValidatedBytes = validatedBytes; + const int64_t batchStartPersistBytes = lastPersistBytes; + const bool batchStartComplete = complete; try { readBytes = 0; rangeToRead = KeyRangeRef(rangeToReadBegin, req.range.end); @@ -4561,8 +4896,12 @@ Future auditRestoreQ(StorageServer* data, AuditStorageRequest req) { lastPersistBytes = validatedBytes; } - // Apply rate limiting + // Apply rate limiting. The limiter is shared by every concurrent audit task on this + // server, so unlike the old per-task limiter it can actually bind; account for the + // wait so a throttled audit is distinguishable from a slow one. + const double rateLimiterBeforeWaitTime = now(); co_await rateLimiter->getAllowance(readBytes); + rateLimiterTotalWaitTime += now() - rateLimiterBeforeWaitTime; // If errors found or complete, break if (!errors.empty() || complete) { @@ -4571,6 +4910,10 @@ Future auditRestoreQ(StorageServer* data, AuditStorageRequest req) { // Move to next range rangeToReadBegin = keyAfter(lastKey); + // The batch succeeded, so let the byte budget grow back toward its ceiling. + consecutiveRetries = 0; + retryBackoff = Backoff(); + limitBytes = nextAuditBatchBytes(limitBytes, true, limitBytesFloor, limitBytesCeiling); if (rangeToReadBegin >= req.range.end) { complete = true; break; @@ -4587,13 +4930,32 @@ Future auditRestoreQ(StorageServer* data, AuditStorageRequest req) { if (nonRetryableError.code() == error_code_future_version || nonRetryableError.code() == error_code_transaction_too_old || nonRetryableError.code() == error_code_server_overloaded) { + // This batch is retried from the same start key, so undo its partial accounting. + numValidatedKeys = batchStartValidatedKeys; + validatedBytes = batchStartValidatedBytes; + lastPersistBytes = batchStartPersistBytes; + complete = batchStartComplete; + const int retriedBytes = limitBytes; + limitBytes = nextAuditBatchBytes(limitBytes, false, limitBytesFloor, limitBytesCeiling); + // A flat 1s per retry was pure latency when the remedy is a smaller batch, and + // these can number in the thousands per server. + ++consecutiveRetries; + ++totalRetryableErrors; + // Suppressed, with the running total reported in SSAuditRestoreComplete. Retrying + // is the normal path -- the budget halving above exists because these are expected. + // Event name kept for greps. TraceEvent(SevWarn, "SSAuditRestoreRetryableError", data->thisServerID) + .suppressFor(10.0) .detail("AuditID", req.id) .detail("Error", nonRetryableError.what()) .detail("Version", version) - .detail("Range", rangeToRead); + .detail("Range", rangeToRead) + .detail("BatchBytes", retriedBytes) + .detail("NextBatchBytes", limitBytes) + .detail("ConsecutiveRetries", consecutiveRetries) + .detail("TotalRetryableErrors", totalRetryableErrors); nonRetryableError = Error(); - co_await delay(1.0); + co_await retryBackoff.onError(); continue; } throw nonRetryableError; @@ -4612,7 +4974,10 @@ Future auditRestoreQ(StorageServer* data, AuditStorageRequest req) { .detail("ValidationErrors", errors.size()) .detail("NumValidatedKeys", numValidatedKeys) .detail("ValidatedBytes", validatedBytes) - .detail("Duration", now() - startTime); + .detail("Duration", now() - startTime) + .detail("RateLimiterTotalWaitTime", rateLimiterTotalWaitTime) + .detail("FinalBatchBytes", limitBytes) + .detail("TotalRetryableErrors", totalRetryableErrors); } else { res.setPhase(AuditPhase::Complete); res.range = req.range; @@ -4622,7 +4987,10 @@ Future auditRestoreQ(StorageServer* data, AuditStorageRequest req) { .detail("Complete", complete) .detail("NumValidatedKeys", numValidatedKeys) .detail("ValidatedBytes", validatedBytes) - .detail("Duration", now() - startTime); + .detail("Duration", now() - startTime) + .detail("RateLimiterTotalWaitTime", rateLimiterTotalWaitTime) + .detail("FinalBatchBytes", limitBytes) + .detail("TotalRetryableErrors", totalRetryableErrors); } // Persist final audit state @@ -4689,7 +5057,8 @@ Future auditStorageShardReplicaQ(StorageServer* data, AuditStorageRequest double lastRateLimiterWaitTime = 0; double rateLimiterBeforeWaitTime = 0; double rateLimiterTotalWaitTime = 0; - Reference rateLimiter = makeReference(SERVER_KNOBS->AUDIT_STORAGE_RATE_PER_SERVER_MAX, 1); + // Server-wide, not per-task: see StorageServer::auditStorageRateLimiter. + Reference rateLimiter = data->auditStorageRateLimiter; try { while (true) { { @@ -5111,6 +5480,18 @@ Future getRangeDataToDump(StorageServer* data, if (e.code() == error_code_actor_cancelled) { throw e; } + TraceEvent(SevWarn, "SSBulkDumpRangeReadFailed", data->thisServerID) + .errorUnsuppressed(e) + .detail("Range", range) + .detail("BeginKey", beginKey) + .detail("Version", version) + .detail("FirstRead", immediateError); + if (immediateError) { + // Rethrow rather than collapsing into retry(): the caller gives up immediately on + // wrong_shard_server -- a stale range assignment, which no amount of retrying against + // this server can fix -- and retries anything else. + throw e; + } break; } @@ -5148,9 +5529,9 @@ Future getRangeDataToDump(StorageServer* data, beginKey = keyAfter(output->lastKey); } - if (immediateError) { - throw retry(); - } + // A first-read failure throws from the catch above, so reaching here means at least one read + // succeeded. immediateError is cleared immediately after the try/catch, before any other break. + ASSERT(!immediateError); } // The SS actor handling bulk dump task sent from DD. @@ -5295,7 +5676,16 @@ Future bulkDumpQ(StorageServer* data, BulkDumpRequest req) { if (e.code() == error_code_actor_cancelled) { throw e; } - TraceEvent(SevWarn, "SSBulkDumpError", data->thisServerID) + // Retrying is the normal path here: this loop allows up to 50 attempts a second apart while + // shards move underneath the dump, and one task routinely burns dozens of them. Only the + // terminal outcome is warning-worthy. Reporting every attempt at SevWarn reads as thousands of + // failures rather than the handful of tasks that actually gave up. + const bool giveUp = e.code() == error_code_bulkdump_task_outdated || + e.code() == error_code_wrong_shard_server || e.code() == error_code_platform_error || + e.code() == error_code_io_error || retryCount >= 50; + TraceEvent(giveUp ? SevWarn : bulkLoadVerboseEventSev(), + giveUp ? "SSBulkDumpError" : "SSBulkDumpRetry", + data->thisServerID) .errorUnsuppressed(e) .detail("TaskID", req.bulkDumpState.getTaskId()) .detail("TaskRange", req.bulkDumpState.getRange()) @@ -5306,8 +5696,7 @@ Future bulkDumpQ(StorageServer* data, BulkDumpRequest req) { req.reply.sendError(bulkdump_task_outdated()); // give up break; // silently exit } - if (e.code() == error_code_wrong_shard_server || e.code() == error_code_platform_error || - e.code() == error_code_io_error || retryCount >= 50) { + if (giveUp) { req.reply.sendError(bulkdump_task_failed()); // give up break; // silently exit } @@ -5720,8 +6109,8 @@ Future getMappedKeyValuesQ(StorageServer* data, GetMappedKeyValuesRequest data->transactionTagCounter.addRequest(req.tags, resultSize); double duration = g_network->timer() - req.requestTime(); - data->counters.readLatencySamples.sample(duration, ReadLatencySamples::READ, trackedReadType(req)); - data->counters.readLatencySamples.sample(duration, ReadLatencySamples::MAPPED_RANGE, trackedReadType(req)); + data->counters.readLatencySamples.samplePair( + duration, ReadLatencySamples::READ, ReadLatencySamples::MAPPED_RANGE, trackedReadType(req)); if (data->latencyBandConfig.present()) { int maxReadBytes = data->latencyBandConfig.get().readConfig.maxReadBytes.orDefault(std::numeric_limits::max()); @@ -6011,8 +6400,8 @@ Future getKeyQ(StorageServer* data, GetKeyRequest req) { data->transactionTagCounter.addRequest(req.tags, resultSize); double duration = g_network->timer() - req.requestTime(); - data->counters.readLatencySamples.sample(duration, ReadLatencySamples::READ, trackedReadType(req)); - data->counters.readLatencySamples.sample(duration, ReadLatencySamples::READ_KEY, trackedReadType(req)); + data->counters.readLatencySamples.samplePair( + duration, ReadLatencySamples::READ, ReadLatencySamples::READ_KEY, trackedReadType(req)); if (data->latencyBandConfig.present()) { int maxReadBytes = @@ -6797,6 +7186,55 @@ static BulkLoadFileSetKeyMap selectOwnedBulkLoadFileSets(const BulkLoadFileSetKe return owned; } +TEST_CASE("/fdbserver/storageserver/nextAuditBatchBytes") { + const int floorBytes = 256000; + const int ceilingBytes = 4000000; + + // Failure halves, and keeps halving, but never below the floor. The floor is not hang protection + // (GetRangeLimits sets minRows=1, so even a zero budget still returns a key): it exists because a + // budget of -1 reads as BYTE_LIMIT_UNLIMITED, below -1 fails as range_limits_invalid, and 0 costs a + // round trip per key. See nextAuditBatchBytes(). + ASSERT_EQ(nextAuditBatchBytes(ceilingBytes, false, floorBytes, ceilingBytes), 2000000); + ASSERT_EQ(nextAuditBatchBytes(2000000, false, floorBytes, ceilingBytes), 1000000); + int budget = ceilingBytes; + for (int i = 0; i < 100; ++i) { + budget = nextAuditBatchBytes(budget, false, floorBytes, ceilingBytes); + ASSERT(budget >= floorBytes && budget <= ceilingBytes); + } + ASSERT_EQ(budget, floorBytes); + + // Success increases additively (a quarter of the ceiling) and saturates at the ceiling rather than + // overshooting it. + ASSERT_EQ(nextAuditBatchBytes(floorBytes, true, floorBytes, ceilingBytes), floorBytes + ceilingBytes / 4); + ASSERT_EQ(nextAuditBatchBytes(ceilingBytes, true, floorBytes, ceilingBytes), ceilingBytes); + budget = floorBytes; + for (int i = 0; i < 100; ++i) { + budget = nextAuditBatchBytes(budget, true, floorBytes, ceilingBytes); + ASSERT(budget >= floorBytes && budget <= ceilingBytes); + } + ASSERT_EQ(budget, ceilingBytes); + + // A huge ceiling must not overflow int when the increment is added. + ASSERT_EQ( + nextAuditBatchBytes(std::numeric_limits::max() - 1, true, floorBytes, std::numeric_limits::max()), + std::numeric_limits::max()); + + // Degenerate configurations: floor above ceiling, and a zero/negative current budget. + ASSERT_EQ(nextAuditBatchBytes(1000, false, 5000, 1000), 5000); // floor wins, clamped up + ASSERT(nextAuditBatchBytes(0, false, floorBytes, ceilingBytes) >= floorBytes); + ASSERT(nextAuditBatchBytes(-1, true, floorBytes, ceilingBytes) >= floorBytes); + ASSERT(nextAuditBatchBytes(0, false, 0, 0) >= 1); // never returns a budget of zero + + // Randomized: the budget stays inside the window no matter the sequence of outcomes. + budget = deterministicRandom()->randomInt(floorBytes, ceilingBytes); + for (int i = 0; i < 1000; ++i) { + budget = nextAuditBatchBytes(budget, deterministicRandom()->coinflip(), floorBytes, ceilingBytes); + ASSERT(budget >= floorBytes && budget <= ceilingBytes); + } + + return Void(); +} + TEST_CASE("/fdbserver/storageserver/selectOwnedBulkLoadFileSets") { auto entry = [](StringRef begin, StringRef end) { return std::make_pair(KeyRange(KeyRangeRef(begin, end)), BulkLoadFileSet()); @@ -7163,7 +7601,14 @@ Future fetchKeys(StorageServer* data, AddingShard* shard) { KeyRange firstUncontainedFileRange; auto ownedBulkLoadFileSets = std::make_shared(selectOwnedBulkLoadFileSets( *localBulkLoadFileSets, keys, &allFilesContained, &firstUncontainedFileRange)); + // Coverage probes: the filtering above only matters when a task spans more than one + // destination shard piece. Deliberately unannotated -- probe::assert::simOnly asserts a path + // is reachable ONLY under simulation, and both of these are normal production behaviour. + if (ownedBulkLoadFileSets->size() < localBulkLoadFileSets->size()) { + CODE_PROBE(true, "bulkload fetchKeys skipped a sibling fetch's file"); + } if (!allFilesContained) { + CODE_PROBE(true, "bulkload file straddles the fetched shard piece"); TraceEvent(SevInfo, "SSBulkLoadFileRangeMismatch", data->thisServerID) .detail("ShardRange", keys) .detail("FileRange", firstUncontainedFileRange) @@ -9305,7 +9750,7 @@ class StorageUpdater { splitMutation(data, data->shards, m, ver, fromFetch); } - if (data->otherError.getFuture().isReady()) + if (data->otherError.isSet()) data->otherError.getFuture().get(); } @@ -9970,7 +10415,7 @@ Future update(StorageServer* data, bool* pReceivedUpdate) { // DEBUG_KEY_RANGE("SSUpdate", ver, KeyRangeRef()); data->mutableData().createNewVersion(ver); - if (data->otherError.getFuture().isReady()) + if (data->otherError.isSet()) data->otherError.getFuture().get(); data->counters.fetchedVersions += (ver - data->version.get()); @@ -9993,7 +10438,7 @@ Future update(StorageServer* data, bool* pReceivedUpdate) { data->version.set(ver); // Triggers replies to waiting gets for new version(s) setDataVersion(data->thisServerID, data->version.get()); - if (data->otherError.getFuture().isReady()) + if (data->otherError.isSet()) data->otherError.getFuture().get(); Version maxVersionsInMemory = (g_network->isSimulated() && g_simulator->speedUpSimulation) @@ -11606,10 +12051,44 @@ Future checkBehind(StorageServer* self) { } } +Future dispatchGetValueRequests(StorageServer* self, TaskPriority priority) { + ASSERT(priority == TaskPriority::DefaultEndpoint || priority == TaskPriority::LowPriorityRead); + auto& pending = + priority == TaskPriority::LowPriorityRead ? self->pendingLowPriorityGetValues : self->pendingDefaultGetValues; + auto& scheduled = priority == TaskPriority::LowPriorityRead ? self->lowPriorityGetValueDispatchScheduled + : self->defaultGetValueDispatchScheduled; + ASSERT(scheduled); + Future readyCheckpoint = Void(); + + while (true) { + // Preserve the asynchronous ReadSocket-to-read-priority boundary while amortizing its scheduler task. + co_await delay(0, priority); + const size_t batchSize = std::min(32, pending.size()); + for (size_t i = 0; i < batchSize; ++i) { + StorageServer::GetValueQuery query = std::move(pending.front()); + pending.pop_front(); + Optional> handler = getValueQDispatched(self, std::move(query)); + if (handler.present()) { + self->actors.add(std::move(handler.get())); + } + // Preserve preemption and the cancellation checkpoint without allocating a ready Future for every GET. + if (check_yield(priority)) { + co_await yield(priority); + } else { + g_network->setCurrentTask(priority); + co_await readyCheckpoint; + } + } + if (pending.empty()) { + scheduled = false; + co_return; + } + } +} + Future serveGetValueRequests(StorageServer* self, FutureStream getValue) { getCurrentLineage()->modify(&TransactionLineage::operation) = TransactionLineage::Operation::GetValue; - while (true) { - GetValueRequest req = co_await getValue; + auto enqueue = [self](GetValueRequest req) { // Warning: This code is executed at extremely high priority (TaskPriority::LoadBalancedEndpoint), so // downgrade before doing real work if (req.options.present() && req.options.get().debugID.present()) { @@ -11620,10 +12099,36 @@ Future serveGetValueRequests(StorageServer* self, FutureStreamgetCurrentTask()); } - if (SHORT_CIRCUT_ACTUAL_STORAGE && normalKeys.contains(req.key)) + if (SHORT_CIRCUT_ACTUAL_STORAGE && normalKeys.contains(req.key)) { req.reply.send(GetValueReply()); - else - self->actors.add(self->readGuard(req, getValueQ)); + } else if (self->shouldRead(req)) { + // Start spans before the priority delay and preserve debug event ordering. + if (req.spanContext.isSampled() || (req.options.present() && req.options.get().debugID.present())) { + self->actors.add(getValueQ(self, std::move(req))); + return; + } + + StorageServer::GetValueQuery query(std::move(req), self->counters); + beginGetValueQ(self, query.getRequest()); + bool lowPriority = self->useLowPriorityRead(); + if (lowPriority) { + ++self->counters.lowPriorityQueries; + } + auto& pending = lowPriority ? self->pendingLowPriorityGetValues : self->pendingDefaultGetValues; + auto& scheduled = + lowPriority ? self->lowPriorityGetValueDispatchScheduled : self->defaultGetValueDispatchScheduled; + pending.emplace_back(std::move(query)); + if (!scheduled) { + scheduled = true; + self->actors.add(dispatchGetValueRequests( + self, lowPriority ? TaskPriority::LowPriorityRead : TaskPriority::DefaultEndpoint)); + } + } + }; + + while (true) { + GetValueRequest req = co_await getValue; + enqueue(std::move(req)); } } @@ -12043,8 +12548,7 @@ Future serveAuditStorageRequests(StorageServer* self, FutureStream ready)` + /// See also `documentation/coro_tutorial/tutorial.cpp::someFuture(Future ready)`. TestCase("1:1 simulate a 'loop choose {}' in Swift, with TaskGroup") { - // ACTOR Future someFuture(Future ready) { - // loop choose { - // when(wait(delay(0.5))) { std::cout << "Still waiting...\n"; } - // when(int r = wait(ready)) { - // std::cout << format("Ready %d\n", r); - // wait(delay(double(r))); - // std::cout << "Done\n"; - // return Void(); - // } - // } - // } let promise = PromiseVoid() Task { diff --git a/fdbserver/tester/TesterServer.cpp b/fdbserver/tester/TesterServer.cpp index e5bc765185b..86f812a67f7 100644 --- a/fdbserver/tester/TesterServer.cpp +++ b/fdbserver/tester/TesterServer.cpp @@ -117,6 +117,7 @@ Future> getWorkloadIface(WorkloadRequest work, wcx.rangesToCheck = work.rangesToCheck; // FIXME: Other stuff not filled in; why isn't this constructed here and passed down to the other // getWorkloadIface()? + ifaces.reserve(work.options.size()); for (int i = 0; i < work.options.size(); i++) { ifaces.push_back(getWorkloadIface(work, ccr, work.options[i], dbInfo)); } diff --git a/fdbserver/worker/MetricClient.cpp b/fdbserver/worker/MetricClient.cpp index a788105dffd..af437445557 100644 --- a/fdbserver/worker/MetricClient.cpp +++ b/fdbserver/worker/MetricClient.cpp @@ -119,6 +119,7 @@ void UDPMetricClient::send(MetricCollection* metrics) { metrics->histMap.clear(); + gauges.reserve(metrics->gaugeMap.size()); for (auto& [_, g] : metrics->gaugeMap) { gauges.push_back(std::move(g)); } diff --git a/fdbserver/workloads/AutomaticIdempotencyWorkload.cpp b/fdbserver/workloads/AutomaticIdempotencyWorkload.cpp index abdf4dbe6aa..01c5ce469fb 100644 --- a/fdbserver/workloads/AutomaticIdempotencyWorkload.cpp +++ b/fdbserver/workloads/AutomaticIdempotencyWorkload.cpp @@ -277,6 +277,7 @@ struct AutomaticIdempotencyWorkload : TestWorkload { Error err; try { std::vector>> futures; + futures.reserve(keys.size()); for (auto const& key : keys) { futures.push_back(tr.get(key)); diff --git a/fdbserver/workloads/BackupToDBCorrectness.cpp b/fdbserver/workloads/BackupToDBCorrectness.cpp index 6ebbe146f48..2efd21e8047 100644 --- a/fdbserver/workloads/BackupToDBCorrectness.cpp +++ b/fdbserver/workloads/BackupToDBCorrectness.cpp @@ -196,6 +196,7 @@ struct BackupToDBCorrectnessWorkload : TestWorkload { co_await waitForAll(results); std::vector ret; + ret.reserve(results.size()); for (const auto& result : results) { ret.push_back(result.get()); } diff --git a/fdbserver/workloads/BulkLoading.cpp b/fdbserver/workloads/BulkLoading.cpp index 73e51da972c..4a04b6930e0 100644 --- a/fdbserver/workloads/BulkLoading.cpp +++ b/fdbserver/workloads/BulkLoading.cpp @@ -442,6 +442,7 @@ struct BulkLoading : TestWorkload { std::vector getAllKeys(const std::vector& kvs) { std::vector res; + res.reserve(kvs.size()); for (const auto& kv : kvs) { res.push_back(kv.key); } @@ -556,6 +557,7 @@ struct BulkLoading : TestWorkload { oldBulkLoadMode = co_await setBulkLoadMode(cx, 1); TraceEvent("BulkLoadingWorkLoadSimpleTestSetMode").detail("OldMode", oldBulkLoadMode).detail("NewMode", 1); std::vector errorTasks = co_await self->waitUntilAllTaskCompleteOrError(self, cx); + errorRanges.reserve(errorTasks.size()); for (const auto& errorTask : errorTasks) { errorRanges.push_back(errorTask.getRange()); } @@ -667,6 +669,7 @@ struct BulkLoading : TestWorkload { } // Wait until all tasks have completed std::vector errorTasks = co_await self->waitUntilAllTaskCompleteOrError(self, cx); + errorRanges.reserve(errorTasks.size()); for (const auto& errorTask : errorTasks) { errorRanges.push_back(errorTask.getRange()); // for any error range, do not check data } @@ -759,6 +762,7 @@ struct BulkLoading : TestWorkload { if (backgroundTrafficEnabled) { std::vector> trafficActors; int actorCount = deterministicRandom()->randomInt(1, 10); + trafficActors.reserve(actorCount); for (int i = 0; i < actorCount; i++) { trafficActors.push_back(backgroundWriteTraffic(this, cx)); } diff --git a/fdbserver/workloads/BulkSetup.h b/fdbserver/workloads/BulkSetup.h index 3d03a062342..fe8d974b8e0 100644 --- a/fdbserver/workloads/BulkSetup.h +++ b/fdbserver/workloads/BulkSetup.h @@ -327,6 +327,7 @@ Future bulkSetup(Database cx, keySaveIncrement = 0; } + fs.reserve(BULK_SETUP_WORKERS); for (int j = 0; j < BULK_SETUP_WORKERS; j++) { fs.push_back(setupRangeWorker(cx, workload, &jobs, maxWorkerInsertRate, keySaveIncrement, j)); } diff --git a/fdbserver/workloads/ConsistencyCheckUrgent.cpp b/fdbserver/workloads/ConsistencyCheckUrgent.cpp index 00bca3c11b0..e2b94b42393 100644 --- a/fdbserver/workloads/ConsistencyCheckUrgent.cpp +++ b/fdbserver/workloads/ConsistencyCheckUrgent.cpp @@ -275,6 +275,7 @@ struct ConsistencyCheckUrgentWorkload : TestWorkload { tr.setOption(FDBTransactionOptions::READ_SYSTEM_KEYS); tr.setOption(FDBTransactionOptions::LOCK_AWARE); std::vector>> serverListEntries; + serverListEntries.reserve(storageServers.size()); for (int s = 0; s < storageServers.size(); s++) serverListEntries.push_back(tr.get(serverListKeyFor(storageServers[s]))); std::vector> serverListValues = co_await getAll(serverListEntries); diff --git a/fdbserver/workloads/DegradedMultiRegionStatus.cpp b/fdbserver/workloads/DegradedMultiRegionStatus.cpp new file mode 100644 index 00000000000..82543d7261e --- /dev/null +++ b/fdbserver/workloads/DegradedMultiRegionStatus.cpp @@ -0,0 +1,409 @@ +/* + * DegradedMultiRegionStatus.cpp + * + * This source file is part of the FoundationDB open source project + * + * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +// Integration test for the "degraded multi-region" status signal. +// +// Setup: a two-usable-region cluster (generateFearless) with primary region "0" +// (with satellite "2") and remote region "1". Coordinators live in the non-primary +// regions because minimumRegions > 1. +// +// Scenario mirroring production: kill ONLY the primary datacenter "0" of the first +// region, leaving its satellite "2" and the remote region "1"/"3" alive. The cluster +// fails over to the remote region, storage servers recover data from the surviving +// satellite "2", and recovery gets stuck at accepting_commits because the dead +// primary's log set cannot be recruited (allLogs == false => RemoteRegionLogsMissing=1). +// After the stall persists past DEGRADED_MULTI_REGION_MIN_STALL_SECONDS the test reads +// the status JSON and asserts: +// - cluster.degraded_multi_region == true +// - cluster.data.state.description contains "Degraded multiregional" +// +// Before killing the primary DC the test also exercises the false-positive regression +// from the remote log set being transiently absent at accepting_commits: it verifies +// (a) a fully healthy cluster is not flagged, and (b) a *normal* recovery of a healthy +// cluster (triggered by killing only the sequencer, no region loss) keeps +// degraded_multi_region == false throughout the recovery. +// +// Finally it resets usable_regions=1 so subsequent checks are not stuck. + +#include "fdbclient/NativeAPI.h" +#include "fdbserver/core/TesterInterface.h" +#include "fdbserver/core/WorkerInterface.h" +#include "fdbserver/tester/workloads.h" +#include "fdbserver/core/FDBSimulationPolicy.h" +#include "fdbserver/core/RecoveryState.h" +#include "fdbserver/core/ServerDBInfo.h" +#include "fdbrpc/simulator.h" +#include "fdbclient/ManagementAPI.h" +#include "flow/CoroUtils.h" +#include "fdbclient/ReadYourWrites.h" +#include "fdbclient/json_spirit/json_spirit_value.h" + +struct DegradedMultiRegionStatusWorkload : TestWorkload { + static constexpr auto NAME = "DegradedMultiRegionStatus"; + bool enabled; + double testDuration; + bool testSucceeded; + + explicit DegradedMultiRegionStatusWorkload(WorkloadContext const& wcx) : TestWorkload(wcx) { + enabled = !clientId && g_network->isSimulated(); + testDuration = getOption(options, "testDuration"_sr, 120.0); + testSucceeded = false; + } + + void disableFailureInjectionWorkloads(std::set& out) const override { out.insert("all"); } + + Future setup(Database const& cx) override { + if (enabled) { + return _setup(cx); + } + return Void(); + } + Future start(Database const& cx) override { + if (enabled) { + return killPrimaryKeepSatellites(cx); + } + return Void(); + } + Future check(Database const& cx) override { return enabled ? testSucceeded : true; } + void getMetrics(std::vector& m) override {} + + // Wait until the cluster is fully recovered (stable starting point) before killing anything. + Future _setup(Database cx) { + double failedWait = 0.0; + while (dbInfo->get().recoveryState < RecoveryState::FULLY_RECOVERED) { + if (failedWait >= 300.0) { + TraceEvent(SevError, "DegradedMultiRegionStatus_FullRecoveryTimeout") + .detail("Elapsed", failedWait) + .detail("RecoveryState", dbInfo->get().recoveryState); + ASSERT(false); + } + co_await delay(1.0); + failedWait += 1.0; + } + TraceEvent("DegradedMultiRegionStatus_Setup").log(); + } + + // Read the degraded_multi_region flag from the status JSON. Returns false on any + // transient read/parse error (treated as "not degraded") so callers only fail on an + // explicit true. + Future readDegradedFlag(Database cx) { + ReadYourWritesTransaction tr(cx); + try { + tr.setOption(FDBTransactionOptions::READ_SYSTEM_KEYS); + Optional statusVal = co_await tr.get("\xff\xff/status/json"_sr); + if (statusVal.present()) { + json_spirit::mValue mv; + json_spirit::read_string(statusVal.get().toString(), mv); + auto& root = mv.get_obj(); + if (root.contains("cluster")) { + auto& clusterObj = root["cluster"].get_obj(); + if (clusterObj.contains("degraded_multi_region")) { + co_return clusterObj["degraded_multi_region"].get_bool(); + } + } + } + } catch (Error& e) { + if (e.code() == error_code_actor_cancelled) { + throw; + } + // Transient read errors: treat as not degraded. + } + co_return false; + } + + // Verify that a healthy (both regions up) cluster is not falsely flagged as + // degraded multi-region while idle. + Future verifyHealthyNotDegraded(Database cx) { + double tStart = now(); + while (true) { + bool degraded = co_await readDegradedFlag(cx); + if (degraded) { + TraceEvent(SevError, "DegradedMultiRegionStatus_HealthyFalsePositive") + .detail("Elapsed", now() - tStart) + .detail("Phase", "idle"); + co_return false; + } + if (now() - tStart > 10.0) { + co_return true; + } + co_await delay(1.0); + } + } + + // Trigger a *normal* recovery with no region loss by killing only the sequencer + // (master), then poll the status across the recovery window and assert that + // degraded_multi_region stays false the whole time. This exercises the exact + // regression where the remote log set is transiently absent at accepting_commits. + Future verifyNormalRecoveryNotDegraded(Database cx) { + ASSERT(g_network->isSimulated()); + + // Kill the current sequencer to force a recovery while keeping every region alive. + NetworkAddress masterAddr = dbInfo->get().master.address(); + ISimulator::ProcessInfo* masterProcess = g_simulator->getProcessByAddress(masterAddr); + if (masterProcess == nullptr) { + TraceEvent(SevError, "DegradedMultiRegionStatus_MasterProcessNotFound").detail("Address", masterAddr); + co_return false; + } + LifetimeToken preKillMasterLifetime = dbInfo->get().masterLifetime; + TraceEvent("DegradedMultiRegionStatus_KillSequencer").detail("Address", masterAddr); + g_simulator->killProcess(masterProcess, ISimulator::KillType::KillInstantly); + + // Wait until the kill is observed as a new recovery start. A changed ServerDBInfo + // alone is not a reliable recovery marker (broadcasts also happen for unrelated + // reasons), so require either a new masterLifetime or a drop from FULLY_RECOVERED. + // onChange() is awaited in a loop because any single change may not belong to + // this recovery; without this wait the polling below could pass on the stale + // fully-recovered state without ever observing the new accepting_commits window. + double observedAt = now(); + constexpr double observeTimeout = 60.0; + while (true) { + // isEqual is not const-qualified, so compare against a local copy. + LifetimeToken currentMasterLifetime = dbInfo->get().masterLifetime; + if (!currentMasterLifetime.isEqual(preKillMasterLifetime) || + dbInfo->get().recoveryState < RecoveryState::FULLY_RECOVERED) { + break; // recovery started: new lifetime or the state dropped + } + if (now() - observedAt > observeTimeout) { + TraceEvent(SevError, "DegradedMultiRegionStatus_NormalRecoveryNotObserved") + .detail("RecoveryState", dbInfo->get().recoveryState); + co_return false; + } + co_await race(dbInfo->onChange(), delay(1.0)); + } + + double tStart = now(); + double deadline = 120.0; + uint8_t cycles = 10; + while (true) { + bool degraded = co_await readDegradedFlag(cx); + if (degraded) { + TraceEvent(SevError, "DegradedMultiRegionStatus_NormalRecoveryFalsePositive") + .detail("Elapsed", now() - tStart) + .detail("RecoveryState", dbInfo->get().recoveryState); + co_return false; + } + if (dbInfo->get().recoveryState >= RecoveryState::ALL_LOGS_RECRUITED) { + TraceEvent("DegradedMultiRegionStatus_NormalRecoveryClean") + .detail("Elapsed", now() - tStart) + .detail("RecoveryState", dbInfo->get().recoveryState); + if (cycles == 0) { + co_return true; + } + cycles--; + } else { + cycles = 10; // reset the hold count if recovery is still in progress + } + if (now() - tStart > deadline) { + TraceEvent(SevError, "DegradedMultiRegionStatus_NormalRecoveryTimeout") + .detail("Elapsed", now() - tStart) + .detail("RecoveryState", dbInfo->get().recoveryState); + co_return false; + } + co_await delay(1.0); + } + } + + // Wait until status JSON reports the desired degraded multi-region state. + Future waitForDegradedStatus(Database cx) { + double tStart = now(); + Optional degradedSince; + while (true) { + ReadYourWritesTransaction tr(cx); + try { + tr.setOption(FDBTransactionOptions::READ_SYSTEM_KEYS); + Optional statusVal = co_await tr.get("\xff\xff/status/json"_sr); + if (statusVal.present()) { + json_spirit::mValue mv; + json_spirit::read_string(statusVal.get().toString(), mv); + auto& root = mv.get_obj(); + if (root.contains("cluster")) { + auto& clusterObj = root["cluster"].get_obj(); + bool degraded = false; + if (clusterObj.contains("degraded_multi_region")) { + degraded = clusterObj["degraded_multi_region"].get_bool(); + } + std::string dataStateDesc; + if (clusterObj.contains("data") && clusterObj["data"].get_obj().contains("state") && + clusterObj["data"].get_obj()["state"].get_obj().contains("description")) { + dataStateDesc = clusterObj["data"].get_obj()["state"].get_obj()["description"].get_str(); + } + + bool acceptingCommits = + clusterObj.contains("recovery_state") && + clusterObj["recovery_state"].get_obj().contains("name") && + clusterObj["recovery_state"].get_obj()["name"].get_str() == "accepting_commits"; + + const bool expectedState = degraded && acceptingCommits && + dataStateDesc.find("Degraded multiregional") != std::string::npos; + + std::string recoveryStateJson; + if (clusterObj.contains("recovery_state")) { + recoveryStateJson = json_spirit::write_string(clusterObj["recovery_state"], + json_spirit::Output_options::none); + } + TraceEvent("DegradedMultiRegionStatus_Probe") + .detail("Degraded", degraded) + .detail("DataStateDesc", dataStateDesc) + .detail("AcceptingCommits", acceptingCommits) + .detail("RecoveryStateJson", recoveryStateJson) + .detail("Elapsed", now() - tStart) + .detail("ExpectedState", expectedState); + if (expectedState) { + if (!degradedSince.present()) { + degradedSince = now(); + TraceEvent("DegradedMultiRegionStatus_DegradedHoldStarted") + .detail("Elapsed", now() - tStart); + } + + const double holdDuration = now() - degradedSince.get(); + if (holdDuration >= 10.0) { + TraceEvent("DegradedMultiRegionStatus_Success") + .detail("Elapsed", now() - tStart) + .detail("Degraded", degraded) + .detail("DataStateDesc", dataStateDesc); + printf("\n=== Degraded Multi-Region Status Found ===\n"); + printf( + "Warning: one region is unavailable; committed data remains safe in the surviving " + "region.\n"); + printf( + "Please restart following tlog interfaces, otherwise storage servers may never be " + "able to catch up.\n"); + printf("\nData:\n"); + printf(" Replication health - %s\n", dataStateDesc.c_str()); + printf("========================================\n\n"); + fflush(stdout); + co_return true; + } + } else { + if (degradedSince.present()) { + TraceEvent("DegradedMultiRegionStatus_DegradedHoldLost") + .detail("Elapsed", now() - tStart) + .detail("HoldDuration", now() - degradedSince.get()); + co_return false; + } + } + } + } + } catch (Error& e) { + if (e.code() == error_code_actor_cancelled) { + throw; + } + // Transient read errors: keep polling. + } + + if (now() - tStart > testDuration) { + TraceEvent(SevError, "DegradedMultiRegionStatus_Timeout").detail("Elapsed", now() - tStart); + co_return false; + } + co_await delay(5.0); + } + } + + Future killPrimaryKeepSatellites(Database cx) { + ASSERT(g_network->isSimulated()); + + // The cluster starts fully healthy (both regions up). Confirm it is not + // falsely marked degraded while idle. + testSucceeded = co_await verifyHealthyNotDegraded(cx); + if (!testSucceeded) { + co_return; + } + + // Trigger a normal recovery with no region loss and confirm the flag stays + // false throughout. This guards the transient accepting_commits window. + testSucceeded = co_await verifyNormalRecoveryNotDegraded(cx); + if (!testSucceeded) { + co_return; + } + + // Give the cluster a moment to settle back to fully recovered before the + // destructive kill below. + co_await _setup(cx); + + // The primary region of the first region is datacenter "0" in the fearless + // topology (with primary satellite "2"). Kill only "0"; keep "2", "1" and + // "3" alive so the remote region can recover data from the surviving + // satellite of the (dead) primary region. + LifetimeToken previousMasterLifetime = dbInfo->get().masterLifetime; + g_simulator->killDataCenter("0"_sr, ISimulator::KillType::KillInstantly, true); + TraceEvent("DegradedMultiRegionStatus_KilledPrimaryDC").log(); + + bool failoverReady = true; + + try { + co_await timeoutError(waitForPrimaryDC(cx, "1"_sr), 120.0); + } catch (Error& e) { + if (e.code() == error_code_actor_cancelled) { + throw; + } + TraceEvent(SevError, "DegradedMultiRegionStatus_WaitForPrimaryDCFailed").error(e); + failoverReady = false; + } + + if (failoverReady) { + double start = now(); + LifetimeToken currentMasterLifetime = dbInfo->get().masterLifetime; + while (currentMasterLifetime.isEqual(previousMasterLifetime)) { + if (now() - start > 120.0) { + TraceEvent(SevError, "DegradedMultiRegionStatus_MasterChangeTimeout").log(); + failoverReady = false; + break; + } + co_await dbInfo->onChange(); + currentMasterLifetime = dbInfo->get().masterLifetime; + } + } + + if (failoverReady) { + testSucceeded = co_await waitForDegradedStatus(cx); + } else { + testSucceeded = false; + } + + // Unstick the cluster regardless of the assertion outcome: recovery is parked + // at accepting_commits while region "0" is dead. A forced recovery in region + // "1" runs updateConfigForForcedRecovery, which forces usable_regions=1 and + // lets the cluster fully recover, so subsequent workloads/checks do not hang. + // This is cleanup, not part of the assertion: a cleanup failure is logged but + // does not overwrite the degraded-status result. + try { + co_await forceRecovery(cx->getConnectionRecord(), "1"_sr); + TraceEvent("DegradedMultiRegionStatus_Unstick").log(); + } catch (Error& e) { + if (e.code() == error_code_actor_cancelled) { + throw; + } + TraceEvent(SevWarnAlways, "DegradedMultiRegionStatus_UnstickFailed").error(e); + } + + // Safety net in case the forced recovery did not take effect. + try { + co_await ManagementAPI::changeConfig(cx.getReference(), "usable_regions=1", true); + TraceEvent("DegradedMultiRegionStatus_Reset").log(); + } catch (Error& e) { + if (e.code() == error_code_actor_cancelled) { + throw; + } + TraceEvent(SevWarnAlways, "DegradedMultiRegionStatus_ResetFailed").error(e); + } + } +}; + +WorkloadFactory DegradedMultiRegionStatusWorkloadFactory; diff --git a/fdbserver/workloads/HighContentionPrefixAllocatorWorkload.cpp b/fdbserver/workloads/HighContentionPrefixAllocatorWorkload.cpp index aee3388648c..fbbb3ed3f44 100644 --- a/fdbserver/workloads/HighContentionPrefixAllocatorWorkload.cpp +++ b/fdbserver/workloads/HighContentionPrefixAllocatorWorkload.cpp @@ -54,6 +54,7 @@ struct HighContentionPrefixAllocatorWorkload : TestWorkload { Error err; try { std::vector> futures; + futures.reserve(numAllocations); for (int i = 0; i < numAllocations; ++i) { futures.push_back(allocator.allocate(tr)); } @@ -105,6 +106,7 @@ struct HighContentionPrefixAllocatorWorkload : TestWorkload { for (int roundNum = 0; roundNum < numRounds; ++roundNum) { std::vector> futures; int numTransactions = deterministicRandom()->randomInt(1, maxTransactionsPerRound + 1); + futures.reserve(numTransactions); for (int i = 0; i < numTransactions; ++i) { futures.push_back(runAllocationTransaction(cx)); } diff --git a/fdbserver/workloads/MiniCycle.cpp b/fdbserver/workloads/MiniCycle.cpp index 0a1aa9e4de3..c1c3dc6c1a1 100644 --- a/fdbserver/workloads/MiniCycle.cpp +++ b/fdbserver/workloads/MiniCycle.cpp @@ -74,6 +74,7 @@ struct MiniCycleWorkload : TestWorkload { Future _check(Database cx, MiniCycleWorkload* self) { std::vector> cycleClients; + cycleClients.reserve(self->clientCount > 0 ? self->clientCount : 0); for (int c = 0; c < self->clientCount; c++) { cycleClients.push_back( timeout(self->cycleClient(cx->clone(), self, self->actorCount / self->transactionsPerSecond), @@ -113,6 +114,7 @@ struct MiniCycleWorkload : TestWorkload { Future _checkCycle(Database cx, MiniCycleWorkload* self, bool ok) { std::vector> checkClients; + checkClients.reserve(self->clientCount > 0 ? self->clientCount : 0); for (int c = 0; c < self->clientCount; c++) checkClients.push_back(self->cycleCheckClient(cx->clone(), self, ok)); bool ret = co_await allTrue(checkClients); diff --git a/fdbserver/workloads/NativeCdcEndToEnd.cpp b/fdbserver/workloads/NativeCdcEndToEnd.cpp index 36884595f39..581d7baf0ac 100644 --- a/fdbserver/workloads/NativeCdcEndToEnd.cpp +++ b/fdbserver/workloads/NativeCdcEndToEnd.cpp @@ -69,6 +69,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { int rounds; int assignmentPublicationChecks; bool testProxyReplacement; + bool testTagOwnership; bool injectUndeliveredProxyHalt; bool testMemoryBound; bool testReplyChunking; @@ -567,6 +568,130 @@ class NativeCdcEndToEndWorkload : public TestWorkload { co_await timeoutError(removeNativeCdcStreamClient(cx, name), operationTimeout); } + Future checkTagOwner(Database cx, Tag tag, Optional expected) { + Transaction tr(cx); + while (true) { + Error err; + try { + tr.setOption(FDBTransactionOptions::READ_LOCK_AWARE); + tr.setOption(FDBTransactionOptions::READ_SYSTEM_KEYS); + Optional value = co_await tr.get(cdcTagOwnerKeyFor(tag)); + ASSERT_EQ(value.present(), expected.present()); + if (value.present()) { + ASSERT_EQ(decodeCDCTagOwnerValue(value.get()), expected.get()); + } + co_return; + } catch (Error& e) { + err = e; + } + co_await tr.onError(err); + } + } + + Future overwriteTagOwnerForTesting(Database cx, Tag tag, Optional anchor) { + Transaction tr(cx); + while (true) { + Error err; + try { + tr.setOption(FDBTransactionOptions::LOCK_AWARE); + tr.setOption(FDBTransactionOptions::ACCESS_SYSTEM_KEYS); + if (anchor.present()) { + tr.set(cdcTagOwnerKeyFor(tag), cdcTagOwnerValue(anchor.get())); + } else { + tr.clear(cdcTagOwnerKeyFor(tag)); + } + co_await tr.commit(); + co_return; + } catch (Error& e) { + err = e; + } + co_await tr.onError(err); + } + } + + Future registerThroughProxy(Database cx, + CDCProxyInterface proxy, + Key name, + KeyRange keys, + Optional sharedTagStream = Optional()) { + const CDCRegisterStreamReply reply = co_await timeoutError( + proxy.registerStream.getReply(CDCRegisterStreamRequest(name, keys)), operationTimeout); + co_await timeoutError(waitForAssignedProxy(cx, reply.streamId), operationTimeout); + // Recovery can replace the owner while registration or consumption is in flight. Compare live streams + // in the same published snapshot instead of retaining a proxy ID across those waits. + const auto& assignments = cx->clientInfo->get().streamToCDCProxyId; + const auto owner = assignments.find(reply.streamId); + ASSERT(owner != assignments.end()); + UID expectedOwner = proxy.id(); + if (sharedTagStream.present()) { + const auto sharedOwner = assignments.find(sharedTagStream.get()); + ASSERT(sharedOwner != assignments.end()); + expectedOwner = sharedOwner->second; + } + ASSERT_EQ(owner->second, expectedOwner); + co_return reply.streamId; + } + + Future validateTagOwnership(Database cx) { + ASSERT(streams.empty()); + ASSERT_EQ(cx->clientInfo->get().nativeCdcTagCount, 1); + const Tag tag(tagLocalityCDC, 0); + const Key firstName = "native-cdc-e2e/tag-owner/first"_sr; + const Key secondName = "native-cdc-e2e/tag-owner/second"_sr; + const Key thirdName = "native-cdc-e2e/tag-owner/third"_sr; + const Key key = "native-cdc-e2e/tag-owner/data"_sr; + const KeyRange keys(KeyRangeRef(key, keyAfter(key))); + const CDCStreamId firstId = + co_await timeoutError(registerNativeCdcStreamClient(cx, firstName, keys), operationTimeout); + CDCProxyInterface owner = co_await timeoutError(waitForAssignedProxy(cx, firstId), operationTimeout); + std::vector proxies = cx->clientInfo->get().cdcProxies; + ASSERT_EQ(proxies.size(), 2); + CDCProxyInterface other = proxies[proxies.front().id() == owner.id() ? 1 : 0]; + co_await checkTagOwner(cx, tag, firstId); + + // The public client usually chooses the first proxy, which cannot distinguish an index hit from caller choice. + const CDCStreamId removedId = co_await registerThroughProxy(cx, other, secondName, keys, firstId); + co_await checkTagOwner(cx, tag, firstId); + co_await timeoutError(removeNativeCdcStreamClient(cx, secondName), operationTimeout); + co_await checkTagOwner(cx, tag, firstId); + + for (const Optional invalidAnchor : + { Optional(), Optional(removedId) }) { + co_await overwriteTagOwnerForTesting(cx, tag, invalidAnchor); + co_await registerThroughProxy(cx, other, secondName, keys, firstId); + co_await checkTagOwner(cx, tag, firstId); + co_await timeoutError(removeNativeCdcStreamClient(cx, secondName), operationTimeout); + } + + Reference consumer = + co_await timeoutError(createNativeCdcConsumer(cx, firstName), operationTimeout); + const Value value = "tag-owner-retained-across-replacement"_sr; + const Version committed = co_await writeValue(cx, key, value); + co_await timeoutError(haltProxyUntilReplaced(cx, owner, false), operationTimeout); + owner = co_await timeoutError(waitForAssignedProxy(cx, firstId, owner.id()), operationTimeout); + proxies = cx->clientInfo->get().cdcProxies; + ASSERT_EQ(proxies.size(), 2); + other = proxies[proxies.front().id() == owner.id() ? 1 : 0]; + const CDCStreamId secondId = co_await registerThroughProxy(cx, other, secondName, keys, firstId); + co_await checkTagOwner(cx, tag, firstId); + co_await consumeThroughValue(consumer, committed, key, value); + + co_await timeoutError(removeNativeCdcStreamClient(cx, firstName), operationTimeout); + co_await checkTagOwner(cx, tag, Optional()); + co_await registerThroughProxy(cx, other, thirdName, keys, secondId); + co_await checkTagOwner(cx, tag, secondId); + co_await timeoutError(removeNativeCdcStreamClient(cx, thirdName), operationTimeout); + co_await checkTagOwner(cx, tag, secondId); + co_await timeoutError(removeNativeCdcStreamClient(cx, secondName), operationTimeout); + co_await checkTagOwner(cx, tag, Optional()); + + const CDCStreamId reusedId = co_await registerThroughProxy(cx, other, firstName, keys); + co_await checkTagOwner(cx, tag, reusedId); + co_await timeoutError(removeNativeCdcStreamClient(cx, firstName), operationTimeout); + co_await checkTagOwner(cx, tag, Optional()); + co_await timeoutError(waitForRetiredTagCleanup(cx), operationTimeout); + } + Future setUnpublishedStreamOwner(Database cx, Key name, CDCStreamId streamId, @@ -1636,6 +1761,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { chosenKeys.insert(deterministicRandom()->randomInt(0, keyCount)); } std::vector> values; + values.reserve(chosenKeys.size()); for (int index : chosenKeys) { values.emplace_back(keyForIndex(index), Value(StringRef(format("round/%04d/key/%04d", round, index)))); } @@ -1663,6 +1789,9 @@ class NativeCdcEndToEndWorkload : public TestWorkload { } else { co_await timeoutError(waitForRetiredTagCleanup(cx), operationTimeout); } + if (testTagOwnership) { + co_await timeoutError(validateTagOwnership(cx), operationTimeout); + } } public: @@ -1677,6 +1806,7 @@ class NativeCdcEndToEndWorkload : public TestWorkload { rounds = getOption(options, "rounds"_sr, 30); assignmentPublicationChecks = getOption(options, "assignmentPublicationChecks"_sr, 0); testProxyReplacement = getOption(options, "testProxyReplacement"_sr, false); + testTagOwnership = getOption(options, "testTagOwnership"_sr, false); injectUndeliveredProxyHalt = getOption(options, "injectUndeliveredProxyHalt"_sr, false); testMemoryBound = getOption(options, "testMemoryBound"_sr, false); testReplyChunking = getOption(options, "testReplyChunking"_sr, false); diff --git a/fdbserver/workloads/PhysicalShardMove.cpp b/fdbserver/workloads/PhysicalShardMove.cpp index 9b6f34658ca..c6c978bcdfd 100644 --- a/fdbserver/workloads/PhysicalShardMove.cpp +++ b/fdbserver/workloads/PhysicalShardMove.cpp @@ -260,6 +260,7 @@ struct PhysicalShardMoveWorkLoad : TestWorkload { Error err; try { std::vector>> checkpointEntries; + checkpointEntries.reserve(checkpointIds.size()); for (const UID& id : checkpointIds) { checkpointEntries.push_back(tr.get(checkpointKeyFor(id))); } @@ -380,6 +381,7 @@ struct PhysicalShardMoveWorkLoad : TestWorkload { } std::vector checkpointIds; + checkpointIds.reserve(records.size()); for (const auto& it : records) { checkpointIds.push_back(it.second.checkpointID); } diff --git a/fdbserver/workloads/RandomRangeLock.cpp b/fdbserver/workloads/RandomRangeLock.cpp index 87d16593e29..b060a93cc47 100644 --- a/fdbserver/workloads/RandomRangeLock.cpp +++ b/fdbserver/workloads/RandomRangeLock.cpp @@ -177,6 +177,7 @@ struct RandomRangeLockWorkload : FailureInjectionWorkload { // and this injected workload should not block other workloads. std::string rangeLockOwnerNamePrefix = "Owner" + std::to_string(clientId); std::vector> actors; + actors.reserve(lockActorCount); for (int i = 0; i < lockActorCount; i++) { actors.push_back(lockActor(cx, this, rangeLockOwnerNamePrefix)); } diff --git a/fdbserver/workloads/RangeLock.cpp b/fdbserver/workloads/RangeLock.cpp index 7b747412de0..b9d9abd2bba 100644 --- a/fdbserver/workloads/RangeLock.cpp +++ b/fdbserver/workloads/RangeLock.cpp @@ -21,6 +21,7 @@ #include "fdbclient/AuditUtils.h" #include "fdbclient/RangeLock.h" #include "fdbclient/FDBTypes.h" +#include "fdbclient/KeyRangeMap.h" #include "fdbclient/SystemData.h" #include "fdbserver/core/TesterInterface.h" #include "fdbserver/tester/workloads.h" @@ -230,6 +231,137 @@ struct RangeLocking : TestWorkload { TraceEvent("RangeLockLogicalIdentityPassed"); } + Future testOwnerRemoval(Database cx) { + const RangeLockOwnerName owner = "RangeLockOwnerRemoval"; + const RangeLockOwnerName foreignOwner = "RangeLockOwnerRemovalForeign"; + const std::vector foreignRanges = { + KeyRangeRef("rangeLockOwnerRemoval/a"_sr, "rangeLockOwnerRemoval/b"_sr), + KeyRangeRef("rangeLockOwnerRemoval/c"_sr, "rangeLockOwnerRemoval/d"_sr), + KeyRangeRef("rangeLockOwnerRemoval/e"_sr, "rangeLockOwnerRemoval/f"_sr) + }; + const KeyRange range = KeyRangeRef("rangeLockOwnerRemoval/y"_sr, normalKeys.end); + co_await registerRangeLockOwner(cx, owner, "owner whose locks must survive unregister"); + co_await registerRangeLockOwner(cx, foreignOwner, foreignOwner); + for (const auto& foreignRange : foreignRanges) { + co_await takeExclusiveReadLockOnRange(cx, foreignRange, foreignOwner); + } + co_await takeExclusiveReadLockOnRange(cx, range, owner); + + Transaction scan(cx); + while (true) { + Error err; + try { + scan.setOption(FDBTransactionOptions::READ_LOCK_AWARE); + scan.setOption(FDBTransactionOptions::READ_SYSTEM_KEYS); + const RangeResult firstPage = co_await krmGetRanges(&scan, rangeLockPrefix, normalKeys); + ASSERT(firstPage.more && firstPage.back().key < range.begin); + break; + } catch (Error& e) { + err = e; + } + co_await scan.onError(err); + } + + const auto ownerBefore = co_await getRangeLockOwner(cx, owner); + ASSERT(ownerBefore.present()); + const auto locksBefore = co_await findExclusiveReadLockOnRange(cx, normalKeys); + ASSERT(locksBefore.size() == foreignRanges.size() + 1); + co_await expectOperationRejected(removeRangeLockOwner(cx, owner), error_code_range_lock_reject); + const auto ownerAfter = co_await getRangeLockOwner(cx, owner); + ASSERT(ownerAfter.present() && rangeLockOwnerValue(ownerAfter.get()) == rangeLockOwnerValue(ownerBefore.get())); + const auto locksAfter = co_await findExclusiveReadLockOnRange(cx, normalKeys); + ASSERT(locksAfter == locksBefore); + co_await expectClearRejected(cx, range); + TraceEvent("RangeLockOwnerRemovalPagedScanPassed"); + + co_await releaseExclusiveReadLockOnRange(cx, range, owner); + co_await removeRangeLockOwner(cx, owner); + co_await removeRangeLockOwner(cx, owner); + const auto removedOwner = co_await getRangeLockOwner(cx, owner); + ASSERT(!removedOwner.present()); + const auto remainingLocks = co_await findExclusiveReadLockOnRange(cx, normalKeys); + ASSERT(remainingLocks.size() == foreignRanges.size()); + for (int i = 0; i < foreignRanges.size(); ++i) { + ASSERT(remainingLocks[i].first == foreignRanges[i]); + ASSERT(remainingLocks[i].second == + RangeLockState(RangeLockType::ExclusiveReadLock, foreignOwner, foreignRanges[i])); + } + co_await testOwnerRemovalRace(cx); + co_await releaseExclusiveReadLockByUser(cx, foreignOwner); + co_await removeRangeLockOwner(cx, foreignOwner); + TraceEvent("RangeLockOwnerRemovalPassed"); + } + + Future testOwnerRemovalConflict(Database cx) { + const RangeLockOwnerName owner = "RangeLockOwnerRemovalConflict"; + const KeyRange range = KeyRangeRef("rangeLockOwnerRemovalConflict/a"_sr, "rangeLockOwnerRemovalConflict/b"_sr); + Transaction acquire(cx); + while (true) { + Error err; + try { + co_await registerRangeLockOwner(cx, owner, owner); + co_await takeExclusiveReadLockOnRange(&acquire, range, owner); + co_await removeRangeLockOwner(cx, owner); + break; + } catch (Error& e) { + err = e; + } + co_await acquire.onError(err); + } + const ErrorOr commitResult = co_await errorOr(acquire.commit()); + ASSERT(commitResult.isError()); + co_await acquire.onError(commitResult.getError()); + co_await expectOperationRejected(takeExclusiveReadLockOnRange(cx, range, owner), error_code_range_lock_failed); + const auto removedOwner = co_await getRangeLockOwner(cx, owner); + const auto remainingLocks = co_await findExclusiveReadLockOnRange(cx, range); + ASSERT(!removedOwner.present() && remainingLocks.empty()); + TraceEvent("RangeLockOwnerRemovalConflictPassed").detail("CommitError", commitResult.getError().code()); + } + + Future testOwnerRemovalRace(Database cx) { + const RangeLockOwnerName owner = "RangeLockOwnerRemovalRace"; + const KeyRange range = KeyRangeRef("rangeLockOwnerRemoval/0"_sr, "rangeLockOwnerRemoval/1"_sr); + int acquireWins = 0; + int removeWins = 0; + for (int i = 0; i < 10; ++i) { + co_await registerRangeLockOwner(cx, owner, owner); + Future> taking; + Future> removing; + if (i % 2 == 0) { + taking = errorOr(takeExclusiveReadLockOnRange(cx, range, owner)); + removing = errorOr(removeRangeLockOwner(cx, owner)); + } else { + removing = errorOr(removeRangeLockOwner(cx, owner)); + taking = errorOr(takeExclusiveReadLockOnRange(cx, range, owner)); + } + const ErrorOr takeResult = co_await taking; + const ErrorOr removeResult = co_await removing; + if (takeResult.isError() && takeResult.getError().code() != error_code_range_lock_failed) { + throw takeResult.getError(); + } + if (removeResult.isError() && removeResult.getError().code() != error_code_range_lock_reject) { + throw removeResult.getError(); + } + ASSERT(takeResult.isError() != removeResult.isError()); + const auto registeredOwner = co_await getRangeLockOwner(cx, owner); + const auto locks = co_await findExclusiveReadLockOnRange(cx, range); + if (!takeResult.isError()) { + ++acquireWins; + ASSERT(registeredOwner.present()); + ASSERT(locks.size() == 1 && locks[0].first == range && + locks[0].second == RangeLockState(RangeLockType::ExclusiveReadLock, owner, range)); + co_await releaseExclusiveReadLockOnRange(cx, range, owner); + co_await removeRangeLockOwner(cx, owner); + } else { + ++removeWins; + ASSERT(!registeredOwner.present() && locks.empty()); + } + } + TraceEvent("RangeLockOwnerRemovalRacePassed") + .detail("AcquireWins", acquireWins) + .detail("RemoveWins", removeWins); + } + std::string getLockRangesString(const std::vector>& locks) { std::string res = ""; int count = 0; @@ -425,6 +557,7 @@ struct RangeLocking : TestWorkload { std::vector> res; res = co_await findExclusiveReadLockOnRange(cx, normalKeys); std::vector ranges; + ranges.reserve(res.size()); for (const auto& [lockedRange, _lockState] : res) { ranges.push_back(lockedRange); } @@ -700,6 +833,8 @@ struct RangeLocking : TestWorkload { } co_await testNormalKeyspaceBoundary(cx); co_await testLogicalIdentity(cx); + co_await testOwnerRemoval(cx); + co_await testOwnerRemovalConflict(cx); co_await complexTest(this, cx); co_await testUnlockByUser(this, cx); } diff --git a/fdbserver/workloads/SaveAndKill.cpp b/fdbserver/workloads/SaveAndKill.cpp index 9f1443ba861..b52b9e97efb 100644 --- a/fdbserver/workloads/SaveAndKill.cpp +++ b/fdbserver/workloads/SaveAndKill.cpp @@ -33,9 +33,7 @@ #include "flow/IConnection.h" #include "fdbrpc/SimulatorProcessInfo.h" -#undef state #include "fdbclient/SimpleIni.h" -#define state #undef max #undef min diff --git a/fdbserver/workloads/SkewedReadWrite.cpp b/fdbserver/workloads/SkewedReadWrite.cpp index 16c988f467c..072cbf30604 100644 --- a/fdbserver/workloads/SkewedReadWrite.cpp +++ b/fdbserver/workloads/SkewedReadWrite.cpp @@ -274,6 +274,7 @@ struct SkewedReadWriteWorkload : ReadWriteCommon { std::vector extra_ranges; int reads = aTransaction ? self->readsPerTransactionA : self->readsPerTransactionB; int writes = aTransaction ? self->writesPerTransactionA : self->writesPerTransactionB; + keys.reserve(reads > 0 ? reads : 0); for (int op = 0; op < reads; op++) keys.push_back(self->getRandomKey(self->nodeCount)); diff --git a/fdbserver/workloads/Throttling.cpp b/fdbserver/workloads/Throttling.cpp index 0570c723e39..549a4fc066c 100644 --- a/fdbserver/workloads/Throttling.cpp +++ b/fdbserver/workloads/Throttling.cpp @@ -184,6 +184,7 @@ struct ThrottlingWorkload : KVWorkload { Future start(Database const& cx) override { std::vector> clientActors; + clientActors.reserve(static_cast(actorsPerClient > 0 ? actorsPerClient : 0) + 2); for (int actorId = 0; actorId < actorsPerClient; ++actorId) { clientActors.push_back(timeout(clientActor(cx), testDuration, Void())); } diff --git a/fdbserver/workloads/TransactionCost.cpp b/fdbserver/workloads/TransactionCost.cpp index ae80a018e17..53a90e59ca8 100644 --- a/fdbserver/workloads/TransactionCost.cpp +++ b/fdbserver/workloads/TransactionCost.cpp @@ -207,6 +207,7 @@ class TransactionCostWorkload : public TestWorkload { Future exec(TransactionCostWorkload const& workload, Reference tr) override { std::vector> futures; + futures.reserve(10); for (int i = 0; i < 10; ++i) { futures.push_back(success(tr->get(workload.getKey(testNumber, i)))); } diff --git a/fdbserver/workloads/TxnTimeout.cpp b/fdbserver/workloads/TxnTimeout.cpp index 1ca37c6b4ad..7e934c60492 100644 --- a/fdbserver/workloads/TxnTimeout.cpp +++ b/fdbserver/workloads/TxnTimeout.cpp @@ -142,6 +142,7 @@ struct TxnTimeout : TestWorkload { // Runs database population concurrently across actors and clients Future populateDatabaseAllActors(Database db) { std::vector> populationActors; + populationActors.reserve(actorsPerClient > 0 ? actorsPerClient : 0); for (int actorIdx = 0; actorIdx < actorsPerClient; ++actorIdx) { populationActors.push_back(populateDatabase(db, actorIdx)); } @@ -308,6 +309,7 @@ struct TxnTimeout : TestWorkload { // Phase 2: Run transaction clients that test timeout behavior std::vector> txnClients; + txnClients.reserve(self->actorsPerClient > 0 ? self->actorsPerClient : 0); for (int actorIdx = 0; actorIdx < self->actorsPerClient; ++actorIdx) { txnClients.emplace_back(txnClient(db, actorIdx)); } diff --git a/fdbserver/workloads/UnitTests.cpp b/fdbserver/workloads/UnitTests.cpp index d164878f7f0..1b3a90ee4fa 100644 --- a/fdbserver/workloads/UnitTests.cpp +++ b/fdbserver/workloads/UnitTests.cpp @@ -45,7 +45,6 @@ void forceLinkDDSketchTests(); void forceLinkCommitProxyTests(); void forceLinkWipedStringTests(); void forceLinkRandomKeyValueUtilsTests(); -void forceLinkActorFuzzUnitTests(); void forceLinkGrpcTests(); void forceLinkGrpcTests2(); void forceLinkSimpleCounterTests(); @@ -127,7 +126,6 @@ struct UnitTestWorkload : TestWorkload { forceLinkDDSketchTests(); forceLinkWipedStringTests(); forceLinkRandomKeyValueUtilsTests(); - forceLinkActorFuzzUnitTests(); forceLinkSimpleCounterTests(); forceLinkLogSystemRecoveryTests(); forceLinkIPagerTests(); diff --git a/flow/CMakeLists.txt b/flow/CMakeLists.txt index c838df3af2c..d9a2c5f724f 100644 --- a/flow/CMakeLists.txt +++ b/flow/CMakeLists.txt @@ -165,7 +165,6 @@ foreach(ft flow flow_sampling flowlinktest flow_test) endforeach() if(OPEN_FOR_IDE) - # AcAC requires actor transpiler add_library(acac OBJECT acac.cpp) else() add_executable(acac acac.cpp) @@ -173,9 +172,6 @@ endif() target_link_libraries(acac PUBLIC flow boost_target_program_options) target_compile_definitions(flow_sampling PRIVATE -DENABLE_SAMPLING) -if(WIN32) - add_dependencies(flow_sampling_actors flow_actors) -endif() if(OPEN_FOR_IDE) add_library(mkcert OBJECT MkCertCli.cpp) @@ -253,7 +249,7 @@ if(WITH_SWIFT) target_compile_options( flow_swift PRIVATE - "$<$:SHELL:-Xcc -std=c++20 -Xfrontend -validate-tbd-against-ir=none -Xcc -DNO_INTELLISENSE -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/flow/include/headeroverlay.yaml>" + "$<$:SHELL:-Xcc -std=c++20 -Xfrontend -validate-tbd-against-ir=none -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/flow/include/headeroverlay.yaml>" ) # Ensure that C++ code in fdbserver can import Swift using a compatibility @@ -271,8 +267,6 @@ if(WITH_SWIFT) -Xcc -std=c++20 -Xcc - -DNO_INTELLISENSE - -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/flow/include/headeroverlay.yaml # Important: This is needed to avoid including headers that depends on this # generated header. @@ -291,16 +285,12 @@ if(WITH_SWIFT) -Xcc -std=c++20 -Xcc - -DNO_INTELLISENSE - -Xcc -ivfsoverlay${CMAKE_BINARY_DIR}/flow/include/headeroverlay.yaml) - add_dependencies(flow_swift_checked_continuation_header flow_actors - boost_target ProtocolVersion) + add_dependencies(flow_swift_checked_continuation_header boost_target ProtocolVersion) add_dependencies(flow_swift_header flow_swift_checked_continuation_header) add_dependencies(flow_swift flow_swift_header) - add_dependencies(flow_swift flow_actors) add_dependencies(flow_swift boost_target) # TODO(swift): rdar://99107402 - this will only work once CMake 3.25 is # released: target_link_libraries(flow PRIVATE flow_swift) diff --git a/flow/CoroTests.cpp b/flow/CoroTests.cpp index a72ee752e2a..2a07f680ad8 100644 --- a/flow/CoroTests.cpp +++ b/flow/CoroTests.cpp @@ -21,6 +21,7 @@ #include "flow/UnitTest.h" #include "flow/IAsyncFile.h" #include "flow/FlowThread.h" +#include "flow/PriorityMultiLock.h" #include "flow/Trace.h" #include "flow/TLSConfig.h" @@ -44,6 +45,76 @@ void forceLinkCoroTests() {} using namespace std::literals::string_literals; +Future holdPriorityMultiLock(Reference pml, Future signal) { + Optional lock = pml->tryLock(); + ASSERT(lock.present()); + co_await signal; +} + +TEST_CASE("/flow/coro/PriorityMultiLock/tryLock") { + auto pml = makeReference(1, std::vector{ 1, 1 }); + Optional first = pml->tryLock(0); + ASSERT(first.present()); + ASSERT_EQ(pml->getRunnersCount(), 1); + ASSERT_EQ(pml->getWaitersCount(), 0); + + Optional unavailable = pml->tryLock(1); + ASSERT(!unavailable.present()); + ASSERT_EQ(pml->getRunnersCount(), 1); + ASSERT_EQ(pml->getWaitersCount(), 0); + + Future waiter = pml->lock(1); + ASSERT(!waiter.isReady()); + ASSERT_EQ(pml->getWaitersCount(), 1); + ASSERT(!pml->tryLock(1).present()); + ASSERT_EQ(pml->getWaitersCount(), 1); + first.get().release(); + PriorityMultiLock::Lock second = co_await waiter; + ASSERT_EQ(pml->getRunnersCount(1), 1); + ASSERT_EQ(pml->getWaitersCount(), 0); + second.release(); + ASSERT_EQ(pml->getRunnersCount(), 0); + + auto movePml = makeReference(2, std::vector{ 1 }); + Optional movedFrom = movePml->tryLock(); + PriorityMultiLock::Releaser moved = std::move(movedFrom.get()); + ASSERT(!movedFrom.get().isLocked()); + ASSERT(moved.isLocked()); + Optional replaced = movePml->tryLock(); + ASSERT_EQ(movePml->getRunnersCount(), 2); + moved.release(); + ASSERT_EQ(movePml->getRunnersCount(), 1); + replaced.get().release(); + ASSERT_EQ(movePml->getRunnersCount(), 0); + + auto cancelPml = makeReference(1, std::vector{ 1 }); + Promise never; + Future holder = holdPriorityMultiLock(cancelPml, never.getFuture()); + Future cancelWaiter = cancelPml->lock(); + ASSERT(!cancelWaiter.isReady()); + holder.cancel(); + PriorityMultiLock::Lock afterCancel = co_await cancelWaiter; + ASSERT_EQ(cancelPml->getRunnersCount(), 1); + afterCancel.release(); + ASSERT_EQ(cancelPml->getRunnersCount(), 0); + + auto haltedPml = makeReference(1, std::vector{ 1 }); + Optional halted = haltedPml->tryLock(); + haltedPml->halt(); + halted.reset(); + ASSERT_EQ(haltedPml->getRunnersCount(), 0); + ASSERT(!haltedPml->tryLock().present()); + + pml->kill(); + try { + pml->tryLock(); + ASSERT(false); + } catch (Error& e) { + ASSERT_EQ(e.code(), error_code_broken_promise); + } + co_return; +} + TEST_CASE("/flow/coro/buggifiedDelay") { if (FLOW_KNOBS->MAX_BUGGIFIED_DELAY == 0) { co_return; @@ -128,6 +199,16 @@ Future consumeOneActor(FutureStream in) { co_return i; } +Future consumeTemporaryStream(PromiseStream* in) { + int i = co_await in->getFuture(); + co_return i; +} + +Future consumeReferencedStream(FutureStream* in) { + int i = co_await *in; + co_return i; +} + Future sumActor(FutureStream in) { int total = 0; try { @@ -516,6 +597,7 @@ TEST_CASE("/flow/coro/yieldedFuture/progress") { Future i = success(u); std::vector> v; + v.reserve(5); for (int j = 0; j < 5; j++) v.push_back(yieldedFuture(u)); auto numReady = [&v]() { return std::count_if(v.begin(), v.end(), [](Future v) { return v.isReady(); }); }; @@ -547,6 +629,7 @@ TEST_CASE("/flow/coro/yieldedFuture/random") { Future i = success(u); std::vector> v; + v.reserve(25); for (int j = 0; j < 25; j++) v.push_back(yieldedFuture(u)); auto numReady = [&v]() { @@ -595,6 +678,7 @@ TEST_CASE("/flow/coro/perf/yieldedFuture") { std::vector> ys; start = timer(); + ys.reserve(N); for (int i = 0; i < N; i++) ys.push_back(yieldedFuture(f)); printf("yieldedFuture(f) create: %0.1f M/sec\n", N / 1e6 / (timer() - start)); @@ -1400,6 +1484,65 @@ TEST_CASE("/flow/coro/PromiseStream/move2") { ASSERT(movedTracker.copied == 0); } +TEST_CASE("/flow/coro/FutureStream/rvalueAwait") { + { + PromiseStream stream; + stream.send(41); + Future result = consumeTemporaryStream(&stream); + ASSERT(result.isReady() && !result.isError() && result.get() == 41); + ASSERT(stream.isEmpty()); + } + + { + PromiseStream stream; + Future result = consumeTemporaryStream(&stream); + ASSERT(!result.isReady()); + ASSERT_EQ(stream.getFutureReferenceCount(), 1); + stream.send(42); + ASSERT(result.isReady() && !result.isError() && result.get() == 42); + ASSERT_EQ(stream.getFutureReferenceCount(), 0); + } + + { + PromiseStream stream; + stream.sendError(operation_failed()); + Future result = consumeTemporaryStream(&stream); + ASSERT(result.isReady() && result.isError() && result.getError().code() == error_code_operation_failed); + ASSERT_EQ(stream.getFutureReferenceCount(), 0); + } + + { + PromiseStream stream; + Future result = consumeTemporaryStream(&stream); + stream.sendError(operation_failed()); + ASSERT(result.isReady() && result.isError() && result.getError().code() == error_code_operation_failed); + ASSERT_EQ(stream.getFutureReferenceCount(), 0); + } + + { + PromiseStream stream; + Future result = consumeTemporaryStream(&stream); + ASSERT_EQ(stream.getFutureReferenceCount(), 1); + result.cancel(); + ASSERT(result.isReady() && result.isError() && result.getError().code() == error_code_actor_cancelled); + ASSERT_EQ(stream.getFutureReferenceCount(), 0); + } + + { + PromiseStream stream; + FutureStream input = stream.getFuture(); + Future result = consumeReferencedStream(&input); + ASSERT(!result.isReady()); + ASSERT_EQ(stream.getFutureReferenceCount(), 2); + input = FutureStream(); + ASSERT_EQ(stream.getFutureReferenceCount(), 1); + stream.send(43); + ASSERT(result.isReady() && !result.isError() && result.get() == 43); + ASSERT_EQ(stream.getFutureReferenceCount(), 0); + } + return Void(); +} + TEST_CASE("/flow/coro/AsyncResult/move") { { Tracker tracker = co_await immediateAsyncResultTracker(); @@ -1475,6 +1618,7 @@ TEST_CASE("/flow/coro/quorumAsyncResultReady") { TEST_CASE("/flow/coro/quorumAsyncResultSuccess") { std::vector> signals(3); std::vector> results; + results.reserve(signals.size()); for (int i = 0; i < signals.size(); ++i) { results.push_back(delayedAsyncResultInt(signals[i].getFuture(), i)); } @@ -1531,6 +1675,7 @@ TEST_CASE("/flow/coro/quorumAsyncResultCancelsRemainingProducers") { int completedCount = 0; std::vector> signals(3); std::vector> results; + results.reserve(signals.size()); for (int i = 0; i < signals.size(); ++i) { results.push_back(trackedAsyncResultInt(signals[i].getFuture(), i, &cancelledCount, &completedCount)); } @@ -1552,6 +1697,7 @@ TEST_CASE("/flow/coro/quorumAsyncResultDropCancelsProducers") { std::vector> signals(2); { std::vector> results; + results.reserve(signals.size()); for (int i = 0; i < signals.size(); ++i) { results.push_back(trackedAsyncResultInt(signals[i].getFuture(), i, &cancelledCount, &completedCount)); } @@ -1741,6 +1887,7 @@ TEST_CASE("/flow/coro/getAllAsyncResultDropCancelsProducers") { std::vector> signals(2); { std::vector> results; + results.reserve(signals.size()); for (int i = 0; i < signals.size(); ++i) { results.push_back(trackedAsyncResultInt(signals[i].getFuture(), i, &cancelledCount, &completedCount)); } @@ -1763,6 +1910,7 @@ TEST_CASE("/flow/coro/getAllAsyncResultMoveDropCancelsProducers") { std::vector> signals(2); { std::vector> results; + results.reserve(signals.size()); for (int i = 0; i < signals.size(); ++i) { results.push_back(trackedAsyncResultInt(signals[i].getFuture(), i, &cancelledCount, &completedCount)); } diff --git a/flow/Net2.cpp b/flow/Net2.cpp index 82ed771b6ee..50d96a5d98d 100644 --- a/flow/Net2.cpp +++ b/flow/Net2.cpp @@ -271,6 +271,7 @@ class Net2 final : public INetwork, public INetworkConnections { std::atomic started; uint64_t numYields; + Future readyYield = Void(); NetworkMetrics::PriorityStats* lastPriorityStats; @@ -1903,7 +1904,7 @@ Future Net2::yield(TaskPriority taskID) { return delay(0, taskID); } g_network->setCurrentTask(taskID); - return Void(); + return readyYield; } // TODO: can we wrap our swift task and insert it in here? diff --git a/flow/Platform.cpp b/flow/Platform.cpp index a83c231c9bf..c742f0fbae6 100644 --- a/flow/Platform.cpp +++ b/flow/Platform.cpp @@ -1206,7 +1206,7 @@ void getNetworkTraffic(const IPAddress& ip, struct if_msghdr* ifm = (struct if_msghdr*)next; next += ifm->ifm_msglen; - if ((ifm->ifm_type = RTM_IFINFO2)) { + if (ifm->ifm_type == RTM_IFINFO2) { struct if_msghdr2* if2m = (struct if_msghdr2*)ifm; struct sockaddr_dl* sdl = (struct sockaddr_dl*)(if2m + 1); @@ -1313,23 +1313,25 @@ DiskStatistics getDiskStatistics(std::string const& directory) { DiskStatistics diskStats; - if ((number = (CFNumberRef)CFDictionaryGetValue(stats_dict, CFSTR(kIOBlockStorageDriverStatisticsReadsKey)))) { + number = (CFNumberRef)CFDictionaryGetValue(stats_dict, CFSTR(kIOBlockStorageDriverStatisticsReadsKey)); + if (number) { CFNumberGetValue(number, kCFNumberSInt64Type, &diskStats.reads); } - if ((number = (CFNumberRef)CFDictionaryGetValue(stats_dict, CFSTR(kIOBlockStorageDriverStatisticsWritesKey)))) { + number = (CFNumberRef)CFDictionaryGetValue(stats_dict, CFSTR(kIOBlockStorageDriverStatisticsWritesKey)); + if (number) { CFNumberGetValue(number, kCFNumberSInt64Type, &diskStats.writes); } uint64_t nanoSecs; - if ((number = - (CFNumberRef)CFDictionaryGetValue(stats_dict, CFSTR(kIOBlockStorageDriverStatisticsTotalReadTimeKey)))) { + number = (CFNumberRef)CFDictionaryGetValue(stats_dict, CFSTR(kIOBlockStorageDriverStatisticsTotalReadTimeKey)); + if (number) { CFNumberGetValue(number, kCFNumberSInt64Type, &nanoSecs); diskStats.readMilliSecs += nanoSecs / 1000000; diskStats.IOMilliSecs += nanoSecs / 1000000; } - if ((number = - (CFNumberRef)CFDictionaryGetValue(stats_dict, CFSTR(kIOBlockStorageDriverStatisticsTotalWriteTimeKey)))) { + number = (CFNumberRef)CFDictionaryGetValue(stats_dict, CFSTR(kIOBlockStorageDriverStatisticsTotalWriteTimeKey)); + if (number) { CFNumberGetValue(number, kCFNumberSInt64Type, &nanoSecs); diskStats.writeMilliSecs += nanoSecs / 1000000; diskStats.IOMilliSecs += nanoSecs / 1000000; @@ -3137,8 +3139,8 @@ int setEnvironmentVar(const char* name, const char* value, int overwrite) { #define getcwd(buf, maxlen) _getcwd(buf, maxlen) #endif std::string getWorkingDirectory() { - char* buf; - if ((buf = getcwd(nullptr, 0)) == nullptr) { + char* buf = getcwd(nullptr, 0); + if (buf == nullptr) { TraceEvent(SevWarnAlways, "GetWorkingDirectoryError").GetLastError(); throw platform_error(); } diff --git a/flow/README.md b/flow/README.md index 768c11f1a99..e234eb1c0d7 100644 --- a/flow/README.md +++ b/flow/README.md @@ -2,25 +2,23 @@ Flow Tutorial ============= * [Using Flow](#using-flow) - * [Keywords/primitives](#keywordsprimitives) + * [Primitives](#primitives) * [Promise, Future](#promise-future) - * [Network traversal](#network-traversal) - * [wait()](#wait) - * [ACTOR](#actor) - * [State Variables](#state-variables) + * [Network messaging](#network-messaging) + * [co_await](#co_await) + * [Coroutines](#coroutines) + * [Local variables and lifetimes](#local-variables-and-lifetimes) * [Void](#void) * [PromiseStream<>, FutureStream<>](#promisestream-futurestream) - * [waitNext()](#waitnext) - * [choose / when](#choose--when) + * [Racing futures](#racing-futures) * [Future composition](#future-composition) * [Design Patterns](#design-patterns) - * [RPC](#rpc) - * [ACTOR return values](#actor-return-values) + * [Request/reply](#requestreply) + * [Flatbuffers/ObjectSerializer](#flatbuffersobjectserializer) + * [Coroutine return values](#coroutine-return-values) * [“gotchas”](#gotchas) - * [Actor compiler](#actor-compiler) - * [Switch statements](#switch-statements) - * [try/catch with no wait()](#trycatch-with-no-wait) - * [ACTOR cancellation](#actor-cancellation) + * [Exception handling](#exception-handling) + * [Coroutine cancellation](#coroutine-cancellation) * [Memory Management](#memory-management) * [Reference Counting](#reference-counting) * [Potential Gotchas](#potential-gotchas) @@ -29,26 +27,28 @@ Flow Tutorial * [Potential Gotchas](#potential-gotchas-1) * [Function Creating and Returning a non-Standalone Ref Object](#function-creating-and-returning-a-non-standalone-ref-object) * [Assigning Returned Standalone Object to non Standalone Variable](#assigning-returned-standalone-object-to-non-standalone-variable) - * [Use of Standalone Objects in ACTOR Functions](#use-of-standalone-objects-in-actor-functions) + * [Use of Standalone Objects in Coroutines](#use-of-standalone-objects-in-coroutines) # Using Flow -Flow introduces some new keywords and flow controls. Combining these into workable units -also introduces some new design patterns to C++ programmers. +Flow provides asynchronous communication and cooperative scheduling using standard C++ +coroutines. Include `flow/flow.h` for the runtime types and `flow/CoroUtils.h` for +`race()`. Coroutine code lives in ordinary `.cpp` and `.h` files. -## Keywords/primitives +See [the coroutine design guide](../design/coroutines.md) for more details and +[the runnable coroutine tutorial](../documentation/coro_tutorial/tutorial.cpp) for +examples that use the network and FoundationDB client APIs. + +## Primitives The essence of Flow is the capability of passing messages asynchronously between components. The basic data types that connect asynchronous senders and receivers are `Promise<>` and `Future<>`. The sender holds a `Promise` to, sometime in the future, deliver a value of type `X` to the holder of the `Future`. A receiver, holding a `Future`, at some point -needs the `X` to continue computation, and invokes the `wait(Future<> f)` statement to pause -until the value is delivered. To use the `wait()` statement, a function needs to be declared as an -ACTOR function, a special flow keyword which directs the flow compiler to create the necessary -internal callbacks, etc. Similarly, When a component wants to deal not with one asynchronously -delivered value, but with a series, there are `PromiseStream<>` and `FutureStream<>`. These -two constructs allow for “reliable delivery” of messages, and play an important role in the design -patterns. +needs the `X` to continue computation, and uses `co_await` to suspend until the value is +delivered. Other coroutines can run while it is suspended. When a component wants to deal +with a series of asynchronously delivered values, it uses `PromiseStream<>` and +`FutureStream<>`. ### Promise, Future @@ -66,230 +66,205 @@ p.send( 4 ); printf( "%d\n", f.get() ); // f is already set ``` -### Network traversal - -Promises and futures can be used within a single process, but their real strength in a distributed -system is that they can traverse the network. For example, one computer could create a -promise/future pair, then send the promise to another computer over the network. The promise -and future will still be connected, and when the promise is fulfilled by the remote computer, the -original holder of the future will see the value appear. +### Network messaging -[TODO: network delivery guarantees] +`Promise` and `Future` are local process handles. FoundationDB's RPC layer builds +network communication on top of these primitives with `RequestStream` and +`ReplyPromise`. A request can carry a reply promise to another process; sending the reply +there makes the caller's future ready. See the [request/reply example](#requestreply) below +for the local form of this pattern. -### wait() +### co_await -Wait allows for the calling code to pause execution while the value of a `Future` is set. This -statement is called with a `Future` as its parameter and returns a `T`; the eventual value of the -`Future`. Errors that are generated in the code that is setting the `Future`, will be thrown from -the location of the `wait()`, so `Error`s must be caught at this location. +`co_await` suspends a coroutine until a future becomes ready. If the future is already ready, +execution continues without suspending. Awaiting a failed future throws its `Error` at the +`co_await` expression. -The following example shows a snippet (from an ACTOR) of waiting on a `Future`: +The following snippet waits on a `Future` inside a coroutine: ```c++ Future f = asyncCalculation(); // defined elsewhere -int count = wait( f ); +int count = co_await f; printf( "%d\n", count ); ``` -It is worth noting that, although the function `wait()` is declared in [actorcompiler.h](include/flow/actorcompiler.h), this -“function” is compiled by the Actor Compiler into a complex set of integrated statements and -callbacks. It is therefore never present in generated code or at link time. -**Note** : because of the way that the actor compiler is built, `wait()` must always assign the -resulting value to a _newly declared variable._ - -From 6.1, `wait()` on `Void` actors shouldn't assign the resulting value. So, the following code +Await a `Future` without assigning the result: ```c++ Future asyncTask(); //defined elsewhere -Void _ = _wait(asyncTask()); +co_await asyncTask(); ``` -becomes - -```c++ -Future asyncTask(); //defined elsewhere -wait(asyncTask()); -``` +### Coroutines -### ACTOR +A function containing `co_await` or `co_return` is a coroutine. A Flow coroutine returning +`Future` produces its result with `co_return value;`. The C++ compiler preserves its +execution state across suspension points. -The only code that can call the `wait()` function are functions that are themselves labeled with -the “ACTOR” tag. This is the essential unit of asynchronous work that can be chained together -to create complex message-passing systems. -An actor, although declared as returning a `Future`, simply returns a `T`. Because an actor -may wait on the results of other actors, an actor must return either a `Future `or `void`. In most -cases returning `void `is less advantageous than returning a `Future`, since there are -implications for actor cancellation. See the Actor Cancellation section for details. +Calling a coroutine that returns `Future` starts it immediately. It runs until it completes +or awaits something that is not ready. Retain the returned future for work that must continue; see +[coroutine cancellation](#coroutine-cancellation). -The following simple actor function waits on the `Future` to be ready, when it is ready adds `offset` and returns the result: +The following function waits for a value, adds `offset`, and returns the result: ```c++ -ACTOR Future asyncAdd(Future f, int offset) { - int value = wait( f ); - return value + offset; +Future asyncAdd(Future f, int offset) { + int value = co_await f; + co_return value + offset; } ``` -### State Variables +### Local variables and lifetimes -Since ACTOR-labeled functions are compiled into a c++ class and numerous supporting -functions, the variable scoping rules that normally apply are altered. The differences arise out -of the fact that control flow is broken at wait() statements. Generally the compiled code is -broken into chunks at wait statements, so scoping variables so that they can be seen in multiple -“chunks” requires the `state `keyword. -The following function waits on two inputs and outputs the sum with an offset attached: +Local variables follow normal C++ scope rules and remain alive across suspension while +their scope is active. The following function retains `value1` while waiting for `f2`: ```c++ -ACTOR Future asyncCalculation(Future f1, Future f2, int offset ) { - state int value1 = wait( f1 ); - int value2 = wait( f2 ); - return value1 + value2 + offset; +Future asyncCalculation(Future f1, Future f2, int offset) { + int value1 = co_await f1; + int value2 = co_await f2; + co_return value1 + value2 + offset; } ``` +Parameters passed by value are stored in the coroutine frame. A reference, pointer, or +non-owning view does not keep the referenced object alive; its owner must outlive all uses, +including uses after suspension. Prefer owning parameters such as `Reference` or +`Standalone` when a coroutine must retain an object or its bytes. + +Captures in a coroutine lambda belong to the lambda's closure, which may be destroyed while +the coroutine is suspended. Prefer a named coroutine with explicit value parameters when +the work can outlive the call that starts it. + ### Void The `Void `type is used as a signalling-only type for coordination of asynchronous processes. The following function waits on an input, sends an output to a `Promise`, and signals completion: ```c++ -ACTOR Future asyncCalculation(Future f, Promise p, int offset ) { - int value = wait( f ); +Future asyncCalculation(Future f, Promise p, int offset) { + int value = co_await f; p.send( value + offset ); - return Void(); + co_return; } ``` ### PromiseStream<>, FutureStream<> -PromiseStream ​and `FutureStream` are groupings of a series of asynchronous messages. - - -These allow for two important features: multiplexing and network reliability, discussed later. -They can be waited on with the `waitNext()` function. - -### waitNext() - -Like `wait()`, `waitNext()` pauses program execution and awaits the next value in a -`FutureStream`. If there is a value ready in the stream, execution continues without delay. The -following “server” waits on input, sends an output to a `PromiseStream`: +`PromiseStream` sends a series of values, and `FutureStream` receives them. Await a +`FutureStream` directly to consume its next value. If a value is already queued, execution +continues without suspension. The following server waits for input and sends the result to a +`PromiseStream`: ```c++ -ACTOR void asyncCalculation(FutureStream f, PromiseStream p, int offset ) { - while( true ) { - int value = waitNext( f ); +Future asyncCalculation(FutureStream f, PromiseStream p, int offset) { + while (true) { + int value = co_await f; p.send( value + offset ); } } ``` -### choose / when +### Racing futures -The `choose / when` construct allows an Actor to wait for multiple `Future `events at once in a -ordered and predictable way. Only the `when` associated with the first future to become ready -will be executed. The following shows the general use of choose and when: +`race()` waits for the first ready input and returns a `std::variant` whose index identifies +the winning argument. Inputs can be futures or streams; a winning stream consumes one +element. If multiple inputs are already ready, the lowest argument index wins. ```c++ -choose { - when( int number = waitNext( futureStreamA ) ) { - // clause A - } - when( std::string text = wait( futureB ) ) { - // clause B - } +auto result = co_await race(futureStreamA, futureB); +if (result.index() == 0) { + int number = std::get<0>(result); + // Handle the stream value. +} else { + std::string text = std::get<1>(result); + // Handle the future value. } ``` -You can put this construct in a loop if you need multiple `when` clauses to execute. +Errors propagate from the winning input. Losing inputs are detached from the race, not +explicitly cancelled. They can still be cancelled if releasing the race drops their last future +reference. Retain a future separately when its operation must continue after losing a race. + +Put the race in a loop to process a sequence of events. A completed non-stream future remains +ready, so replace it or remove it from the race after handling it. ### Future composition Futures can be chained together with the result of one depending on the output of another. ```c++ -ACTOR Future asyncAddition(Future f, int offset ) { - int value = wait( f ); - return value + offset; +Future asyncAddition(Future f, int offset) { + int value = co_await f; + co_return value + offset; } -ACTOR Future asyncDivision(Future f, int divisor ) { - int value = wait( f ); - return value / divisor; +Future asyncDivision(Future f, int divisor) { + int value = co_await f; + co_return value / divisor; } -ACTOR Future asyncCalculation( Future f ) { - int value = wait( asyncDivision( - asyncAddition( f, 10 ), 2 ) ); - return value; +Future asyncCalculation(Future f) { + co_return co_await asyncDivision(asyncAddition(f, 10), 2); } ``` ## Design Patterns -### RPC +### Request/reply -Many of the “servers” in FoundationDB that communicate over the network expose their interfaces as a struct of PromiseStreams--one for each request type. For instance, a logical server that keeps a count could look like this: +Many logical servers expose one request stream per request type. This local example uses +promise streams to maintain a count. A network interface uses the RPC types described +[above](#network-messaging) and also needs serialization. ```c++ struct CountingServerInterface { PromiseStream addCount; PromiseStream subtractCount; PromiseStream> getCount; - - // serialization code required for use on a network - template - void serialize( Ar& ar ) { - serializer(ar, addCount, subtractCount, getCount); - } }; ``` Clients can then pass messages to the server with calls such as this: ```c++ -CountingServerInterface csi = ...; // comes from somewhere -csi.addCount.send(5); -csi.subtractCount.send(2); -Promise finalCount; -csi.getCount.send(finalCount); -int value = wait( finalCount.getFuture() ); -``` - -There is even a utility function to take the place of the last three lines: [TODO: And is necessary -when sending requests over a real network to ensure delivery] - -```c++ -CountingServerInterface csi = ...; // comes from somewhere -csi.addCount.send(5); -csi.subtractCount.send(2); -int value = wait( csi.getCount.getReply() ); +Future updateAndReadCount(CountingServerInterface csi) { + csi.addCount.send(5); + csi.subtractCount.send(2); + Promise finalCount; + csi.getCount.send(finalCount); + co_return co_await finalCount.getFuture(); +} ``` -Canonically, a single server ACTOR that implements the interface is a loop with a choose -statement between all of the request types: +A single server coroutine handles requests by repeatedly racing the request streams: ```c++ -ACTOR void serveCountingServerInterface(CountingServerInterface csi) { - state int count = 0; - loop { - choose { - when (int x = waitNext(csi.addCount.getFuture())){ - count += x; - } - when (int x = waitNext(csi.subtractCount.getFuture())){ - count -= x; - } - when (Promise r = waitNext(csi.getCount.getFuture())){ - r.send( count ); // goes to client - } +Future serveCountingServerInterface(CountingServerInterface csi) { + int count = 0; + while (true) { + auto request = co_await race(csi.addCount.getFuture(), + csi.subtractCount.getFuture(), + csi.getCount.getFuture()); + switch (request.index()) { + case 0: + count += std::get<0>(request); + break; + case 1: + count -= std::get<1>(request); + break; + case 2: + std::get<2>(request).send(count); + break; } } } ``` -In this example, the add and subtract interfaces modify the count itself, stored with a state -variable. The get interface is a bit more complicated, taking a `Promise` instead of just an +The caller must keep the server's returned `Future` alive while the server is needed. +The add and subtract interfaces modify the count, which remains alive across each suspension. +The get interface takes a `Promise` instead of just an int. In the interface class, you can see a `PromiseStream>`. This is a common construct that is analogous to sending someone a self-addressed envelope. You send a promise to a someone else, who then unpacks it and send the answer back to you, because @@ -411,48 +386,42 @@ you are holding the corresponding future. template or something similar so that we can write smaller messages for deprecated fields. -### ACTOR return values +### Coroutine return values -An actor can have only one returned Future, so there is a case that one actor wants to perform -some operation more than once: +A coroutine's returned future completes only once. Use a promise stream to send repeated +results while the coroutine is running: ```c++ -ACTOR Future periodically(PromiseStream ps, int seconds) { - loop { - wait( delay( seconds ) ); +Future periodically(PromiseStream ps, int seconds) { + while (true) { + co_await delay(seconds); ps.send(Void()); } } ``` -In this example, the `PromiseStream `is actually a way for the actor to return data from some -operation that it ongoing. - -By default it is a compiler error to discard the result of a cancellable actor. If you don't think this is appropriate for your actor you can use the `[[flow_allow_discard]]` attribute. -This does not apply to UNCANCELLABLE actors. +Keep the returned `Future` alive for as long as periodic notifications are needed. +Its lifetime controls the work; the stream carries the notifications. ## “gotchas” -### Actor compiler - -There are some things about the actor compiler that can confuse and may change over time - -#### Switch statements +### Exception handling -Do not use these with wait statements inside! +An error from an awaited future is thrown at the `co_await` expression. Catch errors around +the operation that can fail, and propagate errors that the coroutine cannot handle. C++ does +not allow `co_await` inside a `catch` handler. If recovery itself is asynchronous, save the +error and await the recovery operation after leaving the handler. -#### try/catch with no wait() +### Coroutine cancellation -When a `try/catch` block does not `wait()` the blocks are still decomposed into separate -functions. This means that variables that you want to access both before and after such a block -will need to be declared state. +By default, dropping the last reference to a pending coroutine's returned `Future` cancels +that coroutine. An explicit `Future::cancel()` also requests cancellation. A suspended +coroutine resumes by throwing `actor_cancelled` from its await; local objects are destroyed +as their scopes unwind. Do not discard a future when its work must continue. -### ACTOR cancellation - -When the reference to the returned `Future` of an actor is dropped, that actor will be cancelled. -Cancellation of an actor means that any `wait()`s that were currently active (the callback was -currently registered) will be delivered an exception (`actor_cancelled`). In almost every case -this exception should not be caught, though there are certainly exceptions! +Do not swallow `actor_cancelled` in an error or retry handler. Rethrow it after any required +synchronous cleanup so that cancellation can finish. Preserve `broken_promise` and other +errors unless the caller's contract explicitly handles them. # Memory Management @@ -597,27 +566,26 @@ returned from `foo`. When this returned `StringRef` is subsequently deallocated, longer be valid. -#### Use of Standalone Objects in ACTOR Functions +#### Use of Standalone Objects in Coroutines -Special care needs to be taken when using using `Standalone` values in actor functions. -Consider the following example: +An owning local remains alive across suspension while its scope is active. A non-owning +`StringRef` still does not retain its arena. When a coroutine needs to own bytes independently +of its caller, pass a `Standalone` by value: -``` -ACTOR Future foo(StringRef param) -{ - //Do something - return Void(); +```c++ +Future printLater(Standalone text) { + co_await delay(1.0); + printf("%s\n", text.toString().c_str()); + co_return; } -ACTOR Future bar() -{ - Standalone str("string"); - wait(foo(str)); - return Void(); +Future printMessage() { + Standalone text("string"_sr); + co_await printLater(text); + co_return; } ``` -Although it appears at first glance that `bar` keeps the `Arena` for `str` alive during the call to `foo`, -it will actually go out of scope in the class generated by the actor compiler. As a result, `param` in -`foo` will become invalid. To prevent this, either declare `param` to be of type -`Standalone` or make `str` a state variable. +Both coroutines retain the arena in this example. Passing a `StringRef` instead would be safe +only if its owner remained alive until the callee finished using it. The same rule applies to +`KeyRef`, `ValueRef`, and other views into arena-backed storage. diff --git a/flow/SimpleCounter.cpp b/flow/SimpleCounter.cpp index 962036163d6..d63c835c381 100644 --- a/flow/SimpleCounter.cpp +++ b/flow/SimpleCounter.cpp @@ -144,6 +144,7 @@ TEST_CASE("/flow/simplecounter/int64") { }; std::vector threads; + threads.reserve(10); for (int i = 0; i < 10; i++) { threads.emplace_back(inclots); } @@ -184,6 +185,7 @@ TEST_CASE("/flow/simplecounter/double") { }; std::vector threads; + threads.reserve(10); for (int i = 0; i < 10; i++) { threads.emplace_back(double_inclots); } diff --git a/flow/Trace.cpp b/flow/Trace.cpp index a5b62ffc07c..51c8b754a9c 100644 --- a/flow/Trace.cpp +++ b/flow/Trace.cpp @@ -503,6 +503,14 @@ struct TraceLog { } } + // Tracked-latest events first logged before the trace file was opened (e.g. an early + // ProgramStart in backup_agent) were cached without the universal annotation fields + // (LogGroup, Machine, Roles). Now that the log is open, annotate the rolled copy so it + // carries them. Already-annotated cached copies already have those fields (copied + // above), so skip them to avoid duplicating fields. + if (!events[idx].isAnnotated()) + annotateEvent(rolledFields); + eventBuffer.push_back(rolledFields); } } diff --git a/flow/actorcompiler/Actor checklist.txt b/flow/actorcompiler/Actor checklist.txt deleted file mode 100644 index 82ed087cefc..00000000000 --- a/flow/actorcompiler/Actor checklist.txt +++ /dev/null @@ -1,84 +0,0 @@ -Compile issues: - -- wait() must always assign the resulting value to a newly declared variable. - -- Variables used across a "wait() boundary" must be declared state - - -Remember to: - -- Add useful ASSERT()s - -- Add BUGGIFY() statements to expose rare cases to simulation - -- Add TEST() statements to any conditional/rare cases - -- Comment invariants, strategy, preconditions, tricky stuff when you - figure it out, even if it's not "your" code. - -- Factor common asynchronous control flows to use composition - of generic actors such as: - waitForAll - timeout - splitFuture - recurring - smartQuorum - AsyncMap - broadcast - &&, || - etc... - -- Declare classes NonCopyable unless they are, and you know what - that means - - -Run time issues: - -- Is the actor return type "void"? Make sure that some exception or timeout - will trigger eventually to clean up the actor. - -- If you send a future to another future, a long-lived forwardPromise - actor is created--make sure that the event happens eventually to free - this actor. - -- If you return a Future instead of a T from an actor, a forwardPromise - actor is created with the same lifetime issues as above. - -- Remember that parameters are internally passed to the actor as const & - and then copied into actor state variables - -- When you use *GetReply() or LoadBalance(), the "server" responding to - your request might get your request multiple times. - -- When you use getReply() instead of tryGetReply() you must ensure that the - actor will be cancelled if the service you are trying to connect to is - no longer available. (Otherwise, an infinite waiting loop) - -- For each wait: - - - An actor_cancelled exception can be thrown if the actor's return - value future is dropped. - - - An exception can arrive instead of a value - - - What happens if it never returns? - - - If the client fulfilling the wait is coming over the network, you - might get the same request multiple times - - -Performance issues: - -- Wait a little extra time before doing something time-consuming or - irreversible to see if it is still necessary. - -- When waiting for a number of things, wait a little extra time to get - the stragglers. (See the SmartQuorum() generic actor) - -- If asking another asynchronous server to do units of work, don't queue up more - work than is necessary to keep the server busy. Likewise, if you are - busy, let your own work queue fill up to signal your requester - that you are blocked. Also do this personally with managers assigning - you stuff. - -- Pass all variables as "const &" if their size is greater than 8 bytes. diff --git a/flow/actorcompiler/ActorCompiler.cs b/flow/actorcompiler/ActorCompiler.cs deleted file mode 100644 index adf138eb27b..00000000000 --- a/flow/actorcompiler/ActorCompiler.cs +++ /dev/null @@ -1,1409 +0,0 @@ -/* - * ActorCompiler.cs - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -using System; -using System.Collections.Generic; -using System.Linq; -using System.Text; -using System.IO; -using System.Security.Cryptography; - -namespace actorcompiler -{ - class TypeSwitch - { - object value; - R result; - bool ok = false; - public TypeSwitch(object t) { this.value = t; } - public TypeSwitch Case(Func f) - where T : class - { - if (!ok) - { - var t = value as T; - if (t != null) - { - result = f(t); - ok = true; - } - } - return this; - } - public R Return() - { - if (!ok) throw new Exception("Typeswitch didn't match."); - return result; - } - } - - class Context - { - public Function target; - public Function next; - public Function breakF; - public Function continueF; - public Function catchFErr; // Catch function taking Error - public int tryLoopDepth = 0; // The number of (loopDepth-increasing) loops entered inside the innermost try (thus, that will be exited by a throw) - - public void unreachable() { target = null; } - - // Usually we just change the target of a context - public Context WithTarget(Function newTarget) { return new Context { target = newTarget, breakF = breakF, continueF = continueF, next = null, catchFErr = catchFErr, tryLoopDepth = tryLoopDepth }; } - - // When entering a loop, we have to provide new break and continue functions - public Context LoopContext(Function newTarget, Function breakF, Function continueF, int deltaLoopDepth) { return new Context { target = newTarget, breakF = breakF, continueF = continueF, next = null, catchFErr = catchFErr, tryLoopDepth = tryLoopDepth + deltaLoopDepth }; } - - public Context WithCatch(Function newCatchFErr) { return new Context { target = target, breakF = breakF, continueF = continueF, next = null, catchFErr = newCatchFErr, tryLoopDepth = 0 }; } - - public Context Clone() { return new Context { target = target, next = next, breakF = breakF, continueF = continueF, catchFErr = catchFErr, tryLoopDepth = tryLoopDepth }; } - }; - - class Function - { - public string name; - public string returnType; - public string[] formalParameters; - public bool endIsUnreachable = false; - public string exceptionParameterIs = null; - public bool publicName = false; - public string specifiers; - string indentation; - StreamWriter body; - public bool wasCalled { get; protected set; } - public Function overload = null; - - public Function() - { - body = new StreamWriter(new MemoryStream()); - } - - public void setOverload(Function overload) { - this.overload = overload; - } - - public Function popOverload() { - Function result = this.overload; - this.overload = null; - return result; - } - - public void addOverload(params string[] formalParameters) { - setOverload( - new Function { - name = name, - returnType = returnType, - endIsUnreachable = endIsUnreachable, - formalParameters = formalParameters, - indentation = indentation - } - ); - } - - public void Indent(int change) - { - for(int i=0; i0, run the actor beginning at point P until - * (1) it waits, in which case set an appropriate callback and return 0, or - * (2) it returns, in which case destroy the actor and return 0, or - * (3) it reaches the bottom of the Nth innermost loop containing P, in which case - * return max(0, the given loopDepth - N) (N=0 for the innermost loop, N=1 for the next innermost, etc) - * - * Examples: - * Source: - * loop - * [P] - * loop - * [P'] - * loop - * [P''] - * break - * [Q'] - * break - * [Q] - * - * fP(1) should execute everything from [P] to [Q] and then return 1 (since [Q] is at the bottom of the 0th innermost loop containing [P]) - * fP'(2) should execute everything from [P'] to [Q] and then return 1 (since [Q] is at the bottom of the 1st innermost loop containing [P']) - * fP''(3) should execute everything from [P''] to [Q] and then return 1 (since [Q] is at the bottom of the 2nd innermost loop containing [P'']) - * fQ'(2) should execute everything from [Q'] to [Q] and then return 1 (since [Q] is at the bottom of the 1st innermost loop containing [Q']) - * fQ(1) should return 1 (since [Q] is at the bottom of the 0th innermost loop containing [Q]) - */ - }; - - class LiteralBreak : Function - { - public LiteralBreak() { name = "break!"; } - public override string call(params string[] parameters) - { - wasCalled = true; - if (parameters.Length != 0) throw new Exception("LiteralBreak called with parameters!"); - return "break"; - } - }; - - class LiteralContinue : Function - { - public LiteralContinue() { name = "continue!"; } - public override string call(params string[] parameters) - { - wasCalled = true; - if (parameters.Length != 0) throw new Exception("LiteralContinue called with parameters!"); - return "continue"; - } - }; - - class StateVar : VarDeclaration - { - public int SourceLine; - }; - - class CallbackVar : StateVar - { - public int CallbackGroup; - } - - class ActorCompiler - { - Actor actor; - string className, fullClassName, stateClassName; - string sourceFile; - List state; - List callbacks = new List(); - bool isTopLevel; - const string loopDepth0 = "int loopDepth=0"; - const string loopDepth = "int loopDepth"; - const int codeIndent = +2; - const string memberIndentStr = "\t"; - static HashSet usedClassNames = new HashSet(); - bool LineNumbersEnabled; - int chooseGroups = 0, whenCount = 0; - string This; - bool generateProbes; - public Dictionary<(ulong, ulong), string> uidObjects { get; private set; } - - public ActorCompiler(Actor actor, string sourceFile, bool isTopLevel, bool lineNumbersEnabled, bool generateProbes) - { - this.actor = actor; - this.sourceFile = sourceFile; - this.isTopLevel = isTopLevel; - this.LineNumbersEnabled = lineNumbersEnabled; - this.generateProbes = generateProbes; - this.uidObjects = new Dictionary<(ulong, ulong), string>(); - - FindState(); - } - - private ulong ByteToLong(byte[] bytes) { - // NOTE: Always assume big endian. - ulong result = 0; - foreach(var b in bytes) { - result += b; - result <<= 8; - } - return result; - } - - // Generates the identifier for the ACTOR - private Tuple GetUidFromString(string str) { - byte[] sha256Hash = SHA256.Create().ComputeHash(Encoding.UTF8.GetBytes(str)); - byte[] first = sha256Hash.Take(8).ToArray(); - byte[] second = sha256Hash.Skip(8).Take(8).ToArray(); - return new Tuple(ByteToLong(first), ByteToLong(second)); - } - - // Writes the function that returns the Actor object - private void WriteActorFunction(TextWriter writer, string fullReturnType) { - WriteTemplate(writer); - LineNumber(writer, actor.SourceLine); - foreach (string attribute in actor.attributes) { - writer.Write(attribute + " "); - } - if (actor.isStatic) writer.Write("static "); - writer.WriteLine("{0} {3}{1}( {2} ) {{", fullReturnType, actor.name, string.Join(", ", ParameterList()), actor.nameSpace==null ? "" : actor.nameSpace + "::"); - LineNumber(writer, actor.SourceLine); - - string newActor = string.Format("new {0}({1})", - fullClassName, - string.Join(", ", actor.parameters.Select(p => p.name).ToArray())); - - if (actor.returnType != null) - writer.WriteLine("\treturn Future<{1}>({0});", newActor, actor.returnType); - else - writer.WriteLine("\t{0};", newActor); - writer.WriteLine("}"); - } - - // Writes the class of the Actor object - private void WriteActorClass(TextWriter writer, string fullStateClassName, Function body) { - // The final actor class mixes in the State class, the Actor base class and all callback classes - writer.WriteLine("// This generated class is to be used only via {0}()", actor.name); - WriteTemplate(writer); - LineNumber(writer, actor.SourceLine); - - string callback_base_classes = string.Join(", ", callbacks.Select(c=>string.Format("public {0}", c.type))); - if (callback_base_classes != "") callback_base_classes += ", "; - writer.WriteLine("class {0} final : public Actor<{2}>, {3}public FastAllocated<{1}>, public {4} {{", - className, - fullClassName, - actor.returnType == null ? "void" : actor.returnType, - callback_base_classes, - fullStateClassName - ); - writer.WriteLine("public:"); - writer.WriteLine("\tusing FastAllocated<{0}>::operator new;", fullClassName); - writer.WriteLine("\tusing FastAllocated<{0}>::operator delete;", fullClassName); - - var actorIdentifierKey = this.sourceFile + ":" + this.actor.name; - var actorIdentifier = GetUidFromString(actorIdentifierKey); - uidObjects.Add((actorIdentifier.Item1, actorIdentifier.Item2), actorIdentifierKey); - // NOTE UL is required as a u64 postfix for large integers, otherwise Clang would complain - writer.WriteLine("\tstatic constexpr ActorIdentifier __actorIdentifier = UID({0}UL, {1}UL);", actorIdentifier.Item1, actorIdentifier.Item2); - writer.WriteLine("\tActiveActorHelper activeActorHelper;"); - - writer.WriteLine("#pragma clang diagnostic push"); - writer.WriteLine("#pragma clang diagnostic ignored \"-Wdelete-non-virtual-dtor\""); - if (actor.returnType != null) - writer.WriteLine(@" void destroy() override {{ - activeActorHelper.~ActiveActorHelper(); - static_cast*>(this)->~Actor(); - operator delete(this); - }}", actor.returnType); - else - writer.WriteLine(@" void destroy() {{ - activeActorHelper.~ActiveActorHelper(); - static_cast*>(this)->~Actor(); - operator delete(this); - }}"); - writer.WriteLine("#pragma clang diagnostic pop"); - - foreach (var cb in callbacks) - writer.WriteLine("friend struct {0};", cb.type); - - LineNumber(writer, actor.SourceLine); - WriteConstructor(body, writer, fullStateClassName); - WriteCancelFunc(writer); - writer.WriteLine("};"); - - } - - public void Write(TextWriter writer) - { - string fullReturnType = - actor.returnType != null ? string.Format("Future<{0}>", actor.returnType) - : "void"; - for (int i = 0; ; i++) - { - className = string.Format("{3}{0}{1}Actor{2}", - actor.name.Substring(0, 1).ToUpper(), - actor.name.Substring(1), - i != 0 ? i.ToString() : "", - actor.enclosingClass != null && actor.isForwardDeclaration ? actor.enclosingClass.Replace("::", "_") + "_" - : actor.nameSpace != null ? actor.nameSpace.Replace("::", "_") + "_" - : ""); - if (actor.isForwardDeclaration || usedClassNames.Add(className)) - break; - } - - // e.g. SimpleTimerActor - fullClassName = className + GetTemplateActuals(); - var actorClassFormal = new VarDeclaration { name = className, type = "class" }; - This = string.Format("static_cast<{0}*>(this)", actorClassFormal.name); - // e.g. SimpleTimerActorState - stateClassName = className + "State"; - // e.g. SimpleTimerActorState - var fullStateClassName = stateClassName + GetTemplateActuals(new VarDeclaration { type = "class", name = fullClassName }); - - if (actor.isForwardDeclaration) { - foreach (string attribute in actor.attributes) { - writer.Write(attribute + " "); - } - if (actor.isStatic) writer.Write("static "); - writer.WriteLine("{0} {3}{1}( {2} );", fullReturnType, actor.name, string.Join(", ", ParameterList()), actor.nameSpace==null ? "" : actor.nameSpace + "::"); - if (actor.enclosingClass != null) { - writer.WriteLine("template friend class {0};", stateClassName); - } - return; - } - - var body = getFunction("", "body", loopDepth0); - var bodyContext = new Context { - target = body, - catchFErr = getFunction(body.name, "Catch", "Error error", loopDepth0), - }; - - var endContext = TryCatchCompile(actor.body, bodyContext); - - if (endContext.target != null) - { - if (actor.returnType == null) - CompileStatement(new ReturnStatement { FirstSourceLine = actor.SourceLine, expression = "" }, endContext ); - else - throw new Error(actor.SourceLine, "Actor {0} fails to return a value", actor.name); - } - - if (actor.returnType != null) - { - bodyContext.catchFErr.WriteLine("this->~{0}();", stateClassName); - bodyContext.catchFErr.WriteLine("{0}->sendErrorAndDelPromiseRef(error);", This); - } - else - { - bodyContext.catchFErr.WriteLine("delete {0};", This); - } - bodyContext.catchFErr.WriteLine("loopDepth = 0;"); - - if (isTopLevel && actor.nameSpace == null) writer.WriteLine("namespace {"); - - // The "State" class contains all state and user code, to make sure that state names are accessible to user code but - // inherited members of Actor, Callback etc are not. - writer.WriteLine("// This generated class is to be used only via {0}()", actor.name); - WriteTemplate(writer, actorClassFormal); - LineNumber(writer, actor.SourceLine); - writer.WriteLine("class {0} {{", stateClassName); - writer.WriteLine("public:"); - LineNumber(writer, actor.SourceLine); - WriteStateConstructor(writer); - WriteStateDestructor(writer); - WriteFunctions(writer); - foreach (var st in state) - { - LineNumber(writer, st.SourceLine); - writer.WriteLine("\t{0} {1};", st.type, st.name); - } - writer.WriteLine("};"); - - WriteActorClass(writer, fullStateClassName, body); - - if (isTopLevel && actor.nameSpace == null) writer.WriteLine("} // namespace"); // namespace - - WriteActorFunction(writer, fullReturnType); - - if (actor.testCaseParameters != null) - { - writer.WriteLine("ACTOR_TEST_CASE({0}, {1})", actor.name, actor.testCaseParameters); - } - - // Console.WriteLine("\tCompiled ACTOR {0} (line {1})", actor.name, actor.SourceLine); - } - - const string thisAddress = "reinterpret_cast(this)"; - - void ProbeEnter(Function fun, string name, int index = -1) { - if (generateProbes) { - fun.WriteLine("fdb_probe_actor_enter(\"{0}\", {1}, {2});", name, thisAddress, index); - } - var blockIdentifier = GetUidFromString(fun.name); - fun.WriteLine("#ifdef WITH_ACAC"); - fun.WriteLine("static constexpr ActorBlockIdentifier __identifier = UID({0}UL, {1}UL);", blockIdentifier.Item1, blockIdentifier.Item2); - fun.WriteLine("ActorExecutionContextHelper __helper(static_cast<{0}*>(this)->activeActorHelper.actorID, __identifier);", className); - fun.WriteLine("#endif // WITH_ACAC"); - } - - void ProbeExit(Function fun, string name, int index = -1) { - if (generateProbes) { - fun.WriteLine("fdb_probe_actor_exit(\"{0}\", {1}, {2});", name, thisAddress, index); - } - } - - void ProbeCreate(Function fun, string name) { - if (generateProbes) { - fun.WriteLine("fdb_probe_actor_create(\"{0}\", {1});", name, thisAddress); - } - } - - void ProbeDestroy(Function fun, string name) { - if (generateProbes) { - fun.WriteLine("fdb_probe_actor_destroy(\"{0}\", {1});", name, thisAddress); - } - } - - void LineNumber(TextWriter writer, int SourceLine) - { - if(SourceLine == 0) - { - throw new Exception("Internal error: Invalid source line (0)"); - } - if (LineNumbersEnabled) - writer.WriteLine("\t\t\t\t\t\t\t\t\t\t\t\t\t\t\t#line {0} \"{1}\"", SourceLine, sourceFile); - } - void LineNumber(Function writer, int SourceLine) - { - if(SourceLine == 0) - { - throw new Exception("Internal error: Invalid source line (0)"); - } - if (LineNumbersEnabled) - writer.WriteLineUnindented( string.Format("\t\t\t\t\t\t\t\t\t\t\t\t\t\t\t#line {0} \"{1}\"", SourceLine, sourceFile) ); - } - - void TryCatch(Context cx, Function catchFErr, int catchLoopDepth, Action action, bool useLoopDepth = true) - { - if (catchFErr!=null) - { - cx.target.WriteLine("try {"); - cx.target.Indent(+1); - } - action(); - if (catchFErr!=null) - { - cx.target.Indent(-1); - cx.target.WriteLine("}"); - - cx.target.WriteLine("catch (Error& error) {"); - if (useLoopDepth) - cx.target.WriteLine("\tloopDepth = {0};", catchFErr.call("error", AdjustLoopDepth(catchLoopDepth))); - else - cx.target.WriteLine("\t{0};", catchFErr.call("error", "0")); - cx.target.WriteLine("} catch (...) {"); - if (useLoopDepth) - cx.target.WriteLine("\tloopDepth = {0};", catchFErr.call("unknown_error()", AdjustLoopDepth(catchLoopDepth))); - else - cx.target.WriteLine("\t{0};", catchFErr.call("unknown_error()", "0")); - cx.target.WriteLine("}"); - } - } - Context TryCatchCompile(CodeBlock block, Context cx) - { - TryCatch(cx, cx.catchFErr, cx.tryLoopDepth, () => { - cx = Compile(block, cx, true); - if (cx.target != null) - { - var next = getFunction(cx.target.name, "cont", loopDepth); - cx.target.WriteLine("loopDepth = {0};", next.call("loopDepth")); - cx.target = next; - cx.next = null; - } - }); - return cx; - } - - void WriteTemplate(TextWriter writer, params VarDeclaration[] extraParameters) - { - var formals = (actor.templateFormals!=null ? actor.templateFormals.AsEnumerable() : Enumerable.Empty()) - .Concat(extraParameters) - .ToArray(); - - if (formals.Length==0) return; - LineNumber(writer, actor.SourceLine); - writer.WriteLine("template <{0}>", - string.Join(", ", formals.Select( - p => string.Format("{0} {1}", p.type, p.name) - ).ToArray())); - } - string GetTemplateActuals(params VarDeclaration[] extraParameters) - { - var formals = (actor.templateFormals != null ? actor.templateFormals.AsEnumerable() : Enumerable.Empty()) - .Concat(extraParameters) - .ToArray(); - - if (formals.Length == 0) return ""; - else return "<" + - string.Join(", ", formals.Select( - p => p.name - ).ToArray()) - + ">"; - } - - bool WillContinue(Statement stmt) - { - return Flatten(stmt).Any( - st => (st is ChooseStatement || st is WaitStatement || st is TryStatement)); - } - - CodeBlock AsCodeBlock(Statement statement) - { - // SOMEDAY: Is this necessary? Maybe we should just be compiling statements? - var cb = statement as CodeBlock; - if (cb != null) return cb; - return new CodeBlock { statements = new Statement[] { statement } }; - } - - void CompileStatement(PlainOldCodeStatement stmt, Context cx) - { - LineNumber(cx.target,stmt.FirstSourceLine); - cx.target.WriteLine(stmt.code); - } - void CompileStatement(StateDeclarationStatement stmt, Context cx) - { - // if this state declaration is at the very top of the actor body - if (actor.body.statements.Select(x => x as StateDeclarationStatement).TakeWhile(x => x != null).Any(x => x == stmt)) - { - // Initialize the state in the constructor, not here - state.Add(new StateVar { SourceLine = stmt.FirstSourceLine, name = stmt.decl.name, type = stmt.decl.type, - initializer = stmt.decl.initializer, initializerConstructorSyntax = stmt.decl.initializerConstructorSyntax } ); - } - else - { - // State variables declared elsewhere must have a default constructor - state.Add(new StateVar { SourceLine = stmt.FirstSourceLine, name = stmt.decl.name, type = stmt.decl.type, initializer = null }); - if (stmt.decl.initializer != null) - { - LineNumber(cx.target, stmt.FirstSourceLine); - if (stmt.decl.initializerConstructorSyntax || stmt.decl.initializer=="") - cx.target.WriteLine("{0} = {1}({2});", stmt.decl.name, stmt.decl.type, stmt.decl.initializer); - else - cx.target.WriteLine("{0} = {1};", stmt.decl.name, stmt.decl.initializer); - } - } - } - - void CompileStatement(ForStatement stmt, Context cx) - { - // for( initExpression; condExpression; nextExpression ) body; - - bool noCondition = stmt.condExpression == "" || stmt.condExpression == "true" || stmt.condExpression == "1"; - - if (!WillContinue(stmt.body)) - { - // We can write this loop without anything fancy, because there are no wait statements in it - if (EmitNativeLoop(stmt.FirstSourceLine, "for(" + stmt.initExpression + ";" + stmt.condExpression + ";" + stmt.nextExpression+")", stmt.body, cx) - && noCondition ) - cx.unreachable(); - } - else - { - // First compile the initExpression - CompileStatement( new PlainOldCodeStatement {code = stmt.initExpression + ";", FirstSourceLine = stmt.FirstSourceLine}, cx ); - - // fullBody = { if (!(condExpression)) break; body; } - Statement fullBody = noCondition ? stmt.body : - new CodeBlock - { - statements = new Statement[] { - new IfStatement - { - expression = "!(" + stmt.condExpression + ")", - ifBody = new BreakStatement { FirstSourceLine = stmt.FirstSourceLine }, - FirstSourceLine = stmt.FirstSourceLine - } - }.Concat(AsCodeBlock(stmt.body).statements).ToArray(), - FirstSourceLine = stmt.FirstSourceLine - }; - - Function loopF = getFunction(cx.target.name, "loopHead", loopDepth); - Function loopBody = getFunction(cx.target.name, "loopBody", loopDepth); - Function breakF = getFunction(cx.target.name, "break", loopDepth); - Function continueF = stmt.nextExpression == "" ? loopF : getFunction(cx.target.name, "continue", loopDepth); - - // TODO: Could we use EmitNativeLoop() here? - - loopF.WriteLine("int oldLoopDepth = ++loopDepth;"); - loopF.WriteLine("while (loopDepth == oldLoopDepth) loopDepth = {0};", loopBody.call("loopDepth")); - - Function endLoop = Compile(AsCodeBlock(fullBody), cx.LoopContext( loopBody, breakF, continueF, +1 ), true).target; - if (endLoop != null && endLoop != loopBody) { - if (stmt.nextExpression != "") - CompileStatement( new PlainOldCodeStatement {code = stmt.nextExpression + ";", FirstSourceLine = stmt.FirstSourceLine}, cx.WithTarget(endLoop) ); - endLoop.WriteLine("if (loopDepth == 0) return {0};", loopF.call("0")); - } - - cx.target.WriteLine("loopDepth = {0};", loopF.call("loopDepth")); - - if (continueF != loopF && continueF.wasCalled) { - CompileStatement( new PlainOldCodeStatement {code = stmt.nextExpression + ";", FirstSourceLine = stmt.FirstSourceLine}, cx.WithTarget(continueF) ); - continueF.WriteLine("if (loopDepth == 0) return {0};", loopF.call("0")); - } - - if (breakF.wasCalled) - TryCatch(cx.WithTarget(breakF), cx.catchFErr, cx.tryLoopDepth, - () => { breakF.WriteLine("return {0};", cx.next.call("loopDepth")); }); - else - cx.unreachable(); - } - } - - Dictionary iterators = new Dictionary(); - - string getIteratorName(Context cx) - { - string name = "RangeFor" + cx.target.name + "Iterator"; - if (!iterators.ContainsKey(name)) - iterators[name] = 0; - return string.Format("{0}{1}", name, iterators[name]++); - } - - void CompileStatement(RangeForStatement stmt, Context cx) - { - // If stmt does not contain a wait statement, rewrite the original c++11 range-based for loop - // If there is a wait, we need to rewrite the loop as: - // for(a:b) c; ==> for(__iter=std::begin(b); __iter!=std::end(b); ++__iter) { a = *__iter; c; } - // where __iter is stored as a state variable - - if (WillContinue(stmt.body)) - { - StateVar container = state.FirstOrDefault(s => s.name == stmt.rangeExpression); - if (container == null) - { - throw new Error(stmt.FirstSourceLine, "container of range-based for with continuation must be a state variable"); - } - - var iter = getIteratorName(cx); - state.Add(new StateVar { SourceLine = stmt.FirstSourceLine, name = iter, type = "decltype(std::begin(std::declval<" + container.type + ">()))", initializer = null }); - var equivalent = new ForStatement { - initExpression = iter + " = std::begin(" + stmt.rangeExpression + ")", - condExpression = iter + " != std::end(" + stmt.rangeExpression + ")", - nextExpression = "++" + iter, - FirstSourceLine = stmt.FirstSourceLine, - body = new CodeBlock - { - statements = new Statement[] { - new PlainOldCodeStatement { FirstSourceLine = stmt.FirstSourceLine, code = stmt.rangeDecl + " = *" + iter + ";" }, - stmt.body - } - } - }; - CompileStatement(equivalent, cx); - } - else - { - EmitNativeLoop(stmt.FirstSourceLine, "for( " + stmt.rangeDecl + " : " + stmt.rangeExpression + " )", stmt.body, cx); - } - } - - void CompileStatement(WhileStatement stmt, Context cx) - { - // Compile while (x) { y } as for(;x;) { y } - var equivalent = new ForStatement - { - condExpression = stmt.expression, - body = stmt.body, - FirstSourceLine = stmt.FirstSourceLine, - }; - - CompileStatement(equivalent, cx); - } - - void CompileStatement(LoopStatement stmt, Context cx) - { - // Compile loop { body } as for(;;;) { body } - var equivalent = new ForStatement - { - body = stmt.body, - FirstSourceLine = stmt.FirstSourceLine - }; - CompileStatement(equivalent, cx); - } - - // Writes out a loop in native C++ (with no continuation passing) - // Returns true if the loop is known to have no normal exit (is unreachable) - private bool EmitNativeLoop(int sourceLine, string head, Statement body, Context cx) - { - LineNumber(cx.target, sourceLine); - cx.target.WriteLine(head + " {"); - cx.target.Indent(+1); - var literalBreak = new LiteralBreak(); - Compile(AsCodeBlock(body), cx.LoopContext(cx.target, literalBreak, new LiteralContinue(), 0), true); - cx.target.Indent(-1); - cx.target.WriteLine("}"); - return !literalBreak.wasCalled; - } - - void CompileStatement(ChooseStatement stmt, Context cx) - { - int group = ++this.chooseGroups; - //string cbGroup = "ChooseGroup" + getFunction(cx.target.name,"W").name; // SOMEDAY - - var codeblock = stmt.body as CodeBlock; - if (codeblock == null) - throw new Error(stmt.FirstSourceLine, "'choose' must be followed by a compound statement."); - var choices = codeblock.statements - .OfType() - .Select( (ch,i) => new { - Stmt = ch, - Group = group, - Index = this.whenCount+i, - Body = getFunction(cx.target.name, "when", - new string[] { string.Format("{0} const& {2}{1}", ch.wait.result.type, ch.wait.result.name, ch.wait.resultIsState?"__":""), loopDepth }, - new string[] { string.Format("{0} && {2}{1}", ch.wait.result.type, ch.wait.result.name, ch.wait.resultIsState?"__":""), loopDepth } - ), - Future = string.Format("__when_expr_{0}", this.whenCount + i), - CallbackType = string.Format("{3}< {0}, {1}, {2} >", fullClassName, this.whenCount + i, ch.wait.result.type, ch.wait.isWaitNext ? "ActorSingleCallback" : "ActorCallback"), - CallbackTypeInStateClass = string.Format("{3}< {0}, {1}, {2} >", className, this.whenCount + i, ch.wait.result.type, ch.wait.isWaitNext ? "ActorSingleCallback" : "ActorCallback") - }) - .ToArray(); - this.whenCount += choices.Length; - if (choices.Length != codeblock.statements.Length) - throw new Error(codeblock.statements.First(x=>!(x is WhenStatement)).FirstSourceLine, "only 'when' statements are valid in an 'choose' block."); - - var exitFunc = getFunction("exitChoose", ""); - exitFunc.returnType = "void"; - exitFunc.WriteLine("if (actorWaitStateIsWaiting({0}->actor_wait_state)) {0}->actor_wait_state = ACTOR_WAIT_STATE_NOT_WAITING;", This); - foreach(var ch in choices) - exitFunc.WriteLine("{0}->{1}::remove();", This, ch.CallbackTypeInStateClass); - exitFunc.endIsUnreachable = true; - - //state.Add(new StateVar { SourceLine = stmt.FirstSourceLine, type = "CallbackGroup", name = cbGroup, callbackCatchFErr = cx.catchFErr }); - bool reachable = false; - foreach(var ch in choices) { - callbacks.Add(new CallbackVar - { - SourceLine = ch.Stmt.FirstSourceLine, - CallbackGroup = ch.Group, - type = ch.CallbackType - }); - var r = ch.Body; - if (ch.Stmt.wait.resultIsState) - { - Function overload = r.popOverload(); - CompileStatement(new StateDeclarationStatement - { - FirstSourceLine = ch.Stmt.FirstSourceLine, - decl = new VarDeclaration { - type = ch.Stmt.wait.result.type, - name = ch.Stmt.wait.result.name, - initializer = "__" + ch.Stmt.wait.result.name, - initializerConstructorSyntax = false - } - }, cx.WithTarget(r)); - if (overload != null) - { - overload.WriteLine("{0} = std::move(__{0});", ch.Stmt.wait.result.name); - r.setOverload(overload); - } - } - if (ch.Stmt.body != null) - { - r = Compile(AsCodeBlock(ch.Stmt.body), cx.WithTarget(r), true).target; - } - if (r != null) - { - reachable = true; - if (cx.next.formalParameters.Length == 1) - r.WriteLine("loopDepth = {0};", cx.next.call("loopDepth")); - else { - Function overload = r.popOverload(); - r.WriteLine("loopDepth = {0};", cx.next.call(ch.Stmt.wait.result.name, "loopDepth")); - if (overload != null) { - overload.WriteLine("loopDepth = {0};", cx.next.call(string.Format("std::move({0})", ch.Stmt.wait.result.name), "loopDepth")); - r.setOverload(overload); - } - } - } - - var cbFunc = new Function { - name = "callback_fire", - returnType = "void", - formalParameters = new string[] { - ch.CallbackTypeInStateClass + "*", - ch.Stmt.wait.result.type + " const& value" - }, - endIsUnreachable = true - }; - cbFunc.addOverload(ch.CallbackTypeInStateClass + "*", ch.Stmt.wait.result.type + " && value"); - functions.Add(string.Format("{0}#{1}", cbFunc.name, ch.Index), cbFunc); - cbFunc.Indent(codeIndent); - ProbeEnter(cbFunc, actor.name, ch.Index); - cbFunc.WriteLine("{0};", exitFunc.call()); - - Function _overload = cbFunc.popOverload(); - TryCatch(cx.WithTarget(cbFunc), cx.catchFErr, cx.tryLoopDepth, () => { - cbFunc.WriteLine("{0};", ch.Body.call("value", "0")); - }, false); - if (_overload != null) { - TryCatch(cx.WithTarget(_overload), cx.catchFErr, cx.tryLoopDepth, () => { - _overload.WriteLine("{0};", ch.Body.call("std::move(value)", "0")); - }, false); - cbFunc.setOverload(_overload); - } - ProbeExit(cbFunc, actor.name, ch.Index); - - var errFunc = new Function - { - name = "callback_error", - returnType = "void", - formalParameters = new string[] { - ch.CallbackTypeInStateClass + "*", - "Error err" - }, - endIsUnreachable = true - }; - functions.Add(string.Format("{0}#{1}", errFunc.name, ch.Index), errFunc); - errFunc.Indent(codeIndent); - ProbeEnter(errFunc, actor.name, ch.Index); - errFunc.WriteLine("{0};", exitFunc.call()); - TryCatch(cx.WithTarget(errFunc), cx.catchFErr, cx.tryLoopDepth, () => - { - errFunc.WriteLine("{0};", cx.catchFErr.call("err", "0")); - }, false); - ProbeExit(errFunc, actor.name, ch.Index); - } - - bool firstChoice = true; - foreach (var ch in choices) - { - string getFunc = ch.Stmt.wait.isWaitNext ? "pop" : "get"; - LineNumber(cx.target, ch.Stmt.wait.FirstSourceLine); - if (ch.Stmt.wait.isWaitNext) { - cx.target.WriteLine("auto {0} = {1};", ch.Future, ch.Stmt.wait.futureExpression); - cx.target.WriteLine("static_assert(std::is_same>::value || std::is_same>::value, \"invalid type\");", ch.Future, ch.Stmt.wait.result.type); - - } else { - cx.target.WriteLine("{2}<{3}> {0} = {1};", ch.Future, ch.Stmt.wait.futureExpression, "StrictFuture", ch.Stmt.wait.result.type); - } - - if (firstChoice) - { - // Do this check only after evaluating the expression for the first wait expression, so that expression cannot be short circuited by cancellation. - // So wait( expr() ) will always evaluate `expr()`, but choose { when ( wait(success( expr2() )) {} } need - // not evaluate `expr2()`. - firstChoice = false; - LineNumber(cx.target, stmt.FirstSourceLine); - if (actor.IsCancellable()) - cx.target.WriteLine("if (actorWaitStateIsCancelled({1}->actor_wait_state)) return {0};", cx.catchFErr.call("actor_cancelled()", AdjustLoopDepth(cx.tryLoopDepth)), This); - } - - cx.target.WriteLine("if ({0}.isReady()) {{ if ({0}.isError()) return {2}; else return {1}; }};", ch.Future, ch.Body.call(ch.Future + "." + getFunc + "()", "loopDepth"), cx.catchFErr.call(ch.Future + ".getError()", AdjustLoopDepth(cx.tryLoopDepth))); - } - cx.target.WriteLine("{1}->actor_wait_state = {0};", group, This); - foreach (var ch in choices) - { - LineNumber(cx.target, ch.Stmt.wait.FirstSourceLine); - cx.target.WriteLine("{0}.addCallbackAndClear(static_cast<{1}*>({2}));", ch.Future, ch.CallbackTypeInStateClass, This); - } - cx.target.WriteLine("loopDepth = 0;");//cx.target.WriteLine("return 0;"); - - if (!reachable) cx.unreachable(); - } - void CompileStatement(BreakStatement stmt, Context cx) - { - if (cx.breakF == null) - throw new Error(stmt.FirstSourceLine, "break outside loop"); - if (cx.breakF is LiteralBreak) - cx.target.WriteLine("{0};", cx.breakF.call()); - else - { - cx.target.WriteLine("return {0}; // break", cx.breakF.call("loopDepth==0?0:loopDepth-1")); - } - cx.unreachable(); - } - void CompileStatement(ContinueStatement stmt, Context cx) - { - if (cx.continueF == null) - throw new Error(stmt.FirstSourceLine, "continue outside loop"); - if (cx.continueF is LiteralContinue) - cx.target.WriteLine("{0};", cx.continueF.call()); - else - cx.target.WriteLine("return {0}; // continue", cx.continueF.call("loopDepth")); - cx.unreachable(); - } - void CompileStatement(WaitStatement stmt, Context cx) - { - var equiv = new ChooseStatement - { - body = new CodeBlock - { - statements = new Statement[] { - new WhenStatement { - wait = stmt, - body = null, - FirstSourceLine = stmt.FirstSourceLine, - } - }, - FirstSourceLine = stmt.FirstSourceLine - }, - FirstSourceLine = stmt.FirstSourceLine - }; - if (!stmt.resultIsState) { - cx.next.formalParameters = new string[] { - string.Format("{0} const& {1}", stmt.result.type, stmt.result.name), - loopDepth }; - cx.next.addOverload( - string.Format("{0} && {1}", stmt.result.type, stmt.result.name), - loopDepth); - } - CompileStatement(equiv, cx); - } - void CompileStatement(CodeBlock stmt, Context cx) - { - cx.target.WriteLine("{"); - cx.target.Indent(+1); - var end = Compile(stmt, cx, true); - cx.target.Indent(-1); - cx.target.WriteLine("}"); - if (end.target == null) - cx.unreachable(); - else if (end.target != cx.target) - end.target.WriteLine("loopDepth = {0};", cx.next.call("loopDepth")); - } - void CompileStatement(ReturnStatement stmt, Context cx) - { - LineNumber(cx.target, stmt.FirstSourceLine); - if ((stmt.expression == "") != (actor.returnType == null)) - throw new Error(stmt.FirstSourceLine, "Return statement does not match actor declaration"); - if (actor.returnType != null) - { - if (stmt.expression == "Never()") - { - // `return Never();` destroys state immediately but never returns to the caller - cx.target.WriteLine("this->~{0}();", stateClassName); - cx.target.WriteLine("{0}->sendAndDelPromiseRef(Never());", This); - } - else - { - // Short circuit if there are no futures outstanding, but still evaluate the expression - // if it has side effects - cx.target.WriteLine("if (!{0}->SAV<{1}>::futures) {{ (void)({2}); this->~{3}(); {0}->destroy(); return 0; }}", This, actor.returnType, stmt.expression, stateClassName); - // Build the return value directly in SAV::value_storage - // If the expression is exactly the name of a state variable, std::move() it - if (state.Exists(s => s.name == stmt.expression)) - { - cx.target.WriteLine("new (&{0}->SAV< {1} >::value()) {1}(std::move({2})); // state_var_RVO", This, actor.returnType, stmt.expression); - } - else - { - cx.target.WriteLine("new (&{0}->SAV< {1} >::value()) {1}({2});", This, actor.returnType, stmt.expression); - } - // Destruct state - cx.target.WriteLine("this->~{0}();", stateClassName); - // Tell SAV to return the value we already constructed in value_storage - cx.target.WriteLine("{0}->finishSendAndDelPromiseRef();", This); - } - } else - cx.target.WriteLine("delete {0};", This); - cx.target.WriteLine("return 0;"); - cx.unreachable(); - } - void CompileStatement(IfStatement stmt, Context cx) - { - bool useContinuation = WillContinue(stmt.ifBody) || WillContinue(stmt.elseBody); - - LineNumber(cx.target, stmt.FirstSourceLine); - cx.target.WriteLine("if {1}({0})", stmt.expression, stmt.constexpr ? "constexpr " : ""); - cx.target.WriteLine("{"); - cx.target.Indent(+1); - Function ifTarget = Compile(AsCodeBlock(stmt.ifBody), cx, useContinuation).target; - if (useContinuation && ifTarget != null) - ifTarget.WriteLine("loopDepth = {0};", cx.next.call("loopDepth")); - cx.target.Indent(-1); - cx.target.WriteLine("}"); - Function elseTarget = null; - if (stmt.elseBody != null || useContinuation) - { - cx.target.WriteLine("else"); - cx.target.WriteLine("{"); - cx.target.Indent(+1); - elseTarget = cx.target; - if (stmt.elseBody != null) - { - elseTarget = Compile(AsCodeBlock(stmt.elseBody), cx, useContinuation).target; - } - if (useContinuation && elseTarget != null) - elseTarget.WriteLine("loopDepth = {0};", cx.next.call("loopDepth")); - cx.target.Indent(-1); - cx.target.WriteLine("}"); - } - if (ifTarget == null && stmt.elseBody != null && elseTarget == null) - cx.unreachable(); - else if (!cx.next.wasCalled && useContinuation) - throw new Exception("Internal error: IfStatement: next not called?"); - } - void CompileStatement(TryStatement stmt, Context cx) - { - bool reachable = false; - - if (stmt.catches.Count != 1) throw new Error(stmt.FirstSourceLine, "try statement must have exactly one catch clause"); - var c = stmt.catches[0]; - string catchErrorParameterName = ""; - if (c.expression != "...") { - string exp = c.expression.Replace(" ",""); - if (!exp.StartsWith("Error&")) - throw new Error(c.FirstSourceLine, "Only type 'Error' or '...' may be caught in an actor function"); - catchErrorParameterName = exp.Substring(6); - } - if (catchErrorParameterName == "") catchErrorParameterName = "__current_error"; - - var catchFErr = getFunction(cx.target.name, "Catch", "const Error& " + catchErrorParameterName, loopDepth0); - catchFErr.exceptionParameterIs = catchErrorParameterName; - var end = TryCatchCompile(AsCodeBlock(stmt.tryBody), cx.WithCatch(catchFErr)); - if (end.target != null) reachable = true; - - if (end.target!=null) - TryCatch(end, cx.catchFErr, cx.tryLoopDepth, () => - end.target.WriteLine("loopDepth = {0};", cx.next.call("loopDepth"))); - - // Now to write the catch function - TryCatch(cx.WithTarget(catchFErr), cx.catchFErr, cx.tryLoopDepth, () => - { - var cend = Compile(AsCodeBlock(c.body), cx.WithTarget(catchFErr), true); - if (cend.target != null) cend.target.WriteLine("loopDepth = {0};", cx.next.call("loopDepth")); - if (cend.target != null) reachable = true; - }); - - if (!reachable) cx.unreachable(); - } - void CompileStatement(ThrowStatement stmt, Context cx) - { - LineNumber(cx.target, stmt.FirstSourceLine); - - if (stmt.expression == "") - { - if (cx.target.exceptionParameterIs != null) - cx.target.WriteLine("return {0};", cx.catchFErr.call(cx.target.exceptionParameterIs, AdjustLoopDepth( cx.tryLoopDepth ))); - else - throw new Error(stmt.FirstSourceLine, "throw statement with no expression has no current exception in scope"); - } - else - cx.target.WriteLine("return {0};", cx.catchFErr.call(stmt.expression, AdjustLoopDepth( cx.tryLoopDepth ))); - cx.unreachable(); - } - void CompileStatement(Statement stmt, Context cx) - { - // Use reflection for double dispatch. SOMEDAY: Use a Dictionary> and expression trees to memoize - var method = typeof(ActorCompiler).GetMethod("CompileStatement", - System.Reflection.BindingFlags.NonPublic|System.Reflection.BindingFlags.Instance|System.Reflection.BindingFlags.ExactBinding, - null, new Type[] { stmt.GetType(), typeof(Context) }, null); - if (method == null) - throw new Error(stmt.FirstSourceLine, "Statement type {0} not supported yet.", stmt.GetType().Name); - try - { - method.Invoke(this, new object[] { stmt, cx }); - } - catch (System.Reflection.TargetInvocationException e) - { - if (!(e.InnerException is Error)) - Console.Error.WriteLine("\tHit error <{0}> for statement type {1} at line {2}\n\tStack Trace:\n{3}", - e.InnerException.Message, stmt.GetType().Name, stmt.FirstSourceLine, e.InnerException.StackTrace); - throw e.InnerException; - } - } - - // Compile returns a new context based on the one that is passed in, but (unlike CompileStatement) - // does not modify its parameter - // The target of the returned context is null if the end of the CodeBlock is unreachable (otherwise - // it is the target Function to which the end of the CodeBlock was written) - Context Compile(CodeBlock block, Context context, bool okToContinue=true) - { - var cx = context.Clone(); cx.next = null; - foreach (var stmt in block.statements) - { - if (cx.target == null) - { - throw new Error(stmt.FirstSourceLine, "Unreachable code."); - //Console.Error.WriteLine("\t(WARNING) Unreachable code at line {0}.", stmt.FirstSourceLine); - //break; - } - if (cx.next == null) - cx.next = getFunction(cx.target.name, "cont", loopDepth); - CompileStatement(stmt, cx); - if (cx.next.wasCalled) - { - if (cx.target == null) throw new Exception("Unreachable continuation called?"); - if (!okToContinue) throw new Exception("Unexpected continuation"); - cx.target = cx.next; - cx.next = null; - } - } - return cx; - } - - Dictionary functions = new Dictionary(); - - void WriteFunctions(TextWriter writer) - { - foreach (var func in functions.Values) - { - string body = func.BodyText; - if (body.Length != 0) - { - WriteFunction(writer, func, body); - } - if (func.overload != null) - { - string overloadBody = func.overload.BodyText; - if (overloadBody.Length != 0) - { - WriteFunction(writer, func.overload, overloadBody); - } - } - } - } - - private static void WriteFunction(TextWriter writer, Function func, string body) - { - writer.WriteLine(memberIndentStr + "{0}{1}({2}){3}", - func.returnType == "" ? "" : func.returnType + " ", - func.useByName(), - string.Join(",", func.formalParameters), - func.specifiers == "" ? "" : " " + func.specifiers); - if (func.returnType != "") - writer.WriteLine(memberIndentStr + "{"); - writer.WriteLine(body); - if (!func.endIsUnreachable) - writer.WriteLine(memberIndentStr + "\treturn loopDepth;"); - writer.WriteLine(memberIndentStr + "}"); - } - - Function getFunction(string baseName, string addName, string[] formalParameters, string[] overloadFormalParameters) - { - string proposedName; - if (addName == "cont" && baseName.Length>=5 && baseName.Substring(baseName.Length - 5, 4) == "cont") - proposedName = baseName.Substring(0, baseName.Length - 1); - else - proposedName = baseName + addName; - - int i = 0; - while (functions.ContainsKey(string.Format("{0}{1}", proposedName, ++i))) ; - - var f = new Function { - name = string.Format("{0}{1}", proposedName, i), - returnType = "int", - formalParameters = formalParameters - }; - if (overloadFormalParameters != null) { - f.addOverload(overloadFormalParameters); - } - f.Indent(codeIndent); - functions.Add(f.name, f); - return f; - } - - Function getFunction(string baseName, string addName, params string[] formalParameters) - { - return getFunction(baseName, addName, formalParameters, null); - } - - string[] ParameterList() - { - return actor.parameters.Select(p => - { - // SOMEDAY: pass small built in types by value - if (p.initializer != "") - return string.Format("{0} const& {1} = {2}", p.type, p.name, p.initializer); - else - return string.Format("{0} const& {1}", p.type, p.name); - }).ToArray(); - } - void WriteCancelFunc(TextWriter writer) - { - if (actor.IsCancellable()) - { - Function cancelFunc = new Function - { - name = "cancel", - returnType = "void", - formalParameters = new string[] {}, - endIsUnreachable = true, - publicName = true, - specifiers = "override" - }; - cancelFunc.Indent(codeIndent); - cancelFunc.WriteLine("auto wait_state = this->actor_wait_state;"); - cancelFunc.WriteLine("this->actor_wait_state = ACTOR_WAIT_STATE_CANCELLED;"); - cancelFunc.WriteLine("switch (wait_state) {"); - int lastGroup = -1; - foreach (var cb in callbacks.OrderBy(cb => cb.CallbackGroup)) - if (cb.CallbackGroup != lastGroup) - { - lastGroup = cb.CallbackGroup; - cancelFunc.WriteLine("case {0}: this->a_callback_error(({1}*)0, actor_cancelled()); break;", cb.CallbackGroup, cb.type); - } - cancelFunc.WriteLine("}"); - WriteFunction(writer, cancelFunc, cancelFunc.BodyText); - } - } - - void WriteConstructor(Function body, TextWriter writer, string fullStateClassName) - { - Function constructor = new Function - { - name = className, - returnType = "", - formalParameters = ParameterList(), - endIsUnreachable = true, - publicName = true - }; - - // Initializes class member variables - constructor.Indent(codeIndent); - constructor.WriteLine( " : Actor<" + (actor.returnType == null ? "void" : actor.returnType) + ">()," ); - constructor.WriteLine( " {0}({1}),", fullStateClassName, string.Join(", ", actor.parameters.Select(p => p.name))); - constructor.WriteLine( " activeActorHelper(__actorIdentifier)"); - constructor.Indent(-1); - - constructor.WriteLine("{"); - constructor.Indent(+1); - - ProbeEnter(constructor, actor.name); - - constructor.WriteLine("#ifdef ENABLE_SAMPLING"); - constructor.WriteLine("this->lineage.setActorName(\"{0}\");", actor.name); - constructor.WriteLine("LineageScope _(&this->lineage);"); - // constructor.WriteLine("getCurrentLineage()->modify(&StackLineage::actorName) = \"{0}\"_sr;", actor.name); - constructor.WriteLine("#endif"); - - constructor.WriteLine("this->{0};", body.call()); - - ProbeExit(constructor, actor.name); - - WriteFunction(writer, constructor, constructor.BodyText); - } - - void WriteStateConstructor(TextWriter writer) - { - Function constructor = new Function - { - name = stateClassName, - returnType = "", - formalParameters = ParameterList(), - endIsUnreachable = true, - publicName = true - }; - constructor.Indent(codeIndent); - string ini = null; - int line = actor.SourceLine; - var initializers = state.AsEnumerable(); - foreach (var s in initializers) - if (s.initializer != null) - { - LineNumber(constructor, line); - if (ini != null) - { - constructor.WriteLine(ini + ","); - ini = " "; - } - else - { - ini = " : "; - } - - ini += string.Format("{0}({1})", s.name, s.initializer); - line = s.SourceLine; - } - LineNumber(constructor, line); - if (ini != null) - constructor.WriteLine(ini); - constructor.Indent(-1); - constructor.WriteLine("{"); - constructor.Indent(+1); - ProbeCreate(constructor, actor.name); - WriteFunction(writer, constructor, constructor.BodyText); - } - - void WriteStateDestructor(TextWriter writer) { - Function destructor = new Function - { - name = String.Format("~{0}", stateClassName), - returnType = "", - formalParameters = new string[0], - endIsUnreachable = true, - publicName = true, - }; - destructor.Indent(codeIndent); - destructor.Indent(-1); - destructor.WriteLine("{"); - destructor.Indent(+1); - ProbeDestroy(destructor, actor.name); - WriteFunction(writer, destructor, destructor.BodyText); - } - - IEnumerable Flatten(Statement stmt) - { - if (stmt == null) return new Statement[] { }; - var fl = new TypeSwitch>(stmt) - .Case(s => Flatten(s.body)) - .Case(s => Flatten(s.body)) - .Case(s => Flatten(s.body)) - .Case(s => Flatten(s.body)) - .Case(s => s.statements.SelectMany(t=>Flatten(t))) - .Case( s => Flatten(s.ifBody).Concat(Flatten(s.elseBody)) ) - .Case( s => Flatten(s.body) ) - .Case( s => Flatten(s.body) ) - .Case( s => Flatten(s.tryBody).Concat( s.catches.SelectMany(c=>Flatten(c.body)) ) ) - .Case(s => Enumerable.Empty()) - .Return(); - return new Statement[]{stmt}.Concat(fl); - } - - void FindState() - { - state = actor.parameters - .Select( - p=>new StateVar { SourceLine = actor.SourceLine, name=p.name, type=p.type, initializer=p.name, initializerConstructorSyntax=false } ) - .ToList(); - } - - // Generate an expression equivalent to max(0, loopDepth-subtract) for the given constant subtract - string AdjustLoopDepth(int subtract) - { - if (subtract == 0) - return "loopDepth"; - else - return string.Format("std::max(0, loopDepth - {0})", subtract); - } - } -} diff --git a/flow/actorcompiler/ActorCompiler.targets b/flow/actorcompiler/ActorCompiler.targets deleted file mode 100644 index 16eaec6a8af..00000000000 --- a/flow/actorcompiler/ActorCompiler.targets +++ /dev/null @@ -1,54 +0,0 @@ - - - - - - - true - - - - - - - - _ActorCompiler - - - - - - - - - - $(BuildGenerateSourcesTargets); - ComputeACOutput; - - - - - - - - - - - - - - - - diff --git a/flow/actorcompiler/ActorCompiler.xml b/flow/actorcompiler/ActorCompiler.xml deleted file mode 100644 index 19ed212ddcf..00000000000 --- a/flow/actorcompiler/ActorCompiler.xml +++ /dev/null @@ -1,62 +0,0 @@ - - - - - - - - - - Options - - - - - Command Line - - - - - - Compile generated file - - - The resulting file is a C++ module to be compiled, not a header file to be included - - - - - Actor Compiler Options - - - Actor Compiler Options - - - - - Additional C++ Options - - - Options passed to the C++ compiler processing the generated code - - - - - - - - - - \ No newline at end of file diff --git a/flow/actorcompiler/ActorParser.cs b/flow/actorcompiler/ActorParser.cs deleted file mode 100644 index 36392627d0c..00000000000 --- a/flow/actorcompiler/ActorParser.cs +++ /dev/null @@ -1,1094 +0,0 @@ -/* - * ActorParser.cs - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -using System; -using System.Collections.Generic; -using System.Linq; -using System.Text.RegularExpressions; - -namespace actorcompiler -{ - class Error : Exception - { - public int SourceLine { get; private set; } - public Error(int SourceLine, string format, params object[] args) - : base(string.Format(format,args)) - { - this.SourceLine = SourceLine; - } - }; - - class ErrorMessagePolicy - { - public bool DisableDiagnostics = false; - public void HandleActorWithoutWait(String sourceFile, Actor actor) - { - if (!DisableDiagnostics && !actor.isTestCase) - { - // TODO(atn34): Once cmake is the only build system we can make this an error instead of a warning. - Console.Error.WriteLine("{0}:{1}: warning: ACTOR {2} does not contain a wait() statement", sourceFile, actor.SourceLine, actor.name); - } - } - public bool ActorsNoDiscardByDefault() { - return !DisableDiagnostics; - } - } - - class Token - { - public string Value; - public int Position; - public int SourceLine; - public int BraceDepth; - public int ParenDepth; - public bool IsWhitespace { get { return Value == " " || Value == "\n" || Value == "\r" || Value == "\r\n" || Value == "\t" || Value.StartsWith("//") || Value.StartsWith("/*"); } } - public override string ToString() { return Value; } - public Token Assert(string error, Func pred) - { - if (!pred(this)) throw new Error(SourceLine, error); - return this; - } - public TokenRange GetMatchingRangeIn(TokenRange range) - { - Func pred; - int dir; - switch (Value) { - case "(": pred = t=> t.Value != ")" || t.ParenDepth != ParenDepth; dir = +1; break; - case ")": pred = t=> t.Value != "(" || t.ParenDepth != ParenDepth; dir = -1; break; - case "{": pred = t=> t.Value != "}" || t.BraceDepth != BraceDepth; dir = +1; break; - case "}": pred = t=> t.Value != "{" || t.BraceDepth != BraceDepth; dir = -1; break; - case "<": return - new TokenRange(range.GetAllTokens(), - Position+1, - AngleBracketParser.NotInsideAngleBrackets( - new TokenRange(range.GetAllTokens(), Position, range.End)) - .Skip(1) // skip the "<", which is considered "outside" - .First() // get the ">", which is likewise "outside" - .Position); - case "[": return - new TokenRange(range.GetAllTokens(), - Position+1, - BracketParser.NotInsideBrackets( - new TokenRange(range.GetAllTokens(), Position, range.End)) - .Skip(1) // skip the "[", which is considered "outside" - .First() // get the "]", which is likewise "outside" - .Position); - default: throw new NotSupportedException("Can't match this token!"); - } - TokenRange r; - if (dir == -1) - { - r = new TokenRange(range.GetAllTokens(), range.Begin, Position) - .RevTakeWhile(pred); - if (r.Begin == range.Begin) - throw new Error(SourceLine, "Syntax error: Unmatched " + Value); - } - else - { - r = new TokenRange(range.GetAllTokens(), Position+1, range.End) - .TakeWhile(pred); - if (r.End == range.End) - throw new Error(SourceLine, "Syntax error: Unmatched " + Value); - } - return r; - } - }; - - class TokenRange : IEnumerable - { - public TokenRange(Token[] tokens, int beginPos, int endPos) - { - if (beginPos > endPos) throw new InvalidOperationException("Invalid TokenRange"); - this.tokens = tokens; - this.beginPos = beginPos; - this.endPos = endPos; - } - - public bool IsEmpty { get { return beginPos==endPos; } } - public int Begin { get { return beginPos; } } - public int End { get { return endPos; } } - public Token First() { - if (beginPos == endPos) throw new InvalidOperationException("Empty TokenRange"); - return tokens[beginPos]; - } - public Token Last() { - if (beginPos == endPos) throw new InvalidOperationException("Empty TokenRange"); - return tokens[endPos - 1]; - } - public Token Last(Func pred) - { - for (int i = endPos - 1; i >= beginPos; i--) - if (pred(tokens[i])) - return tokens[i]; - throw new Exception("Matching token not found"); - } - public TokenRange Skip(int count) - { - return new TokenRange(tokens, beginPos + count, endPos); - } - public TokenRange Consume(string value) - { - First().Assert("Expected " + value, t => t.Value == value); - return Skip(1); - } - public TokenRange Consume(string error, Func pred) - { - First().Assert(error, pred); - return Skip(1); - } - System.Collections.IEnumerator System.Collections.IEnumerable.GetEnumerator() { return GetEnumerator(); } - public IEnumerator GetEnumerator() - { - for (int i = beginPos; i < endPos; i++) - yield return tokens[i]; - } - public TokenRange SkipWhile(Func pred) - { - for (int e = beginPos; e < endPos; e++) - if (!pred(tokens[e])) - return new TokenRange(tokens, e, endPos); - return new TokenRange(tokens, endPos, endPos); - } - public TokenRange TakeWhile(Func pred) - { - for (int e = beginPos; e < endPos; e++) - if (!pred(tokens[e])) - return new TokenRange(tokens, beginPos, e); - return new TokenRange(tokens, beginPos, endPos); - } - public TokenRange RevTakeWhile(Func pred) - { - for (int e = endPos-1; e >= beginPos; e--) - if (!pred(tokens[e])) - return new TokenRange(tokens, e+1, endPos); - return new TokenRange(tokens, beginPos, endPos); - } - public TokenRange RevSkipWhile(Func pred) - { - for (int e = endPos - 1; e >= beginPos; e--) - if (!pred(tokens[e])) - return new TokenRange(tokens, beginPos, e + 1); - return new TokenRange(tokens, beginPos, beginPos); - } - public Token[] GetAllTokens() { return tokens; } - - public int Length { - get { - return endPos - beginPos; - } - } - - Token[] tokens; - int beginPos; - int endPos; - }; - - static class BracketParser - { - public static IEnumerable NotInsideBrackets(IEnumerable tokens) - { - int BracketDepth = 0; - int? BasePD = null; - foreach (var tok in tokens) - { - if (BasePD == null) BasePD = tok.ParenDepth; - if (tok.ParenDepth == BasePD && tok.Value == "]") BracketDepth--; - if (BracketDepth == 0) - yield return tok; - if (tok.ParenDepth == BasePD && tok.Value == "[") BracketDepth++; - } - } - }; - static class AngleBracketParser - { - public static IEnumerable NotInsideAngleBrackets(IEnumerable tokens) - { - int AngleDepth = 0; - int? BasePD = null; - foreach (var tok in tokens) - { - if (BasePD == null) BasePD = tok.ParenDepth; - if (tok.ParenDepth == BasePD && tok.Value == ">") AngleDepth--; - if (AngleDepth == 0) - yield return tok; - if (tok.ParenDepth == BasePD && tok.Value == "<") AngleDepth++; - } - } - }; - - class ActorParser - { - public bool LineNumbersEnabled = true; - - Token[] tokens; - string sourceFile; - ErrorMessagePolicy errorMessagePolicy; - public bool generateProbes; - public Dictionary<(ulong, ulong), string> uidObjects { get; private set; } - - public ActorParser(string text, string sourceFile, ErrorMessagePolicy errorMessagePolicy, bool generateProbes) - { - this.sourceFile = sourceFile; - this.errorMessagePolicy = errorMessagePolicy; - this.generateProbes = generateProbes; - this.uidObjects = new Dictionary<(ulong, ulong), string>(); - tokens = Tokenize(text).Select(t=>new Token{ Value=t }).ToArray(); - CountParens(); - //if (sourceFile.EndsWith(".h")) LineNumbersEnabled = false; - //Console.WriteLine("{0} chars -> {1} tokens", text.Length, tokens.Length); - //showTokens(); - } - - class ClassContext { - public string name; - public int inBlocks; - } - - private bool ParseClassContext(TokenRange toks, out string name) - { - name = ""; - if (toks.Begin == toks.End) - { - return false; - } - - // http://nongnu.org/hcb/#attribute-specifier-seq - Token first; - while (true) - { - first = toks.First(NonWhitespace); - if (first.Value == "[") - { - var contents = first.GetMatchingRangeIn(toks); - toks = range(contents.End + 1, toks.End); - } - else if (first.Value == "alignas") - { - toks = range(first.Position + 1, toks.End); - first = toks.First(NonWhitespace); - first.Assert("Expected ( after alignas", t => t.Value == "("); - var contents = first.GetMatchingRangeIn(toks); - toks = range(contents.End + 1, toks.End); - } - else - { - break; - } - } - - // http://nongnu.org/hcb/#class-head-name - first = toks.First(NonWhitespace); - if (!identifierPattern.Match(first.Value).Success) { - return false; - } - while (true) { - first.Assert("Expected identifier", t=>identifierPattern.Match(t.Value).Success); - name += first.Value; - toks = range(first.Position + 1, toks.End); - if (toks.First(NonWhitespace).Value == "::") { - name += "::"; - toks = toks.SkipWhile(Whitespace).Skip(1); - } else { - break; - } - first = toks.First(NonWhitespace); - } - // http://nongnu.org/hcb/#class-virt-specifier-seq - toks = toks.SkipWhile(t => Whitespace(t) || t.Value == "final" || t.Value == "explicit"); - - first = toks.First(NonWhitespace); - if (first.Value == ":" || first.Value == "{") { - // At this point we've confirmed that this is a class. - return true; - } - return false; - } - - public void Write(System.IO.TextWriter writer, string destFileName) - { - writer.NewLine = "\n"; - writer.WriteLine("#define POST_ACTOR_COMPILER 1"); - int outLine = 1; - if (LineNumbersEnabled) - { - writer.WriteLine("#line {0} \"{1}\"", tokens[0].SourceLine, sourceFile); - outLine++; - } - int inBlocks = 0; - Stack classContextStack = new Stack(); - for(int i=0; i 0) - { - actor.enclosingClass = String.Join("::", classContextStack.Reverse().Select(t => t.name)); - } - var actorWriter = new System.IO.StringWriter(); - actorWriter.NewLine = "\n"; - var actorCompiler = new ActorCompiler(actor, sourceFile, inBlocks == 0, LineNumbersEnabled, generateProbes); - actorCompiler.Write(actorWriter); - actorCompiler.uidObjects.ToList().ForEach(x => this.uidObjects.TryAdd(x.Key, x.Value)); - - string[] actorLines = actorWriter.ToString().Split('\n'); - - bool hasLineNumber = false; - bool hadLineNumber = true; - foreach (var line in actorLines) - { - if (LineNumbersEnabled) - { - bool isLineNumber = line.Contains("#line"); - if (isLineNumber) hadLineNumber = true; - if (!isLineNumber && !hasLineNumber && hadLineNumber) - { - writer.WriteLine("\t\t\t\t\t\t\t\t\t\t\t\t\t\t\t#line {0} \"{1}\"", outLine + 1, destFileName); - outLine++; - hadLineNumber = false; - } - hasLineNumber = isLineNumber; - } - writer.WriteLine(line.TrimEnd('\n','\r')); - outLine++; - } - - i = end; - if (i != tokens.Length && LineNumbersEnabled) - { - writer.WriteLine("#line {0} \"{1}\"", tokens[i].SourceLine, sourceFile); - outLine++; - } - } - else if (tokens[i].Value == "class" || tokens[i].Value == "struct" || tokens[i].Value == "union") - { - writer.Write(tokens[i].Value); - string name; - if (ParseClassContext(range(i+1, tokens.Length), out name)) - { - classContextStack.Push(new ClassContext { name = name, inBlocks = inBlocks}); - } - } - else - { - if (tokens[i].Value == "{") - { - inBlocks++; - } - else if (tokens[i].Value == "}") - { - inBlocks--; - if (classContextStack.Count > 0 && classContextStack.Peek().inBlocks == inBlocks) - { - classContextStack.Pop(); - } - } - writer.Write(tokens[i].Value); - outLine += tokens[i].Value.Count(c => c == '\n'); - } - } - } - - IEnumerable SplitParameterList( TokenRange toks, string delimiter ) { - if (toks.Begin==toks.End) yield break; - while (true) { - Token comma = AngleBracketParser.NotInsideAngleBrackets( toks ) - .FirstOrDefault( t=> t.Value==delimiter && t.ParenDepth == toks.First().ParenDepth ); - if (comma == null) break; - yield return range(toks.Begin,comma.Position); - toks = range(comma.Position + 1, toks.End); - } - yield return toks; - } - - IEnumerable NormalizeWhitespace(IEnumerable tokens) - { - bool inWhitespace = false; - bool leading = true; - foreach (var tok in tokens) - { - if (!tok.IsWhitespace) - { - if (inWhitespace && !leading) yield return new Token { Value = " " }; - inWhitespace = false; - yield return tok; - leading = false; - } - else - { - inWhitespace = true; - } - } - } - - void ParseDeclaration(TokenRange tokens, - out Token name, - out TokenRange type, - out TokenRange initializer, - out bool constructorSyntax) - { - initializer = null; - TokenRange beforeInitializer = tokens; - constructorSyntax = false; - - Token equals = AngleBracketParser.NotInsideAngleBrackets(tokens) - .FirstOrDefault(t => t.Value == "=" && t.ParenDepth == tokens.First().ParenDepth); - if (equals != null) - { - // type name = initializer; - beforeInitializer = range(tokens.Begin,equals.Position); - initializer = range(equals.Position + 1, tokens.End); - } - else - { - Token paren = AngleBracketParser.NotInsideAngleBrackets(tokens) - .FirstOrDefault(t => t.Value == "("); - if (paren != null) - { - // type name(initializer); - constructorSyntax = true; - beforeInitializer = range(tokens.Begin, paren.Position); - initializer = - range(paren.Position + 1, tokens.End) - .TakeWhile(t => t.ParenDepth > paren.ParenDepth); - } else { - Token brace = AngleBracketParser.NotInsideAngleBrackets(tokens).FirstOrDefault(t => t.Value == "{"); - if (brace != null) { - // type name{initializer}; - throw new Error(brace.SourceLine, "Uniform initialization syntax is not currently supported for state variables (use '(' instead of '}}' ?)"); - } - } - } - name = beforeInitializer.Last(NonWhitespace); - if (beforeInitializer.Begin == name.Position) - throw new Error(beforeInitializer.First().SourceLine, "Declaration has no type."); - type = range(beforeInitializer.Begin, name.Position); - } - - VarDeclaration ParseVarDeclaration(TokenRange tokens) - { - Token name; - TokenRange type, initializer; - bool constructorSyntax; - ParseDeclaration( tokens, out name, out type, out initializer, out constructorSyntax ); - return new VarDeclaration - { - name = name.Value, - type = str(NormalizeWhitespace(type)), - initializer = initializer == null ? "" : str(NormalizeWhitespace(initializer)), - initializerConstructorSyntax = constructorSyntax - }; - } - - readonly Func Whitespace = (Token t) => t.IsWhitespace; - readonly Func NonWhitespace = (Token t) => !t.IsWhitespace; - - void ParseTestCaseHeading(Actor actor, TokenRange toks) - { - actor.isStatic = true; - - // The parameter(s) to the TEST_CASE macro are opaque to the actor compiler - TokenRange paramRange = toks.Last(NonWhitespace) - .Assert("Unexpected tokens after test case parameter list.", - t => t.Value == ")" && t.ParenDepth == toks.First().ParenDepth) - .GetMatchingRangeIn(toks); - actor.testCaseParameters = str(paramRange); - - actor.name = "flowTestCase" + toks.First().SourceLine; - actor.parameters = new VarDeclaration[] { new VarDeclaration { - name = "params", - type = "UnitTestParameters", - initializer = "", - initializerConstructorSyntax = false - } - }; - actor.returnType = "Void"; - } - - void ParseActorHeading(Actor actor, TokenRange toks) - { - var template = toks.First(NonWhitespace); - if (template.Value == "template") - { - var templateParams = range(template.Position+1, toks.End) - .First(NonWhitespace) - .Assert("Invalid template declaration", t=>t.Value=="<") - .GetMatchingRangeIn(toks); - - actor.templateFormals = SplitParameterList(templateParams, ",") - .Select(p => ParseVarDeclaration(p)) //< SOMEDAY: ? - .ToArray(); - - toks = range(templateParams.End + 1, toks.End); - } - var attribute = toks.First(NonWhitespace); - while (attribute.Value == "[") - { - var attributeContents = attribute.GetMatchingRangeIn(toks); - - var asArray = attributeContents.ToArray(); - if (asArray.Length < 2 || asArray[0].Value != "[" || asArray[asArray.Length - 1].Value != "]") - { - throw new Error(actor.SourceLine, "Invalid attribute: Expected [[...]]"); - } - actor.attributes.Add("[" + str(NormalizeWhitespace(attributeContents)) + "]"); - toks = range(attributeContents.End + 1, toks.End); - - attribute = toks.First(NonWhitespace); - } - - var staticKeyword = toks.First(NonWhitespace); - if (staticKeyword.Value == "static") - { - actor.isStatic = true; - toks = range(staticKeyword.Position + 1, toks.End); - } - - var uncancellableKeyword = toks.First(NonWhitespace); - if (uncancellableKeyword.Value == "UNCANCELLABLE") - { - actor.SetUncancellable(); - toks = range(uncancellableKeyword.Position + 1, toks.End); - } - - // Find the parameter list - TokenRange paramRange = toks.Last(NonWhitespace) - .Assert("Unexpected tokens after actor parameter list.", - t => t.Value == ")" && t.ParenDepth == toks.First().ParenDepth) - .GetMatchingRangeIn(toks); - actor.parameters = SplitParameterList(paramRange, ",") - .Select(p => ParseVarDeclaration(p)) - .ToArray(); - - var name = range(toks.Begin,paramRange.Begin-1).Last(NonWhitespace); - actor.name = name.Value; - - // SOMEDAY: refactor? - var returnType = range(toks.First().Position + 1, name.Position).SkipWhile(Whitespace); - var retToken = returnType.First(); - if (retToken.Value == "Future") - { - var ofType = returnType.Skip(1).First(NonWhitespace).Assert("Expected <", tok => tok.Value == "<").GetMatchingRangeIn(returnType); - actor.returnType = str(NormalizeWhitespace(ofType)); - toks = range(ofType.End + 1, returnType.End); - } - else if (retToken.Value == "void"/* && !returnType.Skip(1).Any(NonWhitespace)*/) - { - actor.returnType = null; - toks = returnType.Skip(1); - } - else - throw new Error(actor.SourceLine, "Actor apparently does not return Future"); - - toks = toks.SkipWhile(Whitespace); - if (!toks.IsEmpty) - { - if (toks.Last().Value == "::") - { - actor.nameSpace = str(range(toks.Begin, toks.End - 1)); - } - else - { - Console.WriteLine("Tokens: '{0}' {1} '{2}'", str(toks), toks.Count(), toks.Last().Value); - throw new Error(actor.SourceLine, "Unrecognized tokens preceding parameter list in actor declaration"); - } - } - if (errorMessagePolicy.ActorsNoDiscardByDefault() && !actor.attributes.Contains("[[flow_allow_discard]]")) { - if (actor.IsCancellable()) - { - actor.attributes.Add("[[nodiscard]]"); - } - } - HashSet knownFlowAttributes = new HashSet(); - knownFlowAttributes.Add("[[flow_allow_discard]]"); - foreach (var flowAttribute in actor.attributes.Where(a => a.StartsWith("[[flow_"))) { - if (!knownFlowAttributes.Contains(flowAttribute)) { - throw new Error(actor.SourceLine, "Unknown flow attribute {0}", flowAttribute); - } - } - actor.attributes = actor.attributes.Where(a => !a.StartsWith("[[flow_")).ToList(); - } - - LoopStatement ParseLoopStatement(TokenRange toks) - { - return new LoopStatement { - body = ParseCompoundStatement( toks.Consume("loop") ) - }; - } - - ChooseStatement ParseChooseStatement(TokenRange toks) - { - return new ChooseStatement - { - body = ParseCompoundStatement(toks.Consume("choose")) - }; - } - - WhenStatement ParseWhenStatement(TokenRange toks) - { - var expr = toks.Consume("when") - .SkipWhile(Whitespace) - .First() - .Assert("Expected (", t => t.Value == "(") - .GetMatchingRangeIn(toks) - .SkipWhile(Whitespace); - - return new WhenStatement { - wait = ParseWaitStatement(expr), - body = ParseCompoundStatement(range(expr.End+1, toks.End)) - }; - } - - StateDeclarationStatement ParseStateDeclaration(TokenRange toks) - { - toks = toks.Consume("state").RevSkipWhile(t => t.Value == ";"); - return new StateDeclarationStatement { - decl = ParseVarDeclaration(toks) - }; - } - - ReturnStatement ParseReturnStatement(TokenRange toks) - { - toks = toks.Consume("return").RevSkipWhile(t => t.Value == ";"); - return new ReturnStatement - { - expression = str(NormalizeWhitespace(toks)) - }; - } - - ThrowStatement ParseThrowStatement(TokenRange toks) - { - toks = toks.Consume("throw").RevSkipWhile(t => t.Value == ";"); - return new ThrowStatement - { - expression = str(NormalizeWhitespace(toks)) - }; - } - - WaitStatement ParseWaitStatement(TokenRange toks) - { - WaitStatement ws = new WaitStatement(); - ws.FirstSourceLine = toks.First().SourceLine; - if (toks.First().Value == "state") - { - ws.resultIsState = true; - toks = toks.Consume("state"); - } - TokenRange initializer; - if (toks.First().Value == "wait" || toks.First().Value == "waitNext") - { - initializer = toks.RevSkipWhile(t=>t.Value==";"); - ws.result = new VarDeclaration { - name = "_", - type = "Void", - initializer = "", - initializerConstructorSyntax = false - }; - } else { - Token name; - TokenRange type; - bool constructorSyntax; - ParseDeclaration( toks.RevSkipWhile(t=>t.Value==";"), out name, out type, out initializer, out constructorSyntax ); - - string typestring = str(NormalizeWhitespace(type)); - if (typestring == "Void") { - throw new Error(ws.FirstSourceLine, "Assigning the result of a Void wait is not allowed. Just use a standalone wait statement."); - } - - ws.result = new VarDeclaration - { - name = name.Value, - type = str(NormalizeWhitespace(type)), - initializer = "", - initializerConstructorSyntax = false - }; - } - - if (initializer == null) throw new Error(ws.FirstSourceLine, "Wait statement must be a declaration or standalone statement"); - - var waitParams = initializer - .SkipWhile(Whitespace).Consume("Statement contains a wait, but is not a valid wait statement or a supported compound statement.1", - t=> { - if (t.Value=="wait") return true; - if (t.Value=="waitNext") { ws.isWaitNext = true; return true; } - return false; - }) - .SkipWhile(Whitespace).First().Assert("Expected (", t => t.Value == "(") - .GetMatchingRangeIn(initializer); - if (!range(waitParams.End, initializer.End).Consume(")").All(Whitespace)) { - throw new Error(toks.First().SourceLine, "Statement contains a wait, but is not a valid wait statement or a supported compound statement.2"); - } - - ws.futureExpression = str(NormalizeWhitespace(waitParams)); - return ws; - } - - WhileStatement ParseWhileStatement(TokenRange toks) - { - var expr = toks.Consume("while") - .First(NonWhitespace) - .Assert("Expected (", t => t.Value == "(") - .GetMatchingRangeIn(toks); - return new WhileStatement - { - expression = str(NormalizeWhitespace(expr)), - body = ParseCompoundStatement(range(expr.End + 1, toks.End)) - }; - } - - Statement ParseForStatement(TokenRange toks) - { - var head = - toks.Consume("for") - .First(NonWhitespace) - .Assert("Expected (", t => t.Value == "(") - .GetMatchingRangeIn(toks); - - Token[] delim = - head.Where( - t => t.ParenDepth == head.First().ParenDepth && - t.BraceDepth == head.First().BraceDepth && - t.Value==";" - ).ToArray(); - if (delim.Length == 2) - { - var init = range(head.Begin, delim[0].Position); - var cond = range(delim[0].Position + 1, delim[1].Position); - var next = range(delim[1].Position + 1, head.End); - var body = range(head.End + 1, toks.End); - - return new ForStatement - { - initExpression = str(NormalizeWhitespace(init)), - condExpression = str(NormalizeWhitespace(cond)), - nextExpression = str(NormalizeWhitespace(next)), - body = ParseCompoundStatement(body) - }; - } - - delim = - head.Where( - t => t.ParenDepth == head.First().ParenDepth && - t.BraceDepth == head.First().BraceDepth && - t.Value == ":" - ).ToArray(); - if (delim.Length != 1) - { - throw new Error(head.First().SourceLine, "for statement must be 3-arg style or c++11 2-arg style"); - } - - return new RangeForStatement - { - // The container over which to iterate - rangeExpression = str(NormalizeWhitespace(range(delim[0].Position + 1, head.End).SkipWhile(Whitespace))), - // Type and name of the variable assigned in each iteration - rangeDecl = str(NormalizeWhitespace(range(head.Begin, delim[0].Position - 1).SkipWhile(Whitespace))), - // The body of the for loop - body = ParseCompoundStatement(range(head.End + 1, toks.End)) - }; - } - - Statement ParseIfStatement(TokenRange toks) - { - toks = toks.Consume("if"); - toks = toks.SkipWhile(Whitespace); - bool constexpr = toks.First().Value == "constexpr"; - if(constexpr) { - toks = toks.Consume("constexpr").SkipWhile(Whitespace); - } - - var expr = toks.First(NonWhitespace) - .Assert("Expected (", t => t.Value == "(") - .GetMatchingRangeIn(toks); - return new IfStatement { - expression = str(NormalizeWhitespace(expr)), - constexpr = constexpr, - ifBody = ParseCompoundStatement(range(expr.End+1, toks.End)) - // elseBody will be filled in later if necessary by ParseElseStatement - }; - } - void ParseElseStatement(TokenRange toks, Statement prevStatement) - { - var ifStatement = prevStatement as IfStatement; - while (ifStatement != null && ifStatement.elseBody != null) - ifStatement = ifStatement.elseBody as IfStatement; - if (ifStatement == null) - throw new Error(toks.First().SourceLine, "else without matching if"); - ifStatement.elseBody = ParseCompoundStatement(toks.Consume("else")); - } - - Statement ParseTryStatement(TokenRange toks) - { - return new TryStatement - { - tryBody = ParseCompoundStatement(toks.Consume("try")), - catches = new List() // will be filled in later by ParseCatchStatement - }; - } - void ParseCatchStatement(TokenRange toks, Statement prevStatement) - { - var tryStatement = prevStatement as TryStatement; - if (tryStatement == null) - throw new Error(toks.First().SourceLine, "catch without matching try"); - var expr = toks.Consume("catch") - .First(NonWhitespace) - .Assert("Expected (", t => t.Value == "(") - .GetMatchingRangeIn(toks); - tryStatement.catches.Add( - new TryStatement.Catch - { - expression = str(NormalizeWhitespace(expr)), - body = ParseCompoundStatement(range(expr.End + 1, toks.End)), - FirstSourceLine = expr.First().SourceLine - }); - } - - static readonly HashSet IllegalKeywords = new HashSet { "goto", "do", "finally", "__if_exists", "__if_not_exists" }; - - void ParseStatement(TokenRange toks, List statements) - { - toks = toks.SkipWhile(Whitespace); - - Action Add = stmt => - { - stmt.FirstSourceLine = toks.First().SourceLine; - statements.Add(stmt); - }; - - switch (toks.First().Value) - { - case "loop": Add(ParseLoopStatement(toks)); break; - case "while": Add(ParseWhileStatement(toks)); break; - case "for": Add(ParseForStatement(toks)); break; - case "break": Add(new BreakStatement()); break; - case "continue": Add(new ContinueStatement()); break; - case "return": Add(ParseReturnStatement(toks)); break; - case "{": Add(ParseCompoundStatement(toks)); break; - case "if": Add(ParseIfStatement(toks)); break; - case "else": ParseElseStatement(toks, statements[statements.Count - 1]); break; - case "choose": Add(ParseChooseStatement(toks)); break; - case "when": Add(ParseWhenStatement(toks)); break; - case "try": Add(ParseTryStatement(toks)); break; - case "catch": ParseCatchStatement(toks, statements[statements.Count - 1]); break; - case "throw": Add(ParseThrowStatement(toks)); break; - default: - if (IllegalKeywords.Contains(toks.First().Value)) - throw new Error(toks.First().SourceLine, "Statement '{0}' not supported in actors.", toks.First().Value); - if (toks.Any(t => t.Value == "wait" || t.Value == "waitNext")) - Add(ParseWaitStatement(toks)); - else if (toks.First().Value == "state") - Add(ParseStateDeclaration(toks)); - else if (toks.First().Value == "switch" && toks.Any(t => t.Value == "return")) - throw new Error(toks.First().SourceLine, "Unsupported compound statement containing return."); - else if (toks.First().Value.StartsWith("#")) - throw new Error(toks.First().SourceLine, "Found \"{0}\". Preprocessor directives are not supported within ACTORs", toks.First().Value); - else if (toks.RevSkipWhile(t => t.Value == ";").Any(NonWhitespace)) - Add(new PlainOldCodeStatement - { - code = str(NormalizeWhitespace(toks.RevSkipWhile(t => t.Value == ";"))) + ";" - }); - break; - }; - } - - Statement ParseCompoundStatement(TokenRange toks) - { - var first = toks.First(NonWhitespace); - if (first.Value == "{") { - var inBraces = first.GetMatchingRangeIn(toks); - if (!range(inBraces.End, toks.End).Consume("}").All(Whitespace)) - throw new Error(inBraces.Last().SourceLine, "Unexpected tokens after compound statement"); - return ParseCodeBlock(inBraces); - } else { - List statements = new List(); - ParseStatement( toks.Skip(1), statements ); - return statements[0]; - } - } - - CodeBlock ParseCodeBlock(TokenRange toks) - { - List statements = new List(); - while (true) - { - Token delim = toks - .FirstOrDefault( - t=> t.ParenDepth == toks.First().ParenDepth && - t.BraceDepth == toks.First().BraceDepth && - (t.Value==";" || t.Value == "}") - ); - if (delim == null) - break; - ParseStatement(range(toks.Begin, delim.Position + 1), statements); - toks = range(delim.Position + 1, toks.End); - } - if (!toks.All(Whitespace)) - throw new Error(toks.First(NonWhitespace).SourceLine, "Trailing unterminated statement in code block"); - return new CodeBlock { statements = statements.ToArray() }; - } - - TokenRange range(int beginPos, int endPos) - { - return new TokenRange(tokens, beginPos, endPos); - } - - Actor ParseActor( int pos, out int end ) { - var actor = new Actor(); - var head_token = tokens[pos]; - actor.SourceLine = head_token.SourceLine; - - var toks = range(pos+1, tokens.Length); - var heading = toks.TakeWhile(t => t.Value != "{"); - var toSemicolon = toks.TakeWhile(t => t.Value != ";"); - actor.isForwardDeclaration = toSemicolon.Length < heading.Length; - if (actor.isForwardDeclaration) { - heading = toSemicolon; - if (head_token.Value == "ACTOR" || head_token.Value == "SWIFT_ACTOR") { - ParseActorHeading(actor, heading); - } else { - head_token.Assert("ACTOR expected!", t => false); - } - end = heading.End + 1; - } else { - var body = range(heading.End+1, tokens.Length) - .TakeWhile(t => t.BraceDepth > toks.First().BraceDepth); - - if (head_token.Value == "ACTOR" || head_token.Value == "SWIFT_ACTOR") - { - ParseActorHeading(actor, heading); - } - else if (head_token.Value == "TEST_CASE") { - ParseTestCaseHeading(actor, heading); - actor.isTestCase = true; - } - else - head_token.Assert("ACTOR or TEST_CASE expected!", t => false); - - actor.body = ParseCodeBlock(body); - - if (!actor.body.containsWait()) - this.errorMessagePolicy.HandleActorWithoutWait(sourceFile, actor); - - end = body.End + 1; - } - return actor; - } - - string str(IEnumerable tokens) - { - return string.Join("", tokens.Select(x => x.Value).ToArray()); - } - string str(int begin, int end) - { - return str(range(begin,end)); - } - - void CountParens() - { - int BraceDepth = 0, ParenDepth = 0, LineCount = 1; - Token lastParen = null, lastBrace = null; - for (int i = 0; i < tokens.Length; i++) - { - switch (tokens[i].Value) - { - case "}": BraceDepth--; break; - case ")": ParenDepth--; break; - case "\r\n": LineCount++; break; - case "\n": LineCount++; break; - } - if (BraceDepth < 0) throw new Error(LineCount, "Mismatched braces"); - if (ParenDepth < 0) throw new Error(LineCount, "Mismatched parenthesis"); - tokens[i].Position = i; - tokens[i].SourceLine = LineCount; - tokens[i].BraceDepth = BraceDepth; - tokens[i].ParenDepth = ParenDepth; - if (tokens[i].Value.StartsWith("/*")) LineCount += tokens[i].Value.Count(c=>c=='\n'); - switch (tokens[i].Value) - { - case "{": BraceDepth++; if (BraceDepth==1) lastBrace = tokens[i]; break; - case "(": ParenDepth++; if (ParenDepth==1) lastParen = tokens[i]; break; - } - } - if (BraceDepth != 0) throw new Error(lastBrace.SourceLine, "Unmatched brace"); - if (ParenDepth != 0) throw new Error(lastParen.SourceLine, "Unmatched parenthesis"); - } - - void showTokens() - { - foreach (var t in tokens) - { - if (t.Value == "\r\n") - Console.WriteLine(); - else if (t.Value.Length == 1) - Console.Write(t.Value); - else - Console.Write("|{0}|", t.Value); - } - } - - readonly Regex identifierPattern = new Regex(@"\G[a-zA-Z_][a-zA-Z_0-9]*", RegexOptions.Singleline); - - readonly Regex[] tokenExpressions = (new string[] { - @"\{", - @"\}", - @"\(", - @"\)", - @"\[", - @"\]", - @"//[^\n]*", - @"/[*]([*][^/]|[^*])*[*]/", - @"'(\\.|[^\'\n])*'", //< SOMEDAY: Not fully restrictive - @"""(\\.|[^\""\n])*""", - @"[a-zA-Z_][a-zA-Z_0-9]*", - @"\r\n", - @"\n", - @"::", - @":", - @"#[a-z]*", // Recognize preprocessor directives so that we can reject them - @".", - }).Select( x=>new Regex(@"\G"+x, RegexOptions.Singleline) ).ToArray(); - - IEnumerable Tokenize(string text) - { - int pos = 0; - while (pos < text.Length) - { - bool ok = false; - foreach (var re in tokenExpressions) - { - var m = re.Match(text, pos); - if (m.Success) - { - yield return m.Value; - pos += m.Value.Length; - ok = true; - break; - } - } - if (!ok) - throw new Exception( String.Format("Can't tokenize! {0}", pos)); - } - } - } -} diff --git a/flow/actorcompiler/ParseTree.cs b/flow/actorcompiler/ParseTree.cs deleted file mode 100644 index bcc4f137e23..00000000000 --- a/flow/actorcompiler/ParseTree.cs +++ /dev/null @@ -1,238 +0,0 @@ -/* - * ParseTree.cs - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -using System; -using System.Collections.Generic; -using System.Linq; -using System.Text; -using System.Text.RegularExpressions; - -namespace actorcompiler -{ - class VarDeclaration - { - public string type; - public string name; - public string initializer; - public bool initializerConstructorSyntax; - }; - - abstract class Statement - { - public int FirstSourceLine; - public virtual bool containsWait() - { - return false; - } - }; - class PlainOldCodeStatement : Statement - { - public string code; - public override string ToString() - { - return code; - } - }; - class StateDeclarationStatement : Statement - { - public VarDeclaration decl; - public override string ToString() - { - if (decl.initializerConstructorSyntax) - return string.Format("State {0} {1}({2});", decl.type, decl.name, decl.initializer); - else - return string.Format("State {0} {1} = {2};", decl.type, decl.name, decl.initializer); - } - }; - class WhileStatement : Statement - { - public string expression; - public Statement body; - public override bool containsWait() - { - return body.containsWait(); - } - }; - class ForStatement : Statement - { - public string initExpression = ""; - public string condExpression = ""; - public string nextExpression = ""; - public Statement body; - public override bool containsWait() - { - return body.containsWait(); - } - }; - class RangeForStatement : Statement - { - public string rangeExpression; - public string rangeDecl; - public Statement body; - public override bool containsWait() - { - return body.containsWait(); - } - }; - class LoopStatement : Statement - { - public Statement body; - public override string ToString() - { - return "Loop " + body.ToString(); - } - public override bool containsWait() - { - return body.containsWait(); - } - }; - class BreakStatement : Statement - { - }; - class ContinueStatement : Statement - { - }; - class IfStatement : Statement - { - public string expression; - public bool constexpr; - public Statement ifBody; - public Statement elseBody; // might be null - public override bool containsWait() - { - return ifBody.containsWait() || (elseBody != null && elseBody.containsWait()); - } - }; - class ReturnStatement : Statement - { - public string expression; - public override string ToString() - { - return "Return " + expression; - } - }; - class WaitStatement : Statement - { - public VarDeclaration result; - public string futureExpression; - public bool resultIsState; - public bool isWaitNext; - public override string ToString() - { - return string.Format("Wait {0} {1} <- {2} ({3})", result.type, result.name, futureExpression, resultIsState ? "state" : "local"); - } - public override bool containsWait() - { - return true; - } - }; - class ChooseStatement : Statement - { - public Statement body; - public override string ToString() - { - return "Choose " + body.ToString(); - } - public override bool containsWait() - { - return body.containsWait(); - } - }; - class WhenStatement : Statement - { - public WaitStatement wait; - public Statement body; - public override string ToString() - { - return string.Format("When ({0}) {1}", wait, body); - } - public override bool containsWait() - { - return true; - } - }; - class TryStatement : Statement - { - public struct Catch - { - public string expression; - public Statement body; - public int FirstSourceLine; - }; - - public Statement tryBody; - public List catches; - public override bool containsWait() - { - if (tryBody.containsWait()) - return true; - foreach (Catch c in catches) - if (c.body.containsWait()) - return true; - return false; - } - }; - class ThrowStatement : Statement - { - public string expression; - }; - - class CodeBlock : Statement - { - public Statement[] statements; - public override string ToString() - { - return string.Join("\n", - new string[] { "CodeBlock" } - .Concat(statements.Select(s => s.ToString())) - .Concat(new string[] { "EndCodeBlock" }) - .ToArray()); - } - public override bool containsWait() - { - foreach (Statement s in statements) - if (s.containsWait()) - return true; - return false; - } - }; - - - class Actor - { - public List attributes = new List(); - public string returnType; - public string name; - public string enclosingClass = null; - public VarDeclaration[] parameters; - public VarDeclaration[] templateFormals; //< null if not a template - public CodeBlock body; - public int SourceLine; - public bool isStatic = false; - private bool isUncancellable; - public string testCaseParameters = null; - public string nameSpace = null; - public bool isForwardDeclaration = false; - public bool isTestCase = false; - - public bool IsCancellable() { return returnType != null && !isUncancellable; } - public void SetUncancellable() { isUncancellable = true; } - }; -}; diff --git a/flow/actorcompiler/Program.cs b/flow/actorcompiler/Program.cs deleted file mode 100644 index fc49544254b..00000000000 --- a/flow/actorcompiler/Program.cs +++ /dev/null @@ -1,104 +0,0 @@ -/* - * Program.cs - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -using System; -using System.IO; - -namespace actorcompiler -{ - class Program - { - private static void OverwriteByMove(string target, string temporaryFile) { - if (File.Exists(target)) - { - File.SetAttributes(target, FileAttributes.Normal); - File.Delete(target); - } - File.Move(temporaryFile, target); - File.SetAttributes(target, FileAttributes.ReadOnly); - } - - public static int Main(string[] args) - { - bool generateProbes = false; - if (args.Length < 2) - { - Console.WriteLine("Usage:"); - Console.WriteLine(" actorcompiler [--disable-diagnostics] [--generate-probes]"); - return 100; - } - Console.WriteLine("actorcompiler {0}", string.Join(" ", args)); - string input = args[0], output = args[1], outputtmp = args[1] + ".tmp", outputUid = args[1] + ".uid"; - ErrorMessagePolicy errorMessagePolicy = new ErrorMessagePolicy(); - foreach (var arg in args) { - if (arg.StartsWith("--")) { - if (arg.Equals("--disable-diagnostics")) { - errorMessagePolicy.DisableDiagnostics = true; - } else if (arg.Equals("--generate-probes")) { - generateProbes = true; - } - } - } - try - { - var inputData = File.ReadAllText(input); - var parser = new ActorParser(inputData, input.Replace('\\', '/'), errorMessagePolicy, generateProbes); - - using (var outputStream = new StreamWriter(outputtmp)) { - parser.Write(outputStream, output.Replace('\\', '/')); - } - OverwriteByMove(output, outputtmp); - - using (var outputStream = new StreamWriter(outputtmp)) { - foreach(var entry in parser.uidObjects) { - outputStream.WriteLine("{0}|{1}|{2}", entry.Key.Item1, entry.Key.Item2, entry.Value); - } - } - OverwriteByMove(outputUid, outputtmp); - - return 0; - } - catch (actorcompiler.Error e) - { - Console.Error.WriteLine("{0}({1}): error FAC1000: {2}", input, e.SourceLine, e.Message); - if (File.Exists(outputtmp)) - File.Delete(outputtmp); - if (File.Exists(output)) - { - File.SetAttributes(output, FileAttributes.Normal); - File.Delete(output); - } - return 1; - } - catch (Exception e) - { - Console.Error.WriteLine("{0}({1}): error FAC2000: Internal {2}", input, 1, e.ToString()); - if (File.Exists(outputtmp)) - File.Delete(outputtmp); - if (File.Exists(output)) - { - File.SetAttributes(output, FileAttributes.Normal); - File.Delete(output); - } - return 3; - } - } - } -} diff --git a/flow/actorcompiler/Properties/AssemblyInfo.cs b/flow/actorcompiler/Properties/AssemblyInfo.cs deleted file mode 100644 index 7548c19016d..00000000000 --- a/flow/actorcompiler/Properties/AssemblyInfo.cs +++ /dev/null @@ -1,56 +0,0 @@ -/* - * AssemblyInfo.cs - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -using System.Reflection; -using System.Runtime.CompilerServices; -using System.Runtime.InteropServices; - -// General Information about an assembly is controlled through the following -// set of attributes. Change these attribute values to modify the information -// associated with an assembly. -[assembly: AssemblyTitle("actorcompiler")] -[assembly: AssemblyDescription("Compile Flow code to C++")] -[assembly: AssemblyConfiguration("")] -[assembly: AssemblyCompany("Apple Inc")] -[assembly: AssemblyProduct("actorcompiler")] -[assembly: AssemblyCopyright("Copyright (c) 2013-2025 Apple Inc.")] -[assembly: AssemblyTrademark("")] -[assembly: AssemblyCulture("")] - -// Setting ComVisible to false makes the types in this assembly not visible -// to COM components. If you need to access a type in this assembly from -// COM, set the ComVisible attribute to true on that type. -[assembly: ComVisible(false)] - -// The following GUID is for the ID of the typelib if this project is exposed to COM -[assembly: Guid("fd8d34c8-46ca-43de-bfb3-b437e46943da")] - -// Version information for an assembly consists of the following four values: -// -// Major Version -// Minor Version -// Build Number -// Revision -// -// You can specify all the values or you can default the Build and Revision Numbers -// by using the '*' as shown below: -// [assembly: AssemblyVersion("1.0.*")] -[assembly: AssemblyVersion("1.0.0.0")] -[assembly: AssemblyFileVersion("1.0.0.0")] diff --git a/flow/actorcompiler/actorcompiler.csproj b/flow/actorcompiler/actorcompiler.csproj deleted file mode 100644 index ae866447097..00000000000 --- a/flow/actorcompiler/actorcompiler.csproj +++ /dev/null @@ -1,10 +0,0 @@ - - - - Exe - net9.0 - false - true - - - diff --git a/flow/actorcompiler/usertype.dat b/flow/actorcompiler/usertype.dat deleted file mode 100644 index 613d82fe794..00000000000 --- a/flow/actorcompiler/usertype.dat +++ /dev/null @@ -1,7 +0,0 @@ -ACTOR -loop -state -choose -when -wait -waitNext diff --git a/flow/actorcompiler_py/Actor checklist.txt b/flow/actorcompiler_py/Actor checklist.txt deleted file mode 100644 index 82ed087cefc..00000000000 --- a/flow/actorcompiler_py/Actor checklist.txt +++ /dev/null @@ -1,84 +0,0 @@ -Compile issues: - -- wait() must always assign the resulting value to a newly declared variable. - -- Variables used across a "wait() boundary" must be declared state - - -Remember to: - -- Add useful ASSERT()s - -- Add BUGGIFY() statements to expose rare cases to simulation - -- Add TEST() statements to any conditional/rare cases - -- Comment invariants, strategy, preconditions, tricky stuff when you - figure it out, even if it's not "your" code. - -- Factor common asynchronous control flows to use composition - of generic actors such as: - waitForAll - timeout - splitFuture - recurring - smartQuorum - AsyncMap - broadcast - &&, || - etc... - -- Declare classes NonCopyable unless they are, and you know what - that means - - -Run time issues: - -- Is the actor return type "void"? Make sure that some exception or timeout - will trigger eventually to clean up the actor. - -- If you send a future to another future, a long-lived forwardPromise - actor is created--make sure that the event happens eventually to free - this actor. - -- If you return a Future instead of a T from an actor, a forwardPromise - actor is created with the same lifetime issues as above. - -- Remember that parameters are internally passed to the actor as const & - and then copied into actor state variables - -- When you use *GetReply() or LoadBalance(), the "server" responding to - your request might get your request multiple times. - -- When you use getReply() instead of tryGetReply() you must ensure that the - actor will be cancelled if the service you are trying to connect to is - no longer available. (Otherwise, an infinite waiting loop) - -- For each wait: - - - An actor_cancelled exception can be thrown if the actor's return - value future is dropped. - - - An exception can arrive instead of a value - - - What happens if it never returns? - - - If the client fulfilling the wait is coming over the network, you - might get the same request multiple times - - -Performance issues: - -- Wait a little extra time before doing something time-consuming or - irreversible to see if it is still necessary. - -- When waiting for a number of things, wait a little extra time to get - the stragglers. (See the SmartQuorum() generic actor) - -- If asking another asynchronous server to do units of work, don't queue up more - work than is necessary to keep the server busy. Likewise, if you are - busy, let your own work queue fill up to signal your requester - that you are blocked. Also do this personally with managers assigning - you stuff. - -- Pass all variables as "const &" if their size is greater than 8 bytes. diff --git a/flow/actorcompiler_py/__main__.py b/flow/actorcompiler_py/__main__.py deleted file mode 100644 index b6a03591a27..00000000000 --- a/flow/actorcompiler_py/__main__.py +++ /dev/null @@ -1,101 +0,0 @@ -""" -Allow running the actorcompiler as a module: - python3 -m flow.actorcompiler input.actor.cpp output.g.cpp -""" -from __future__ import annotations - -import argparse -import os -import stat -import sys -from pathlib import Path - -from .actor_parser import ActorParser, ErrorMessagePolicy -from .errors import ActorCompilerError - - -def overwrite_by_move(target: Path, temporary: Path) -> None: - if target.exists(): - target.chmod(stat.S_IWUSR | stat.S_IRUSR | stat.S_IRGRP | stat.S_IROTH) - target.unlink() - os.replace(temporary, target) - target.chmod(stat.S_IRUSR | stat.S_IRGRP | stat.S_IROTH) - - -def parse_arguments() -> argparse.Namespace: - parser = argparse.ArgumentParser( - prog="actorcompiler", - description="Python port of the Flow actor compiler", - add_help=False, - usage="actorcompiler [--disable-diagnostics] [--generate-probes]", - ) - parser.add_argument("input", nargs="?") - parser.add_argument("output", nargs="?") - parser.add_argument("--disable-diagnostics", action="store_true") - parser.add_argument("--generate-probes", action="store_true") - parser.add_argument("--help", action="help", help=argparse.SUPPRESS) - args = parser.parse_args() - if not args.input or not args.output: - parser.print_usage(sys.stderr) - sys.exit(100) - return args - - -def main() -> int: - args = parse_arguments() - input_path = Path(args.input) - output_path = Path(args.output) - output_tmp = output_path.with_suffix(output_path.suffix + ".tmp") - output_uid = output_path.with_suffix(output_path.suffix + ".uid") - - policy = ErrorMessagePolicy() - policy.disable_diagnostics = args.disable_diagnostics - - try: - print("actorcompiler", " ".join(sys.argv[1:])) - text = input_path.read_text() - parser = ActorParser( - text, - str(input_path).replace("\\", "/"), - policy, - args.generate_probes, - ) - - with output_tmp.open("w", newline="\n") as out_file: - parser.write(out_file, str(output_path).replace("\\", "/")) - overwrite_by_move(output_path, output_tmp) - - with output_tmp.open("w", newline="\n") as uid_file: - for (hi, lo), value in parser.uid_objects.items(): - uid_file.write(f"{hi}|{lo}|{value}\n") - overwrite_by_move(output_uid, output_tmp) - - return 0 - except ActorCompilerError as exc: - print( - f"{input_path}({exc.source_line}): error FAC1000: {exc}", - file=sys.stderr, - ) - if output_tmp.exists(): - output_tmp.unlink() - if output_path.exists(): - output_path.chmod(stat.S_IWUSR | stat.S_IRUSR) - output_path.unlink() - return 1 - except Exception as exc: # pylint: disable=broad-except - import traceback - traceback.print_exc() - print( - f"{input_path}(1): error FAC2000: Internal {exc}", - file=sys.stderr, - ) - if output_tmp.exists(): - output_tmp.unlink() - if output_path.exists(): - output_path.chmod(stat.S_IWUSR | stat.S_IRUSR) - output_path.unlink() - return 3 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/flow/actorcompiler_py/actor_compiler.py b/flow/actorcompiler_py/actor_compiler.py deleted file mode 100644 index 67964b115b6..00000000000 --- a/flow/actorcompiler_py/actor_compiler.py +++ /dev/null @@ -1,1517 +0,0 @@ -from __future__ import annotations - -import hashlib -import io -from dataclasses import dataclass -from typing import Dict, Iterable, List, Optional, Sequence, Tuple - -from .errors import ActorCompilerError -from .parse_tree import ( - Actor, - BreakStatement, - ChooseStatement, - CodeBlock, - ContinueStatement, - ForStatement, - IfStatement, - LoopStatement, - PlainOldCodeStatement, - RangeForStatement, - ReturnStatement, - Statement, - StateDeclarationStatement, - ThrowStatement, - TryStatement, - VarDeclaration, - WaitStatement, - WhenStatement, - WhileStatement, -) - - -class Function: - def __init__( - self, - name: str = "", - return_type: str = "int", - formal_parameters: Optional[Sequence[str]] = None, - ) -> None: - self.name = name - self.return_type = return_type - self.formal_parameters = list(formal_parameters or []) - self.end_is_unreachable = False - self.exception_parameter_is: Optional[str] = None - self.public_name = False - self.specifiers = "" - self._indentation = "" - self._body = io.StringIO() - self.was_called = False - self.overload: Optional["Function"] = None - - def set_overload(self, overload: "Function") -> None: - self.overload = overload - - def pop_overload(self) -> Optional["Function"]: - overload = self.overload - self.overload = None - return overload - - def add_overload(self, *formal_parameters: str) -> None: - overload = Function( - name=self.name, - return_type=self.return_type, - formal_parameters=formal_parameters, - ) - overload.end_is_unreachable = self.end_is_unreachable - overload._indentation = self._indentation - self.set_overload(overload) - - def indent(self, change: int) -> None: - if change > 0: - self._indentation += "\t" * change - elif change < 0: - self._indentation = self._indentation[:change] - if self.overload is not None: - self.overload.indent(change) - - def write_line_unindented(self, text: str) -> None: - self._body.write(f"{text}\n") - if self.overload is not None: - self.overload.write_line_unindented(text) - - def write_line(self, text: str, *args: object) -> None: - if args: - text = text.format(*args) - self._body.write(f"{self._indentation}{text}\n") - if self.overload is not None: - self.overload.write_line(text) - - @property - def body_text(self) -> str: - value = self._body.getvalue() - if self.overload is not None: - _ = self.overload.body_text - return value - - def use_by_name(self) -> str: - self.was_called = True - return self.name if self.public_name else f"a_{self.name}" - - def call(self, *parameters: str) -> str: - params = ", ".join(parameters) - return f"{self.use_by_name()}({params})" - - -class LiteralBreak(Function): - def __init__(self) -> None: - super().__init__(name="break!", return_type="") - - def call(self, *parameters: str) -> str: - if parameters: - raise ActorCompilerError(0, "LiteralBreak called with parameters!") - self.was_called = True - return "break" - - -class LiteralContinue(Function): - def __init__(self) -> None: - super().__init__(name="continue!", return_type="") - - def call(self, *parameters: str) -> str: - if parameters: - raise ActorCompilerError(0, "LiteralContinue called with parameters!") - self.was_called = True - return "continue" - - -@dataclass -class StateVar: - type: str = "" - name: str = "" - initializer: Optional[str] = None - initializer_constructor_syntax: bool = False - source_line: int = 0 - - -@dataclass -class CallbackVar: - type: str = "" - callback_group: int = 0 - source_line: int = 0 - - -class Context: - def __init__( - self, - target: Optional[Function] = None, - next_func: Optional[Function] = None, - break_f: Optional[Function] = None, - continue_f: Optional[Function] = None, - catch_f_err: Optional[Function] = None, - try_loop_depth: int = 0, - ) -> None: - self.target = target - self.next = next_func - self.break_f = break_f - self.continue_f = continue_f - self.catch_f_err = catch_f_err - self.try_loop_depth = try_loop_depth - - def unreachable(self) -> None: - self.target = None - - def with_target(self, new_target: Function) -> "Context": - return Context( - target=new_target, - break_f=self.break_f, - continue_f=self.continue_f, - catch_f_err=self.catch_f_err, - try_loop_depth=self.try_loop_depth, - ) - - def loop_context( - self, - new_target: Function, - break_f: Function, - continue_f: Function, - deltaLoopDepth: int, - ) -> "Context": - return Context( - target=new_target, - break_f=break_f, - continue_f=continue_f, - catch_f_err=self.catch_f_err, - try_loop_depth=self.try_loop_depth + deltaLoopDepth, - ) - - def with_catch(self, new_catch: Function) -> "Context": - return Context( - target=self.target, - break_f=self.break_f, - continue_f=self.continue_f, - catch_f_err=new_catch, - ) - - def clone(self) -> "Context": - return Context( - target=self.target, - next_func=self.next, - break_f=self.break_f, - continue_f=self.continue_f, - catch_f_err=self.catch_f_err, - try_loop_depth=self.try_loop_depth, - ) - - def copy_from(self, other: "Context") -> None: - self.target = other.target - self.next = other.next - self.break_f = other.break_f - self.continue_f = other.continue_f - self.catch_f_err = other.catch_f_err - self.try_loop_depth = other.try_loop_depth - - -class ActorCompiler: - member_indent_str = "\t" - loop_depth_0 = "int loopDepth=0" - loop_depth = "int loopDepth" - code_indent = 2 - used_class_names: set[str] = set() - - def __init__( - self, - actor: Actor, - source_file: str, - is_top_level: bool, - line_numbers_enabled: bool, - generate_probes: bool, - ) -> None: - self.actor = actor - self.source_file = source_file - self.is_top_level = is_top_level - self.line_numbers_enabled = line_numbers_enabled - self.generate_probes = generate_probes - self.class_name = "" - self.full_class_name = "" - self.state_class_name = "" - self.state: List[StateVar] = [] - self.callbacks: List[CallbackVar] = [] - self.choose_groups = 0 - self.when_count = 0 - self.this = "" - self.uid_objects: Dict[Tuple[int, int], str] = {} - self.functions: Dict[str, Function] = {} - self.iterators: Dict[str, int] = {} - self.find_state() - - def byte_to_long(self, data: bytes) -> int: - result = 0 - for b in data: - result += b - result <<= 8 - return result & ((1 << 64) - 1) - - def get_uid_from_string(self, value: str) -> Tuple[int, int]: - digest = hashlib.sha256(value.encode("utf-8")).digest() - first = self.byte_to_long(digest[:8]) - second = self.byte_to_long(digest[8:16]) - return (first, second) - - def writeActorFunction(self, writer, full_return_type: str) -> None: - self.writeTemplate(writer) - self.line_number(writer, self.actor.source_line) - for attribute in self.actor.attributes: - writer.write(f"{attribute} ") - if self.actor.is_static: - writer.write("static ") - namespace_prefix = ( - "" if self.actor.name_space is None else f"{self.actor.name_space}::" - ) - params = ", ".join(self.parameter_list()) - writer.write( - f"{full_return_type} {namespace_prefix}{self.actor.name}( {params} ) {{\n" - ) - self.line_number(writer, self.actor.source_line) - ctor_args = ", ".join(param.name for param in self.actor.parameters) - new_actor = f"new {self.full_class_name}({ctor_args})" - if self.actor.return_type is not None: - writer.write(f"\treturn Future<{self.actor.return_type}>({new_actor});\n") - else: - writer.write(f"\t{new_actor};\n") - writer.write("}\n") - - def writeActorClass( - self, writer, full_state_class_name: str, body: Function - ) -> None: - writer.write( - f"// This generated class is to be used only via {self.actor.name}()\n" - ) - self.writeTemplate(writer) - self.line_number(writer, self.actor.source_line) - callback_bases = ", ".join(f"public {cb.type}" for cb in self.callbacks) - if callback_bases: - callback_bases += ", " - writer.write( - "class {0} final : public Actor<{2}>, {3}public FastAllocated<{1}>, public {4} {{\n".format( - self.class_name, - self.full_class_name, - "void" if self.actor.return_type is None else self.actor.return_type, - callback_bases, - full_state_class_name, - ) - ) - writer.write("public:\n") - writer.write(f"\tusing FastAllocated<{self.full_class_name}>::operator new;\n") - writer.write(f"\tusing FastAllocated<{self.full_class_name}>::operator delete;\n") - actor_identifier_key = f"{self.source_file}:{self.actor.name}" - uid = self.get_uid_from_string(actor_identifier_key) - self.uid_objects[(uid[0], uid[1])] = actor_identifier_key - writer.write( - f"\tstatic constexpr ActorIdentifier __actorIdentifier = UID({uid[0]}UL, {uid[1]}UL);\n" - ) - writer.write("\tActiveActorHelper activeActorHelper;\n") - writer.write("#pragma clang diagnostic push\n") - writer.write('#pragma clang diagnostic ignored "-Wdelete-non-virtual-dtor"\n') - if self.actor.return_type is not None: - writer.write(" void destroy() override {\n") - writer.write(" activeActorHelper.~ActiveActorHelper();\n") - writer.write( - f" static_cast*>(this)->~Actor();\n" - ) - writer.write(" operator delete(this);\n") - writer.write(" }\n") - else: - writer.write(" void destroy() {{\n") - writer.write(" activeActorHelper.~ActiveActorHelper();\n") - writer.write(" static_cast*>(this)->~Actor();\n") - writer.write(" operator delete(this);\n") - writer.write(" }}\n") - writer.write("#pragma clang diagnostic pop\n") - for cb in self.callbacks: - writer.write(f"friend struct {cb.type};\n") - self.line_number(writer, self.actor.source_line) - self.writeConstructor(body, writer, full_state_class_name) - self.writeCancelFunc(writer) - writer.write("};\n") - - def write(self, writer) -> None: - full_return_type = ( - f"Future<{self.actor.return_type}>" - if self.actor.return_type is not None - else "void" - ) - for i in range(1 << 16): - class_name = "{3}{0}{1}Actor{2}".format( - self.actor.name[:1].upper(), - self.actor.name[1:], - str(i) if i != 0 else "", - ( - self.actor.enclosing_class.replace("::", "_") + "_" - if self.actor.enclosing_class is not None - and self.actor.is_forward_declaration - else ( - self.actor.name_space.replace("::", "_") + "_" - if self.actor.name_space is not None - else "" - ) - ), - ) - if self.actor.is_forward_declaration: - self.class_name = class_name - break - if class_name not in ActorCompiler.used_class_names: - self.class_name = class_name - ActorCompiler.used_class_names.add(class_name) - break - self.full_class_name = self.class_name + self.get_template_actuals() - actor_class_formal = VarDeclaration(name=self.class_name, type="class") - self.this = f"static_cast<{actor_class_formal.name}*>(this)" - self.state_class_name = self.class_name + "State" - full_state = self.state_class_name + self.get_template_actuals( - VarDeclaration(type="class", name=self.full_class_name) - ) - if self.actor.is_forward_declaration: - for attribute in self.actor.attributes: - writer.write(f"{attribute} ") - if self.actor.is_static: - writer.write("static ") - ns = "" if self.actor.name_space is None else f"{self.actor.name_space}::" - writer.write( - f"{full_return_type} {ns}{self.actor.name}( {', '.join(self.parameter_list())} );\n" - ) - if self.actor.enclosing_class is not None: - writer.write(f"template friend class {self.state_class_name};\n") - return - body = self.get_function("", "body", self.loop_depth_0) - body_context = Context( - target=body, - catch_f_err=self.get_function( - body.name, "Catch", "Error error", self.loop_depth_0 - ), - ) - end_context = self.try_catch_compile(self.actor.body, body_context) - if end_context.target is not None: - if self.actor.return_type is None: - self.compile_statement( - ReturnStatement( - first_source_line=self.actor.source_line, expression="" - ), - end_context, - ) - else: - raise ActorCompilerError( - self.actor.source_line, - "Actor {0} fails to return a value", - self.actor.name, - ) - if self.actor.return_type is not None: - body_context.catch_f_err.write_line(f"this->~{self.state_class_name}();") - body_context.catch_f_err.write_line( - f"{self.this}->sendErrorAndDelPromiseRef(error);" - ) - else: - body_context.catch_f_err.write_line(f"delete {self.this};") - body_context.catch_f_err.write_line("loopDepth = 0;") - if self.is_top_level and self.actor.name_space is None: - writer.write("namespace {\n") - writer.write( - f"// This generated class is to be used only via {self.actor.name}()\n" - ) - self.writeTemplate(writer, actor_class_formal) - self.line_number(writer, self.actor.source_line) - writer.write(f"class {self.state_class_name} {{\n") - writer.write("public:\n") - self.line_number(writer, self.actor.source_line) - self.writeStateConstructor(writer) - self.writeStateDestructor(writer) - self.writeFunctions(writer) - for st in self.state: - self.line_number(writer, st.source_line) - writer.write(f"\t{st.type} {st.name};\n") - writer.write("};\n") - self.writeActorClass(writer, full_state, body) - if self.is_top_level and self.actor.name_space is None: - writer.write("} // namespace\n") - self.writeActorFunction(writer, full_return_type) - if self.actor.test_case_parameters is not None: - writer.write( - f"ACTOR_TEST_CASE({self.actor.name}, {self.actor.test_case_parameters})\n" - ) - - thisAddress = "reinterpret_cast(this)" - - def probe_enter(self, fun: Function, name: str, index: int = -1) -> None: - if self.generate_probes: - fun.write_line( - 'fdb_probe_actor_enter("{0}", {1}, {2});', - name, - self.thisAddress, - index, - ) - block_identifier = self.get_uid_from_string(fun.name) - fun.write_line("#ifdef WITH_ACAC") - fun.write_line( - "static constexpr ActorBlockIdentifier __identifier = UID({0}UL, {1}UL);", - block_identifier[0], - block_identifier[1], - ) - fun.write_line( - "ActorExecutionContextHelper __helper(static_cast<{0}*>(this)->activeActorHelper.actorID, __identifier);", - self.class_name, - ) - fun.write_line("#endif // WITH_ACAC") - - def probe_exit(self, fun: Function, name: str, index: int = -1) -> None: - if self.generate_probes: - fun.write_line( - 'fdb_probe_actor_exit("{0}", {1}, {2});', - name, - self.thisAddress, - index, - ) - - def probe_create(self, fun: Function, name: str) -> None: - if self.generate_probes: - fun.write_line( - 'fdb_probe_actor_create("{0}", {1});', name, self.thisAddress - ) - - def probe_destroy(self, fun: Function, name: str) -> None: - if self.generate_probes: - fun.write_line( - 'fdb_probe_actor_destroy("{0}", {1});', - name, - self.thisAddress, - ) - - def line_number(self, writer, source_line: int) -> None: - if source_line == 0: - raise ActorCompilerError(0, "Invalid source line (0)") - if self.line_numbers_enabled: - writer.write( - '\t\t\t\t\t\t\t\t\t\t\t\t\t\t\t#line {0} "{1}"\n'.format( - source_line, self.source_file - ) - ) - - def line_numberFunction(self, func: Function, source_line: int) -> None: - if source_line == 0: - raise ActorCompilerError(0, "Invalid source line (0)") - if self.line_numbers_enabled: - func.write_line_unindented( - '\t\t\t\t\t\t\t\t\t\t\t\t\t\t\t#line {0} "{1}"'.format( - source_line, self.source_file - ) - ) - - def try_catch( - self, - cx: Context, - catch_function: Optional[Function], - catch_loop_depth: int, - action, - use_loop_depth: bool = True, - ) -> None: - if catch_function is not None: - cx.target.write_line("try {") - cx.target.indent(+1) - action() - if catch_function is not None: - cx.target.indent(-1) - cx.target.write_line("}") - cx.target.write_line("catch (Error& error) {") - if use_loop_depth: - cx.target.write_line( - "\tloopDepth = {0};", - catch_function.call( - "error", self.adjust_loop_depth(catch_loop_depth) - ), - ) - else: - cx.target.write_line("\t{0};", catch_function.call("error", "0")) - cx.target.write_line("} catch (...) {") - if use_loop_depth: - cx.target.write_line( - "\tloopDepth = {0};", - catch_function.call( - "unknown_error()", self.adjust_loop_depth(catch_loop_depth) - ), - ) - else: - cx.target.write_line( - "\t{0};", catch_function.call("unknown_error()", "0") - ) - cx.target.write_line("}") - - def try_catch_compile(self, block: CodeBlock, cx: Context) -> Context: - result_holder = {"ctx": cx} - - def action() -> None: - result_holder["ctx"] = self._try_catch_compile_body(block, cx) - - self.try_catch(cx, cx.catch_f_err, cx.try_loop_depth, action) - return result_holder["ctx"] - - def _try_catch_compile_body(self, block: CodeBlock, cx: Context) -> Context: - compiled = self.compile(block, cx, True) - if compiled.target is not None: - next_func = self.get_function(compiled.target.name, "cont", self.loop_depth) - compiled.target.write_line("loopDepth = {0};", next_func.call("loopDepth")) - compiled.target = next_func - compiled.next = None - return compiled - - def writeTemplate(self, writer, *extra_parameters: VarDeclaration) -> None: - formals = list(self.actor.template_formals or []) + list(extra_parameters) - if not formals: - return - self.line_number(writer, self.actor.source_line) - writer.write( - "template <{0}>\n".format(", ".join(f"{p.type} {p.name}" for p in formals)) - ) - - def get_template_actuals(self, *extra_parameters: VarDeclaration) -> str: - formals = list(self.actor.template_formals or []) + list(extra_parameters) - if not formals: - return "" - return "<{0}>".format(", ".join(p.name for p in formals)) - - def will_continue(self, stmt: Statement) -> bool: - return any( - isinstance(sub, (ChooseStatement, WaitStatement, TryStatement)) - for sub in self.flatten(stmt) - ) - - def as_code_block(self, statement: Statement) -> CodeBlock: - if isinstance(statement, CodeBlock): - return statement - return CodeBlock(statements=[statement]) - - def flatten(self, stmt: Optional[Statement]) -> Iterable[Statement]: - if stmt is None: - return [] - - def _flatten(s: Statement) -> Iterable[Statement]: - yield s - if isinstance(s, LoopStatement): - yield from self.flatten(s.body) - elif isinstance(s, WhileStatement): - yield from self.flatten(s.body) - elif isinstance(s, ForStatement): - yield from self.flatten(s.body) - elif isinstance(s, RangeForStatement): - yield from self.flatten(s.body) - elif isinstance(s, CodeBlock): - for child in s.statements: - yield from self.flatten(child) - elif isinstance(s, IfStatement): - yield from self.flatten(s.if_body) - if s.else_body: - yield from self.flatten(s.else_body) - elif isinstance(s, ChooseStatement): - yield from self.flatten(s.body) - elif isinstance(s, WhenStatement): - if s.body: - yield from self.flatten(s.body) - elif isinstance(s, TryStatement): - yield from self.flatten(s.try_body) - for c in s.catches: - yield from self.flatten(c.body) - - return list(_flatten(stmt)) - - def find_state(self) -> None: - self.state = [ - StateVar( - source_line=self.actor.source_line, - name=p.name, - type=p.type, - initializer=p.name, - ) - for p in self.actor.parameters - ] - - def adjust_loop_depth(self, subtract: int) -> str: - if subtract == 0: - return "loopDepth" - return f"std::max(0, loopDepth - {subtract})" - - def parameter_list(self) -> List[str]: - params = [] - for p in self.actor.parameters: - if p.initializer: - params.append(f"{p.type} const& {p.name} = {p.initializer}") - else: - params.append(f"{p.type} const& {p.name}") - return params - - def get_function( - self, - base_name: str, - add_name: str, - *formal_parameters: str, - overload_formal_parameters: Optional[Sequence[str]] = None, - ) -> Function: - if len(formal_parameters) == 1 and isinstance( - formal_parameters[0], (list, tuple) - ): - params = list(formal_parameters[0]) - else: - params = [p for p in formal_parameters if p] - - if add_name == "cont" and len(base_name) >= 5 and base_name[-5:-1] == "cont": - proposed_name = base_name[:-1] - else: - proposed_name = base_name + add_name - idx = 1 - while f"{proposed_name}{idx}" in self.functions: - idx += 1 - func = Function( - name=f"{proposed_name}{idx}", - return_type="int", - formal_parameters=params, - ) - func.indent(self.code_indent) - if overload_formal_parameters: - func.add_overload(*overload_formal_parameters) - self.functions[func.name] = func - return func - - def writeFunctions(self, writer) -> None: - for func in self.functions.values(): - body = func.body_text - if body: - self.writeFunction(writer, func, body) - if func.overload: - overload_body = func.overload.body_text - if overload_body: - self.writeFunction(writer, func.overload, overload_body) - - def writeFunction(self, writer, func: Function, body: str) -> None: - spec = "" if not func.specifiers else f" {func.specifiers}" - trailing = " " if spec == "" else "" - signature = ( - f"{self.member_indent_str}" - f"{'' if func.return_type == '' else f'{func.return_type} '}" - f"{func.use_by_name()}({','.join(func.formal_parameters)})" - f"{spec}{trailing}\n" - ) - writer.write(signature) - if func.return_type != "": - writer.write(f"{self.member_indent_str}{{\n") - writer.write(body) - writer.write("\n") - if not func.end_is_unreachable: - writer.write(f"{self.member_indent_str}\treturn loopDepth;\n") - writer.write(f"{self.member_indent_str}}}\n") - - def writeCancelFunc(self, writer) -> None: - if not self.actor.is_cancellable(): - return - cancel_func = Function( - name="cancel", - return_type="void", - formal_parameters=[], - ) - cancel_func.end_is_unreachable = True - cancel_func.public_name = True - cancel_func.specifiers = "override" - cancel_func.indent(self.code_indent) - cancel_func.write_line("auto wait_state = this->actor_wait_state;") - cancel_func.write_line("this->actor_wait_state = ACTOR_WAIT_STATE_CANCELLED;") - cancel_func.write_line("switch (wait_state) {") - last_group = -1 - for cb in sorted(self.callbacks, key=lambda c: c.callback_group): - if cb.callback_group != last_group: - last_group = cb.callback_group - cancel_func.write_line( - "case {0}: this->a_callback_error(({1}*)0, actor_cancelled()); break;", - cb.callback_group, - cb.type, - ) - cancel_func.write_line("}") - self.writeFunction(writer, cancel_func, cancel_func.body_text) - - def writeConstructor( - self, body: Function, writer, full_state_class_name: str - ) -> None: - constructor = Function( - name=self.class_name, - return_type="", - formal_parameters=self.parameter_list(), - ) - constructor.end_is_unreachable = True - constructor.public_name = True - constructor.indent(self.code_indent) - constructor.write_line( - " : Actor<{0}>(),".format( - "void" if self.actor.return_type is None else self.actor.return_type - ) - ) - constructor.write_line( - " {0}({1}),".format( - full_state_class_name, - ", ".join(p.name for p in self.actor.parameters), - ) - ) - constructor.write_line(" activeActorHelper(__actorIdentifier)") - constructor.indent(-1) - constructor.write_line("{") - constructor.indent(+1) - self.probe_enter(constructor, self.actor.name) - constructor.write_line("#ifdef ENABLE_SAMPLING") - constructor.write_line('this->lineage.setActorName("{0}");', self.actor.name) - constructor.write_line("LineageScope _(&this->lineage);") - constructor.write_line("#endif") - constructor.write_line("this->{0};", body.call()) - self.probe_exit(constructor, self.actor.name) - self.writeFunction(writer, constructor, constructor.body_text) - - def writeStateConstructor(self, writer) -> None: - constructor = Function( - name=self.state_class_name, - return_type="", - formal_parameters=self.parameter_list(), - ) - constructor.end_is_unreachable = True - constructor.public_name = True - constructor.indent(self.code_indent) - ini = None - line = self.actor.source_line - for state_var in self.state: - if state_var.initializer is None: - continue - self.line_numberFunction(constructor, line) - if ini is None: - ini = " : " - else: - constructor.write_line(ini + ",") - ini = " " - ini += f"{state_var.name}({state_var.initializer})" - line = state_var.source_line - self.line_numberFunction(constructor, line) - if ini: - constructor.write_line(ini) - constructor.indent(-1) - constructor.write_line("{") - constructor.indent(+1) - self.probe_create(constructor, self.actor.name) - self.writeFunction(writer, constructor, constructor.body_text) - - def writeStateDestructor(self, writer) -> None: - destructor = Function( - name=f"~{self.state_class_name}", - return_type="", - formal_parameters=[], - ) - destructor.end_is_unreachable = True - destructor.public_name = True - destructor.indent(self.code_indent) - destructor.indent(-1) - destructor.write_line("{") - destructor.indent(+1) - self.probe_destroy(destructor, self.actor.name) - self.writeFunction(writer, destructor, destructor.body_text) - - def compile( - self, block: CodeBlock, context: Context, ok_to_continue: bool = True - ) -> Context: - cx = context.clone() - cx.next = None - for stmt in block.statements: - if cx.target is None: - raise ActorCompilerError(stmt.first_source_line, "Unreachable code.") - if cx.next is None: - cx.next = self.get_function(cx.target.name, "cont", self.loop_depth) - self.compile_statement(stmt, cx) - if cx.next.was_called: - if cx.target is None: - raise ActorCompilerError( - stmt.first_source_line, "Unreachable continuation called?" - ) - if not ok_to_continue: - raise ActorCompilerError( - stmt.first_source_line, "Unexpected continuation" - ) - cx.target = cx.next - cx.next = None - return cx - - def compile_statement(self, stmt: Statement, cx: Context) -> None: - handler_name = f"_compile_{stmt.__class__.__name__}" - handler = getattr(self, handler_name, None) - if handler is None: - raise ActorCompilerError( - stmt.first_source_line, - "Statement type {0} not supported yet.", - stmt.__class__.__name__, - ) - handler(stmt, cx) - - def _is_top_level_state_decl(self, stmt: StateDeclarationStatement) -> bool: - for candidate in self.actor.body.statements: - if not isinstance(candidate, StateDeclarationStatement): - break - if candidate is stmt: - return True - return False - - def _compile_PlainOldCodeStatement( - self, stmt: PlainOldCodeStatement, cx: Context - ) -> None: - self.line_numberFunction(cx.target, stmt.first_source_line) - cx.target.write_line(stmt.code) - - def _compile_StateDeclarationStatement( - self, stmt: StateDeclarationStatement, cx: Context - ) -> None: - if self._is_top_level_state_decl(stmt): - self.state.append( - StateVar( - source_line=stmt.first_source_line, - name=stmt.decl.name, - type=stmt.decl.type, - initializer=stmt.decl.initializer, - initializer_constructor_syntax=stmt.decl.initializer_constructor_syntax, - ) - ) - else: - self.state.append( - StateVar( - source_line=stmt.first_source_line, - name=stmt.decl.name, - type=stmt.decl.type, - initializer=None, - ) - ) - if stmt.decl.initializer is not None: - self.line_numberFunction(cx.target, stmt.first_source_line) - if ( - stmt.decl.initializer_constructor_syntax - or stmt.decl.initializer == "" - ): - cx.target.write_line( - "{0} = {1}({2});", - stmt.decl.name, - stmt.decl.type, - stmt.decl.initializer, - ) - else: - cx.target.write_line( - "{0} = {1};", stmt.decl.name, stmt.decl.initializer - ) - - def get_iterator_name(self, cx: Context) -> str: - name = f"RangeFor{cx.target.name}Iterator" - self.iterators.setdefault(name, 0) - idx = self.iterators[name] - self.iterators[name] += 1 - return f"{name}{idx}" - - def emit_native_loop( - self, source_line: int, head: str, body: Statement, cx: Context - ) -> bool: - self.line_numberFunction(cx.target, source_line) - cx.target.write_line(head + " {") - cx.target.indent(+1) - literal_break = LiteralBreak() - literal_continue = LiteralContinue() - self.compile( - self.as_code_block(body), - cx.loop_context(cx.target, literal_break, literal_continue, 0), - True, - ) - cx.target.indent(-1) - cx.target.write_line("}") - return not literal_break.was_called - - def _compile_ForStatement(self, stmt: ForStatement, cx: Context) -> None: - no_condition = stmt.cond_expression in ("", "true", "1") - if not self.will_continue(stmt.body): - if ( - self.emit_native_loop( - stmt.first_source_line, - f"for({stmt.init_expression};{stmt.cond_expression};{stmt.next_expression})", - stmt.body, - cx, - ) - and no_condition - ): - cx.unreachable() - return - - init_stmt = PlainOldCodeStatement( - code=f"{stmt.init_expression};", first_source_line=stmt.first_source_line - ) - self._compile_PlainOldCodeStatement(init_stmt, cx) - - if no_condition: - full_body = stmt.body - else: - condition = IfStatement( - expression=f"!({stmt.cond_expression})", - if_body=BreakStatement(first_source_line=stmt.first_source_line), - first_source_line=stmt.first_source_line, - ) - full_body = CodeBlock( - statements=[condition] + list(self.as_code_block(stmt.body).statements), - first_source_line=stmt.first_source_line, - ) - - loopF = self.get_function(cx.target.name, "loopHead", self.loop_depth) - loopBody = self.get_function(cx.target.name, "loopBody", self.loop_depth) - break_f = self.get_function(cx.target.name, "break", self.loop_depth) - continue_f = ( - loopF - if stmt.next_expression == "" - else self.get_function(cx.target.name, "continue", self.loop_depth) - ) - - loopF.write_line("int oldLoopDepth = ++loopDepth;") - loopF.write_line( - "while (loopDepth == oldLoopDepth) loopDepth = {0};", - loopBody.call("loopDepth"), - ) - - endLoop = self.compile( - self.as_code_block(full_body), - cx.loop_context(loopBody, break_f, continue_f, +1), - True, - ).target - - if endLoop is not None and endLoop is not loopBody: - if stmt.next_expression: - self._compile_PlainOldCodeStatement( - PlainOldCodeStatement( - code=f"{stmt.next_expression};", - first_source_line=stmt.first_source_line, - ), - cx.with_target(endLoop), - ) - endLoop.write_line("if (loopDepth == 0) return {0};", loopF.call("0")) - - cx.target.write_line("loopDepth = {0};", loopF.call("loopDepth")) - - if continue_f is not loopF and continue_f.was_called: - self._compile_PlainOldCodeStatement( - PlainOldCodeStatement( - code=f"{stmt.next_expression};", first_source_line=stmt.first_source_line - ), - cx.with_target(continue_f), - ) - continue_f.write_line("if (loopDepth == 0) return {0};", loopF.call("0")) - - if break_f.was_called: - self.try_catch( - cx.with_target(break_f), - cx.catch_f_err, - cx.try_loop_depth, - lambda: break_f.write_line("return {0};", cx.next.call("loopDepth")), - ) - else: - cx.unreachable() - - def _compile_RangeForStatement(self, stmt: RangeForStatement, cx: Context) -> None: - if self.will_continue(stmt.body): - container = next( - (s for s in self.state if s.name == stmt.range_expression), None - ) - if container is None: - raise ActorCompilerError( - stmt.first_source_line, - "container of range-based for with continuation must be a state variable", - ) - iterator_name = self.get_iterator_name(cx) - self.state.append( - StateVar( - source_line=stmt.first_source_line, - name=iterator_name, - type=f"decltype(std::begin(std::declval<{container.type}>()))", - initializer=None, - ) - ) - equivalent = ForStatement( - init_expression=f"{iterator_name} = std::begin({stmt.range_expression})", - cond_expression=f"{iterator_name} != std::end({stmt.range_expression})", - next_expression=f"++{iterator_name}", - first_source_line=stmt.first_source_line, - body=CodeBlock( - statements=[ - PlainOldCodeStatement( - code=f"{stmt.range_decl} = *{iterator_name};", - first_source_line=stmt.first_source_line, - ), - stmt.body, - ], - first_source_line=stmt.first_source_line, - ), - ) - self._compile_ForStatement(equivalent, cx) - else: - self.emit_native_loop( - stmt.first_source_line, - f"for( {stmt.range_decl} : {stmt.range_expression} )", - stmt.body, - cx, - ) - - def _compile_WhileStatement(self, stmt: WhileStatement, cx: Context) -> None: - equivalent = ForStatement( - init_expression="", - cond_expression=stmt.expression, - next_expression="", - body=stmt.body, - first_source_line=stmt.first_source_line, - ) - self._compile_ForStatement(equivalent, cx) - - def _compile_LoopStatement(self, stmt: LoopStatement, cx: Context) -> None: - equivalent = ForStatement( - init_expression="", - cond_expression="", - next_expression="", - body=stmt.body, - first_source_line=stmt.first_source_line, - ) - self._compile_ForStatement(equivalent, cx) - - def _compile_BreakStatement(self, stmt: BreakStatement, cx: Context) -> None: - if cx.break_f is None: - raise ActorCompilerError(stmt.first_source_line, "break outside loop") - if isinstance(cx.break_f, LiteralBreak): - cx.target.write_line("{0};", cx.break_f.call()) - else: - cx.target.write_line( - "return {0}; // break", cx.break_f.call("loopDepth==0?0:loopDepth-1") - ) - cx.unreachable() - - def _compile_ContinueStatement(self, stmt: ContinueStatement, cx: Context) -> None: - if cx.continue_f is None: - raise ActorCompilerError(stmt.first_source_line, "continue outside loop") - if isinstance(cx.continue_f, LiteralContinue): - cx.target.write_line("{0};", cx.continue_f.call()) - else: - cx.target.write_line( - "return {0}; // continue", cx.continue_f.call("loopDepth") - ) - cx.unreachable() - - def _compile_CodeBlock(self, stmt: CodeBlock, cx: Context) -> None: - cx.target.write_line("{") - cx.target.indent(+1) - end = self.compile(stmt, cx, True) - cx.target.indent(-1) - cx.target.write_line("}") - if end.target is None: - cx.unreachable() - elif end.target is not cx.target: - end.target.write_line("loopDepth = {0};", cx.next.call("loopDepth")) - - def _compile_ReturnStatement(self, stmt: ReturnStatement, cx: Context) -> None: - self.line_numberFunction(cx.target, stmt.first_source_line) - if (stmt.expression == "") != (self.actor.return_type is None): - raise ActorCompilerError( - stmt.first_source_line, - "Return statement does not match actor declaration", - ) - if self.actor.return_type is not None: - if stmt.expression == "Never()": - cx.target.write_line("this->~{0}();", self.state_class_name) - cx.target.write_line("{0}->sendAndDelPromiseRef(Never());", self.this) - else: - cx.target.write_line( - "if (!{0}->SAV<{1}>::futures) {{ (void)({2}); this->~{3}(); {0}->destroy(); return 0; }}", - self.this, - self.actor.return_type, - stmt.expression, - self.state_class_name, - ) - if any(s.name == stmt.expression for s in self.state): - cx.target.write_line( - "new (&{0}->SAV< {1} >::value()) {1}(std::move({2})); // state_var_RVO", - self.this, - self.actor.return_type, - stmt.expression, - ) - else: - cx.target.write_line( - "new (&{0}->SAV< {1} >::value()) {1}({2});", - self.this, - self.actor.return_type, - stmt.expression, - ) - cx.target.write_line("this->~{0}();", self.state_class_name) - cx.target.write_line("{0}->finishSendAndDelPromiseRef();", self.this) - else: - cx.target.write_line("delete {0};", self.this) - cx.target.write_line("return 0;") - cx.unreachable() - - def _compile_ThrowStatement(self, stmt: ThrowStatement, cx: Context) -> None: - self.line_numberFunction(cx.target, stmt.first_source_line) - if stmt.expression == "": - if cx.target.exception_parameter_is is not None: - cx.target.write_line( - "return {0};", - cx.catch_f_err.call( - cx.target.exception_parameter_is, - self.adjust_loop_depth(cx.try_loop_depth), - ), - ) - else: - raise ActorCompilerError( - stmt.first_source_line, - "throw statement with no expression has no current exception in scope", - ) - else: - cx.target.write_line( - "return {0};", - cx.catch_f_err.call( - stmt.expression, self.adjust_loop_depth(cx.try_loop_depth) - ), - ) - cx.unreachable() - - def _compile_IfStatement(self, stmt: IfStatement, cx: Context) -> None: - use_continuation = self.will_continue(stmt.if_body) or ( - stmt.else_body is not None and self.will_continue(stmt.else_body) - ) - self.line_numberFunction(cx.target, stmt.first_source_line) - constexpr = "constexpr " if stmt.constexpr else "" - cx.target.write_line(f"if {constexpr}({stmt.expression})") - cx.target.write_line("{") - cx.target.indent(+1) - if_target = self.compile( - self.as_code_block(stmt.if_body), cx, use_continuation - ).target - if use_continuation and if_target is not None: - if_target.write_line("loopDepth = {0};", cx.next.call("loopDepth")) - cx.target.indent(-1) - cx.target.write_line("}") - else_target = None - if stmt.else_body is not None or use_continuation: - cx.target.write_line("else") - cx.target.write_line("{") - cx.target.indent(+1) - else_target = cx.target - if stmt.else_body is not None: - else_target = self.compile( - self.as_code_block(stmt.else_body), cx, use_continuation - ).target - if use_continuation and else_target is not None: - else_target.write_line("loopDepth = {0};", cx.next.call("loopDepth")) - cx.target.indent(-1) - cx.target.write_line("}") - if if_target is None and stmt.else_body is not None and else_target is None: - cx.unreachable() - elif not cx.next.was_called and use_continuation: - raise ActorCompilerError( - stmt.first_source_line, "Internal error: IfStatement: next not called?" - ) - - def _compile_TryStatement(self, stmt: TryStatement, cx: Context) -> None: - if len(stmt.catches) != 1: - raise ActorCompilerError( - stmt.first_source_line, - "try statement must have exactly one catch clause", - ) - reachable = False - catch_clause = stmt.catches[0] - catch_expression = catch_clause.expression.replace(" ", "") - catch_param = "" - if catch_expression != "...": - if not catch_expression.startswith("Error&"): - raise ActorCompilerError( - catch_clause.first_source_line, - "Only type 'Error' or '...' may be caught in an actor function", - ) - catch_param = catch_expression[6:] - if not catch_param: - catch_param = "__current_error" - catch_f_err = self.get_function( - cx.target.name, "Catch", f"const Error& {catch_param}", self.loop_depth_0 - ) - catch_f_err.exception_parameter_is = catch_param - end = self.try_catch_compile( - self.as_code_block(stmt.try_body), cx.with_catch(catch_f_err) - ) - if end.target is not None: - reachable = True - self.try_catch( - end, - cx.catch_f_err, - cx.try_loop_depth, - lambda: end.target.write_line( - "loopDepth = {0};", cx.next.call("loopDepth") - ), - ) - - def handle_catch() -> None: - nonlocal reachable - cend = self._compile_try_catch_body(catch_clause, catch_f_err, cx) - if cend.target is not None: - reachable = True - - self.try_catch( - cx.with_target(catch_f_err), - cx.catch_f_err, - cx.try_loop_depth, - handle_catch, - ) - if not reachable: - cx.unreachable() - - def _compile_try_catch_body( - self, catch_clause: TryStatement.Catch, catch_f_err: Function, cx: Context - ) -> Context: - cend = self.compile( - self.as_code_block(catch_clause.body), cx.with_target(catch_f_err), True - ) - if cend.target is not None: - cend.target.write_line("loopDepth = {0};", cx.next.call("loopDepth")) - return cend - - def _compile_WaitStatement(self, stmt: WaitStatement, cx: Context) -> None: - equivalent = ChooseStatement( - body=CodeBlock( - statements=[ - WhenStatement( - wait=stmt, - body=None, - first_source_line=stmt.first_source_line, - ) - ], - first_source_line=stmt.first_source_line, - ), - first_source_line=stmt.first_source_line, - ) - if not stmt.result_is_state: - cx.next.formal_parameters = [ - f"{stmt.result.type} const& {stmt.result.name}", - self.loop_depth, - ] - cx.next.add_overload( - f"{stmt.result.type} && {stmt.result.name}", self.loop_depth - ) - self._compile_ChooseStatement(equivalent, cx) - - def _compile_ChooseStatement(self, stmt: ChooseStatement, cx: Context) -> None: - group = self.choose_groups + 1 - self.choose_groups = group - codeblock = stmt.body if isinstance(stmt.body, CodeBlock) else None - if codeblock is None: - raise ActorCompilerError( - stmt.first_source_line, - "'choose' must be followed by a compound statement.", - ) - choices = [] - for idx, choice_stmt in enumerate(codeblock.statements): - if not isinstance(choice_stmt, WhenStatement): - raise ActorCompilerError( - choice_stmt.first_source_line, - "only 'when' statements are valid in a 'choose' block.", - ) - index = self.when_count + idx - param_prefix = "__" if choice_stmt.wait.result_is_state else "" - const_param = f"{choice_stmt.wait.result.type} const& {param_prefix}{choice_stmt.wait.result.name}" - rvalue_param = f"{choice_stmt.wait.result.type} && {param_prefix}{choice_stmt.wait.result.name}" - body_func = self.get_function( - cx.target.name, - "when", - [const_param, self.loop_depth], - overload_formal_parameters=[rvalue_param, self.loop_depth], - ) - future_name = f"__when_expr_{index}" - callback_template = ( - "ActorSingleCallback" - if choice_stmt.wait.is_wait_next - else "ActorCallback" - ) - callback_type = f"{callback_template}< {self.full_class_name}, {index}, {choice_stmt.wait.result.type} >" - callback_type_state = f"{callback_template}< {self.class_name}, {index}, {choice_stmt.wait.result.type} >" - choices.append( - { - "Stmt": choice_stmt, - "Group": group, - "Index": index, - "Body": body_func, - "Future": future_name, - "CallbackType": callback_type, - "CallbackTypeState": callback_type_state, - } - ) - self.when_count += len(choices) - exit_func = self.get_function("exitChoose", "", []) - exit_func.return_type = "void" - exit_func.write_line( - "if (actorWaitStateIsWaiting({0}->actor_wait_state)) {0}->actor_wait_state = ACTOR_WAIT_STATE_NOT_WAITING;", - self.this, - ) - for choice in choices: - exit_func.write_line( - "{0}->{1}::remove();", self.this, choice["CallbackTypeState"] - ) - exit_func.end_is_unreachable = True - - reachable = False - for choice in choices: - self.callbacks.append( - CallbackVar( - source_line=choice["Stmt"].first_source_line, - callback_group=choice["Group"], - type=choice["CallbackType"], - ) - ) - r = choice["Body"] - if choice["Stmt"].wait.result_is_state: - overload = r.pop_overload() - decl_stmt = StateDeclarationStatement( - decl=VarDeclaration( - type=choice["Stmt"].wait.result.type, - name=choice["Stmt"].wait.result.name, - initializer=f"__{choice['Stmt'].wait.result.name}", - initializer_constructor_syntax=False, - ), - first_source_line=choice["Stmt"].first_source_line, - ) - self._compile_StateDeclarationStatement(decl_stmt, cx.with_target(r)) - if overload is not None: - overload.write_line( - "{0} = std::move(__{0});", - choice["Stmt"].wait.result.name, - ) - r.set_overload(overload) - if choice["Stmt"].body is not None: - r = self.compile( - self.as_code_block(choice["Stmt"].body), cx.with_target(r), True - ).target - if r is not None: - reachable = True - if len(cx.next.formal_parameters) == 1: - r.write_line("loopDepth = {0};", cx.next.call("loopDepth")) - else: - overload = r.pop_overload() - r.write_line( - "loopDepth = {0};", - cx.next.call(choice["Stmt"].wait.result.name, "loopDepth"), - ) - if overload is not None: - overload.write_line( - "loopDepth = {0};", - cx.next.call( - f"std::move({choice['Stmt'].wait.result.name})", - "loopDepth", - ), - ) - r.set_overload(overload) - - cb_func = Function( - name="callback_fire", - return_type="void", - formal_parameters=[ - f"{choice['CallbackTypeState']}*", - f"{choice['Stmt'].wait.result.type} const& value", - ], - ) - cb_func.end_is_unreachable = True - cb_func.add_overload( - f"{choice['CallbackTypeState']}*", - f"{choice['Stmt'].wait.result.type} && value", - ) - cb_func.indent(self.code_indent) - self.probe_enter(cb_func, self.actor.name, choice["Index"]) - cb_func.write_line("{0};", exit_func.call()) - overload_func = cb_func.pop_overload() - - def fire_body(target_func: Function, expr: str, choice_dict=choice) -> None: - self.try_catch( - cx.with_target(target_func), - cx.catch_f_err, - cx.try_loop_depth, - lambda ch=choice_dict, tf=target_func, expression=expr: tf.write_line( - "{0};", ch["Body"].call(expression, "0") - ), - False, - ) - - fire_body(cb_func, "value") - if overload_func is not None: - fire_body(overload_func, "std::move(value)") - cb_func.set_overload(overload_func) - self.probe_exit(cb_func, self.actor.name, choice["Index"]) - self.functions[f"{cb_func.name}#{choice['Index']}"] = cb_func - - err_func = Function( - name="callback_error", - return_type="void", - formal_parameters=[f"{choice['CallbackTypeState']}*", "Error err"], - ) - err_func.end_is_unreachable = True - err_func.indent(self.code_indent) - self.probe_enter(err_func, self.actor.name, choice["Index"]) - err_func.write_line("{0};", exit_func.call()) - self.try_catch( - cx.with_target(err_func), - cx.catch_f_err, - cx.try_loop_depth, - lambda: err_func.write_line("{0};", cx.catch_f_err.call("err", "0")), - False, - ) - self.probe_exit(err_func, self.actor.name, choice["Index"]) - self.functions[f"{err_func.name}#{choice['Index']}"] = err_func - - first_choice = True - for choice in choices: - get_func = "pop" if choice["Stmt"].wait.is_wait_next else "get" - self.line_numberFunction(cx.target, choice["Stmt"].wait.first_source_line) - if choice["Stmt"].wait.is_wait_next: - cx.target.write_line( - "auto {0} = {1};", - choice["Future"], - choice["Stmt"].wait.future_expression, - ) - cx.target.write_line( - 'static_assert(std::is_same>::value || std::is_same>::value, "invalid type");', - choice["Future"], - choice["Stmt"].wait.result.type, - ) - else: - cx.target.write_line( - "StrictFuture<{0}> {1} = {2};", - choice["Stmt"].wait.result.type, - choice["Future"], - choice["Stmt"].wait.future_expression, - ) - if first_choice: - first_choice = False - self.line_numberFunction(cx.target, stmt.first_source_line) - if self.actor.is_cancellable(): - cx.target.write_line( - "if (actorWaitStateIsCancelled({1}->actor_wait_state)) return {0};", - cx.catch_f_err.call( - "actor_cancelled()", self.adjust_loop_depth(cx.try_loop_depth) - ), - self.this, - ) - cx.target.write_line( - "if ({0}.isReady()) {{ if ({0}.isError()) return {2}; else return {1}; }};", - choice["Future"], - choice["Body"].call(f"{choice['Future']}.{get_func}()", "loopDepth"), - cx.catch_f_err.call( - f"{choice['Future']}.getError()", - self.adjust_loop_depth(cx.try_loop_depth), - ), - ) - cx.target.write_line("{1}->actor_wait_state = {0};", group, self.this) - for choice in choices: - self.line_numberFunction(cx.target, choice["Stmt"].wait.first_source_line) - cx.target.write_line( - "{0}.addCallbackAndClear(static_cast<{1}*>({2}));", - choice["Future"], - choice["CallbackTypeState"], - self.this, - ) - cx.target.write_line("loopDepth = 0;") - if not reachable: - cx.unreachable() diff --git a/flow/actorcompiler_py/actor_parser.py b/flow/actorcompiler_py/actor_parser.py deleted file mode 100644 index 2be13009759..00000000000 --- a/flow/actorcompiler_py/actor_parser.py +++ /dev/null @@ -1,1039 +0,0 @@ -from __future__ import annotations - -import io -import re -import sys -from dataclasses import dataclass -from typing import Dict, Iterable, Iterator, List, Tuple - -from .errors import ActorCompilerError -from .actor_compiler import ActorCompiler - -from .parse_tree import ( - Actor, - BreakStatement, - ChooseStatement, - CodeBlock, - ContinueStatement, - ForStatement, - IfStatement, - LoopStatement, - PlainOldCodeStatement, - RangeForStatement, - ReturnStatement, - StateDeclarationStatement, - Statement, - ThrowStatement, - TryStatement, - VarDeclaration, - WaitStatement, - WhenStatement, - WhileStatement, -) - - -class ErrorMessagePolicy: - def __init__(self) -> None: - self.disable_diagnostics = False - - def handle_actor_without_wait(self, source_file: str, actor: Actor) -> None: - if not self.disable_diagnostics and not actor.is_test_case: - print( - f"{source_file}:{actor.source_line}: warning: ACTOR {actor.name} does not contain a wait() statement", - file=sys.stderr, - ) - - def actors_no_discard_by_default(self) -> bool: - return not self.disable_diagnostics - - -@dataclass -class Token: - value: str - position: int = 0 - source_line: int = 0 - brace_depth: int = 0 - paren_depth: int = 0 - - @property - def is_whitespace(self) -> bool: - return ( - self.value in (" ", "\n", "\r", "\r\n", "\t") - or self.value.startswith("//") - or self.value.startswith("/*") - ) - - def ensure(self, error: str, pred) -> "Token": - if not pred(self): - raise ActorCompilerError(self.source_line, error) - return self - - def get_matching_range_in(self, token_range: "TokenRange") -> "TokenRange": - if self.value == "<": - sub = TokenRange(token_range.tokens, self.position, token_range.end_pos) - gen = AngleBracketParser.not_inside_angle_brackets(sub) - next(gen, None) # skip the "<" - closing = next(gen, None) - if closing is None: - raise ActorCompilerError(self.source_line, "Syntax error: Unmatched <") - return TokenRange(token_range.tokens, self.position + 1, closing.position) - if self.value == "[": - sub = TokenRange(token_range.tokens, self.position, token_range.end_pos) - gen = BracketParser.not_inside_brackets(sub) - next(gen, None) # skip the "[" - closing = next(gen, None) - if closing is None: - raise ActorCompilerError(self.source_line, "Syntax error: Unmatched [") - return TokenRange(token_range.tokens, self.position + 1, closing.position) - - pairs = {"(": ")", ")": "(", "{": "}", "}": "{"} - if self.value not in pairs: - raise RuntimeError("Can't match this token") - pred = ( - (lambda t: t.value != ")" or t.paren_depth != self.paren_depth) - if self.value == "(" - else ( - (lambda t: t.value != "(" or t.paren_depth != self.paren_depth) - if self.value == ")" - else ( - (lambda t: t.value != "}" or t.brace_depth != self.brace_depth) - if self.value == "{" - else (lambda t: t.value != "{" or t.brace_depth != self.brace_depth) - ) - ) - ) - direction = 1 if self.value in ("(", "{") else -1 - if direction == -1: - rng = token_range.sub_range(token_range.begin_pos, self.position).Revtake_while(pred) - if rng.begin_pos == token_range.begin_pos: - raise ActorCompilerError( - self.source_line, f"Syntax error: Unmatched {self.value}" - ) - return rng - rng = token_range.sub_range(self.position + 1, token_range.end_pos).take_while(pred) - if rng.end_pos == token_range.end_pos: - raise ActorCompilerError( - self.source_line, f"Syntax error: Unmatched {self.value}" - ) - return rng - - -class TokenRange: - def __init__(self, tokens: List[Token], begin: int, end: int) -> None: - if begin > end: - raise RuntimeError("Invalid TokenRange") - self.tokens = tokens - self.begin_pos = begin - self.end_pos = end - - def __iter__(self) -> Iterator[Token]: - for i in range(self.begin_pos, self.end_pos): - yield self.tokens[i] - - def first(self, predicate=None) -> Token: - if self.begin_pos == self.end_pos: - raise RuntimeError("Empty TokenRange") - if predicate is None: - return self.tokens[self.begin_pos] - for t in self: - if predicate(t): - return t - raise RuntimeError("Matching token not found") - - def last(self, predicate=None) -> Token: - if self.begin_pos == self.end_pos: - raise RuntimeError("Empty TokenRange") - if predicate is None: - return self.tokens[self.end_pos - 1] - for i in range(self.end_pos - 1, self.begin_pos - 1, -1): - if predicate(self.tokens[i]): - return self.tokens[i] - raise RuntimeError("Matching token not found") - - def skip(self, count: int) -> "TokenRange": - return TokenRange(self.tokens, self.begin_pos + count, self.end_pos) - - def consume(self, value_or_error, predicate=None) -> "TokenRange": - if predicate is None: - self.first().ensure( - f"Expected {value_or_error}", lambda t: t.value == value_or_error - ) - else: - self.first().ensure(value_or_error, predicate) - return self.skip(1) - - def skipWhile(self, predicate) -> "TokenRange": - e = self.begin_pos - while e < self.end_pos and predicate(self.tokens[e]): - e += 1 - return TokenRange(self.tokens, e, self.end_pos) - - def take_while(self, predicate) -> "TokenRange": - e = self.begin_pos - while e < self.end_pos and predicate(self.tokens[e]): - e += 1 - return TokenRange(self.tokens, self.begin_pos, e) - - def Revtake_while(self, predicate) -> "TokenRange": - e = self.end_pos - 1 - while e >= self.begin_pos and predicate(self.tokens[e]): - e -= 1 - return TokenRange(self.tokens, e + 1, self.end_pos) - - def Revskip_while(self, predicate) -> "TokenRange": - e = self.end_pos - 1 - while e >= self.begin_pos and predicate(self.tokens[e]): - e -= 1 - return TokenRange(self.tokens, self.begin_pos, e + 1) - - def sub_range(self, begin: int, end: int) -> "TokenRange": - return TokenRange(self.tokens, begin, end) - - def is_empty(self) -> bool: - return self.begin_pos == self.end_pos - - def length(self) -> int: - return self.end_pos - self.begin_pos - - def all_match(self, predicate) -> bool: - return all(predicate(t) for t in self) - - def any_match(self, predicate) -> bool: - return any(predicate(t) for t in self) - - -class BracketParser: - @staticmethod - def not_inside_brackets(tokens: Iterable[Token]) -> Iterator[Token]: - bracket_depth = 0 - base_pd = None - for tok in tokens: - if base_pd is None: - base_pd = tok.paren_depth - if tok.paren_depth == base_pd and tok.value == "]": - bracket_depth -= 1 - if bracket_depth == 0: - yield tok - if tok.paren_depth == base_pd and tok.value == "[": - bracket_depth += 1 - - -class AngleBracketParser: - @staticmethod - def not_inside_angle_brackets(tokens: Iterable[Token]) -> Iterator[Token]: - angle_depth = 0 - base_pd = None - for tok in tokens: - if base_pd is None: - base_pd = tok.paren_depth - if tok.paren_depth == base_pd and tok.value == ">": - angle_depth -= 1 - if angle_depth == 0: - yield tok - if tok.paren_depth == base_pd and tok.value == "<": - angle_depth += 1 - - -class ActorParser: - token_expressions = [ - r"\{", - r"\}", - r"\(", - r"\)", - r"\[", - r"\]", - r"//[^\n]*", - r"/[*]([*][^/]|[^*])*[*]/", - r"'(\\.|[^\'\n])*'", - r'"(\\.|[^"\n])*"', - r"[a-zA-Z_][a-zA-Z_0-9]*", - r"\r\n", - r"\n", - r"::", - r":", - r"#[a-z]*", - r".", - ] - - identifier_pattern = re.compile(r"^[a-zA-Z_][a-zA-Z_0-9]*$") - - def __init__( - self, - text: str, - source_file: str, - error_message_policy: ErrorMessagePolicy, - generate_probes: bool, - ) -> None: - self.source_file = source_file - self.error_message_policy = error_message_policy - self.generate_probes = generate_probes - self.tokens = [Token(value=t) for t in self.tokenize(text)] - self.line_numbers_enabled = True - self.uid_objects: Dict[Tuple[int, int], str] = {} - self.token_array = self.tokens - self.count_parens() - - def tokenize(self, text: str) -> List[str]: - regexes = [re.compile(pattern, re.S) for pattern in self.token_expressions] - pos = 0 - tokens: List[str] = [] - while pos < len(text): - for regex in regexes: - m = regex.match(text, pos) - if m: - tokens.append(m.group(0)) - pos += len(m.group(0)) - break - else: - raise RuntimeError(f"Can't tokenize! {pos}") - return tokens - - def count_parens(self) -> None: - brace_depth = 0 - paren_depth = 0 - line_count = 1 - last_paren = None - last_brace = None - for i, token in enumerate(self.tokens): - value = token.value - if value == "}": - brace_depth -= 1 - elif value == ")": - paren_depth -= 1 - elif value == "\r\n": - line_count += 1 - elif value == "\n": - line_count += 1 - - if brace_depth < 0: - raise ActorCompilerError(line_count, "Mismatched braces") - if paren_depth < 0: - raise ActorCompilerError(line_count, "Mismatched parenthesis") - - token.position = i - token.source_line = line_count - token.brace_depth = brace_depth - token.paren_depth = paren_depth - - if value.startswith("/*"): - line_count += value.count("\n") - if value == "{": - brace_depth += 1 - if brace_depth == 1: - last_brace = token - elif value == "(": - paren_depth += 1 - if paren_depth == 1: - last_paren = token - if brace_depth != 0: - raise ActorCompilerError( - last_brace.source_line if last_brace else line_count, "Unmatched brace" - ) - if paren_depth != 0: - raise ActorCompilerError( - last_paren.source_line if last_paren else line_count, - "Unmatched parenthesis", - ) - - def write(self, writer: io.TextIOBase, destFileName: str) -> None: - ActorCompiler.used_class_names.clear() - writer.write("#define POST_ACTOR_COMPILER 1\n") - outLine = 1 - if self.line_numbers_enabled: - writer.write(f'#line {self.tokens[0].source_line} "{self.source_file}"\n') - outLine += 1 - inBlocks = 0 - classContextStack: List[Tuple[str, int]] = [] - i = 0 - while i < len(self.tokens): - tok = self.tokens[i] - if tok.source_line == 0: - raise RuntimeError("Invalid source line (0)") - if tok.value in ("ACTOR", "SWIFT_ACTOR", "TEST_CASE"): - actor = self.parse_actor(i) - end = self._parse_end - if classContextStack: - actor.enclosing_class = "::".join( - name for name, _ in classContextStack - ) - actor_writer = io.StringIO() - actorCompiler = ActorCompiler( - actor, - self.source_file, - inBlocks == 0, - self.line_numbers_enabled, - self.generate_probes, - ) - actorCompiler.write(actor_writer) - for key, value in actorCompiler.uid_objects.items(): - self.uid_objects.setdefault(key, value) - actor_lines = actor_writer.getvalue().split("\n") - hasLineNumber = False - hadLineNumber = True - for line in actor_lines: - if self.line_numbers_enabled: - isLine = "#line" in line - if isLine: - hadLineNumber = True - if not isLine and not hasLineNumber and hadLineNumber: - writer.write( - '\t\t\t\t\t\t\t\t\t\t\t\t\t\t\t#line {0} "{1}"\n'.format( - outLine + 1, destFileName - ) - ) - outLine += 1 - hadLineNumber = False - hasLineNumber = isLine - writer.write(line.rstrip("\n\r") + "\n") - outLine += 1 - i = end - if i < len(self.tokens) and self.line_numbers_enabled: - writer.write( - f'#line {self.tokens[i].source_line} "{self.source_file}"\n' - ) - outLine += 1 - elif tok.value in ("class", "struct", "union"): - writer.write(tok.value) - success, name = self.parse_class_context( - self.range(i + 1, len(self.tokens)) - ) - if success: - classContextStack.append((name, inBlocks)) - else: - if tok.value == "{": - inBlocks += 1 - elif tok.value == "}": - inBlocks -= 1 - if classContextStack and classContextStack[-1][1] == inBlocks: - classContextStack.pop() - writer.write(tok.value) - outLine += tok.value.count("\n") - i += 1 - - # Parsing helpers - - def range(self, begin: int, end: int) -> TokenRange: - return TokenRange(self.tokens, begin, end) - - def parse_actor(self, pos: int) -> Actor: - token = self.tokens[pos] - actor = Actor() - actor.source_line = token.source_line - toks = self.range(pos + 1, len(self.tokens)) - heading = toks.take_while(lambda t: t.value != "{") - toSemicolon = toks.take_while(lambda t: t.value != ";") - actor.is_forward_declaration = toSemicolon.length() < heading.length() - if actor.is_forward_declaration: - heading = toSemicolon - if token.value in ("ACTOR", "SWIFT_ACTOR"): - self.parse_actorHeading(actor, heading) - else: - token.ensure("ACTOR expected!", lambda _: False) - self._parse_end = heading.end_pos + 1 - else: - body = self.range(heading.end_pos + 1, len(self.tokens)).take_while( - lambda t: t.brace_depth > toks.first().brace_depth - ) - if token.value in ("ACTOR", "SWIFT_ACTOR"): - self.parse_actorHeading(actor, heading) - elif token.value == "TEST_CASE": - self.parse_test_caseHeading(actor, heading) - actor.is_test_case = True - else: - token.ensure("ACTOR or TEST_CASE expected!", lambda _: False) - actor.body = self.parse_code_block(body) - if not actor.body.contains_wait(): - self.error_message_policy.handle_actor_without_wait(self.source_file, actor) - self._parse_end = body.end_pos + 1 - return actor - - def parse_test_caseHeading(self, actor: Actor, toks: TokenRange) -> None: - actor.is_static = True - nonWhitespace = lambda t: not t.is_whitespace - paramRange = ( - toks.last(nonWhitespace) - .ensure( - "Unexpected tokens after test case parameter list.", - lambda t: t.value == ")" and t.paren_depth == toks.first().paren_depth, - ) - .get_matching_range_in(toks) - ) - actor.test_case_parameters = self.str(paramRange) - actor.name = f"flowTestCase{toks.first().source_line}" - actor.parameters = [ - VarDeclaration( - name="params", - type="UnitTestParameters", - initializer="", - initializer_constructor_syntax=False, - ) - ] - actor.return_type = "Void" - - def parse_actorHeading(self, actor: Actor, toks: TokenRange) -> None: - nonWhitespace = lambda t: not t.is_whitespace - template = toks.first(nonWhitespace) - if template.value == "template": - templateParams = ( - self.range(template.position + 1, toks.end_pos) - .first(nonWhitespace) - .ensure("Invalid template declaration", lambda t: t.value == "<") - .get_matching_range_in(toks) - ) - actor.template_formals = [ - self.parse_var_declaration(p) - for p in self.split_parameter_list(templateParams, ",") - ] - toks = self.range(templateParams.end_pos + 1, toks.end_pos) - attribute = toks.first(nonWhitespace) - while attribute.value == "[": - contents = attribute.get_matching_range_in(toks) - as_array = list(contents) - if ( - len(as_array) < 2 - or as_array[0].value != "[" - or as_array[-1].value != "]" - ): - raise ActorCompilerError( - actor.source_line, "Invalid attribute: Expected [[...]]" - ) - actor.attributes.append( - "[" + self.str(self.normalize_whitespace(contents)) + "]" - ) - toks = self.range(contents.end_pos + 1, toks.end_pos) - attribute = toks.first(nonWhitespace) - static_keyword = toks.first(nonWhitespace) - if static_keyword.value == "static": - actor.is_static = True - toks = self.range(static_keyword.position + 1, toks.end_pos) - uncancellable = toks.first(nonWhitespace) - if uncancellable.value == "UNCANCELLABLE": - actor.set_uncancellable() - toks = self.range(uncancellable.position + 1, toks.end_pos) - paramRange = ( - toks.last(nonWhitespace) - .ensure( - "Unexpected tokens after actor parameter list.", - lambda t: t.value == ")" and t.paren_depth == toks.first().paren_depth, - ) - .get_matching_range_in(toks) - ) - actor.parameters = [ - self.parse_var_declaration(p) - for p in self.split_parameter_list(paramRange, ",") - ] - nameToken = self.range(toks.begin_pos, paramRange.begin_pos - 1).last(nonWhitespace) - actor.name = nameToken.value - return_range = self.range( - toks.first().position + 1, nameToken.position - ).skipWhile(lambda t: t.is_whitespace) - retToken = return_range.first() - if retToken.value == "Future": - ofType = ( - return_range.skip(1) - .first(nonWhitespace) - .ensure("Expected <", lambda tok: tok.value == "<") - .get_matching_range_in(return_range) - ) - actor.return_type = self.str(self.normalize_whitespace(ofType)) - toks = self.range(ofType.end_pos + 1, return_range.end_pos) - elif retToken.value == "void": - actor.return_type = None - toks = return_range.skip(1) - else: - raise ActorCompilerError( - actor.source_line, "Actor apparently does not return Future" - ) - toks = toks.skipWhile(lambda t: t.is_whitespace) - if not toks.is_empty(): - if toks.last().value == "::": - actor.name_space = self.str(self.range(toks.begin_pos, toks.end_pos - 1)) - else: - raise ActorCompilerError( - actor.source_line, - "Unrecognized tokens preceding parameter list in actor declaration", - ) - if ( - self.error_message_policy.actors_no_discard_by_default() - and "[[flow_allow_discard]]" not in actor.attributes - ): - if actor.is_cancellable(): - actor.attributes.append("[[nodiscard]]") - known_flow_attributes = {"[[flow_allow_discard]]"} - for flow_attribute in [a for a in actor.attributes if a.startswith("[[flow_")]: - if flow_attribute not in known_flow_attributes: - raise ActorCompilerError( - actor.source_line, f"Unknown flow attribute {flow_attribute}" - ) - actor.attributes = [a for a in actor.attributes if not a.startswith("[[flow_")] - - def parse_var_declaration(self, tokens: TokenRange) -> VarDeclaration: - name, typeRange, initializer, constructorSyntax = self.parse_declaration(tokens) - return VarDeclaration( - name=name.value, - type=self.str(self.normalize_whitespace(typeRange)), - initializer=( - "" - if initializer is None - else self.str(self.normalize_whitespace(initializer)) - ), - initializer_constructor_syntax=constructorSyntax, - ) - - def parse_declaration(self, tokens: TokenRange): - nonWhitespace = lambda t: not t.is_whitespace - initializer = None - beforeInitializer = tokens - constructorSyntax = False - equals = next( - ( - t - for t in AngleBracketParser.not_inside_angle_brackets(tokens) - if t.value == "=" and t.paren_depth == tokens.first().paren_depth - ), - None, - ) - if equals: - beforeInitializer = self.range(tokens.begin_pos, equals.position) - initializer = self.range(equals.position + 1, tokens.end_pos) - else: - paren = next( - ( - t - for t in AngleBracketParser.not_inside_angle_brackets(tokens) - if t.value == "(" - ), - None, - ) - if paren: - constructorSyntax = True - beforeInitializer = self.range(tokens.begin_pos, paren.position) - initializer = self.range(paren.position + 1, tokens.end_pos).take_while( - lambda t, p=paren.paren_depth: t.paren_depth > p - ) - else: - brace = next( - ( - t - for t in AngleBracketParser.not_inside_angle_brackets(tokens) - if t.value == "{" - ), - None, - ) - if brace: - raise ActorCompilerError( - brace.source_line, - "Uniform initialization syntax is not currently supported for state variables (use '(' instead of '}' ?)", - ) - name = beforeInitializer.last(nonWhitespace) - if beforeInitializer.begin_pos == name.position: - raise ActorCompilerError( - beforeInitializer.first().source_line, "Declaration has no type." - ) - typeRange = self.range(beforeInitializer.begin_pos, name.position) - return name, typeRange, initializer, constructorSyntax - - def normalize_whitespace(self, tokens: Iterable[Token]) -> Iterable[Token]: - inWhitespace = False - leading = True - for tok in tokens: - if not tok.is_whitespace: - if inWhitespace and not leading: - yield Token(value=" ") - inWhitespace = False - yield tok - leading = False - else: - inWhitespace = True - - def split_parameter_list( - self, toks: TokenRange, delimiter: str - ) -> Iterable[TokenRange]: - if toks.begin_pos == toks.end_pos: - return [] - ranges: List[TokenRange] = [] - while True: - comma = next( - ( - t - for t in AngleBracketParser.not_inside_angle_brackets(toks) - if t.value == delimiter and t.paren_depth == toks.first().paren_depth - ), - None, - ) - if comma is None: - break - ranges.append(self.range(toks.begin_pos, comma.position)) - toks = self.range(comma.position + 1, toks.end_pos) - ranges.append(toks) - return ranges - - def parse_loop_statement(self, toks: TokenRange) -> LoopStatement: - return LoopStatement(body=self.parse_compound_statement(toks.consume("loop"))) - - def parse_choose_statement(self, toks: TokenRange) -> ChooseStatement: - return ChooseStatement(body=self.parse_compound_statement(toks.consume("choose"))) - - def parse_when_statement(self, toks: TokenRange) -> WhenStatement: - expr = ( - toks.consume("when") - .skipWhile(lambda t: t.is_whitespace) - .first() - .ensure("Expected (", lambda t: t.value == "(") - .get_matching_range_in(toks) - .skipWhile(lambda t: t.is_whitespace) - ) - return WhenStatement( - wait=self.parse_wait_statement(expr), - body=self.parse_compound_statement(self.range(expr.end_pos + 1, toks.end_pos)), - ) - - def parse_state_declaration(self, toks: TokenRange) -> StateDeclarationStatement: - toks = toks.consume("state").Revskip_while(lambda t: t.value == ";") - return StateDeclarationStatement(decl=self.parse_var_declaration(toks)) - - def parse_return_statement(self, toks: TokenRange) -> ReturnStatement: - toks = toks.consume("return").Revskip_while(lambda t: t.value == ";") - return ReturnStatement(expression=self.str(self.normalize_whitespace(toks))) - - def parse_throw_statement(self, toks: TokenRange) -> ThrowStatement: - toks = toks.consume("throw").Revskip_while(lambda t: t.value == ";") - return ThrowStatement(expression=self.str(self.normalize_whitespace(toks))) - - def parse_wait_statement(self, toks: TokenRange) -> WaitStatement: - ws = WaitStatement() - ws.first_source_line = toks.first().source_line - if toks.first().value == "state": - ws.result_is_state = True - toks = toks.consume("state") - initializer = None - if toks.first().value in ("wait", "waitNext"): - initializer = toks.Revskip_while(lambda t: t.value == ";") - ws.result = VarDeclaration( - name="_", - type="Void", - initializer="", - initializer_constructor_syntax=False, - ) - else: - name, typeRange, initializer, constructorSyntax = self.parse_declaration( - toks.Revskip_while(lambda t: t.value == ";") - ) - type_str = self.str(self.normalize_whitespace(typeRange)) - if type_str == "Void": - raise ActorCompilerError( - ws.first_source_line, - "Assigning the result of a Void wait is not allowed. Just use a standalone wait statement.", - ) - ws.result = VarDeclaration( - name=name.value, - type=self.str(self.normalize_whitespace(typeRange)), - initializer="", - initializer_constructor_syntax=False, - ) - if initializer is None: - raise ActorCompilerError( - ws.first_source_line, - "Wait statement must be a declaration or standalone statement", - ) - waitParams = ( - initializer.skipWhile(lambda t: t.is_whitespace) - .consume( - "Statement contains a wait, but is not a valid wait statement or a supported compound statement.1", - lambda t: True if t.value in ("wait", "waitNext") else False, - ) - .skipWhile(lambda t: t.is_whitespace) - .first() - .ensure("Expected (", lambda t: t.value == "(") - .get_matching_range_in(initializer) - ) - if ( - not self.range(waitParams.end_pos, initializer.end_pos) - .consume(")") - .all_match(lambda t: t.is_whitespace) - ): - raise ActorCompilerError( - toks.first().source_line, - "Statement contains a wait, but is not a valid wait statement or a supported compound statement.2", - ) - ws.future_expression = self.str(self.normalize_whitespace(waitParams)) - ws.is_wait_next = "waitNext" in [t.value for t in initializer] - return ws - - def parse_while_statement(self, toks: TokenRange) -> WhileStatement: - expr = ( - toks.consume("while") - .first(lambda t: not t.is_whitespace) - .ensure("Expected (", lambda t: t.value == "(") - .get_matching_range_in(toks) - ) - return WhileStatement( - expression=self.str(self.normalize_whitespace(expr)), - body=self.parse_compound_statement(self.range(expr.end_pos + 1, toks.end_pos)), - ) - - def parse_for_statement(self, toks: TokenRange) -> Statement: - head = ( - toks.consume("for") - .first(lambda t: not t.is_whitespace) - .ensure("Expected (", lambda t: t.value == "(") - .get_matching_range_in(toks) - ) - delim = [ - t - for t in head - if t.paren_depth == head.first().paren_depth - and t.brace_depth == head.first().brace_depth - and t.value == ";" - ] - if len(delim) == 2: - init = self.range(head.begin_pos, delim[0].position) - cond = self.range(delim[0].position + 1, delim[1].position) - next_expr = self.range(delim[1].position + 1, head.end_pos) - body = self.range(head.end_pos + 1, toks.end_pos) - return ForStatement( - init_expression=self.str(self.normalize_whitespace(init)), - cond_expression=self.str(self.normalize_whitespace(cond)), - next_expression=self.str(self.normalize_whitespace(next_expr)), - body=self.parse_compound_statement(body), - ) - delim = [ - t - for t in head - if t.paren_depth == head.first().paren_depth - and t.brace_depth == head.first().brace_depth - and t.value == ":" - ] - if len(delim) != 1: - raise ActorCompilerError( - head.first().source_line, - "for statement must be 3-arg style or c++11 2-arg style", - ) - return RangeForStatement( - range_expression=self.str( - self.normalize_whitespace( - self.range(delim[0].position + 1, head.end_pos).skipWhile( - lambda t: t.is_whitespace - ) - ) - ), - range_decl=self.str( - self.normalize_whitespace( - self.range(head.begin_pos, delim[0].position - 1).skipWhile( - lambda t: t.is_whitespace - ) - ) - ), - body=self.parse_compound_statement(self.range(head.end_pos + 1, toks.end_pos)), - ) - - def parse_if_statement(self, toks: TokenRange) -> IfStatement: - toks = toks.consume("if").skipWhile(lambda t: t.is_whitespace) - constexpr = toks.first().value == "constexpr" - if constexpr: - toks = toks.consume("constexpr").skipWhile(lambda t: t.is_whitespace) - expr = ( - toks.first(lambda t: not t.is_whitespace) - .ensure("Expected (", lambda t: t.value == "(") - .get_matching_range_in(toks) - ) - return IfStatement( - expression=self.str(self.normalize_whitespace(expr)), - constexpr=constexpr, - if_body=self.parse_compound_statement(self.range(expr.end_pos + 1, toks.end_pos)), - ) - - def parse_else_statement(self, toks: TokenRange, prev: Statement) -> None: - if_stmt = prev - while isinstance(if_stmt, IfStatement) and if_stmt.else_body is not None: - if_stmt = if_stmt.else_body - if not isinstance(if_stmt, IfStatement): - raise ActorCompilerError( - toks.first().source_line, "else without matching if" - ) - if_stmt.else_body = self.parse_compound_statement(toks.consume("else")) - - def parse_try_statement(self, toks: TokenRange) -> TryStatement: - return TryStatement( - try_body=self.parse_compound_statement(toks.consume("try")), - catches=[], - ) - - def parse_catch_statement(self, toks: TokenRange, prev: Statement) -> None: - if not isinstance(prev, TryStatement): - raise ActorCompilerError( - toks.first().source_line, "catch without matching try" - ) - expr = ( - toks.consume("catch") - .first(lambda t: not t.is_whitespace) - .ensure("Expected (", lambda t: t.value == "(") - .get_matching_range_in(toks) - ) - prev.catches.append( - TryStatement.Catch( - expression=self.str(self.normalize_whitespace(expr)), - body=self.parse_compound_statement(self.range(expr.end_pos + 1, toks.end_pos)), - first_source_line=expr.first().source_line, - ) - ) - - illegal_keywords = {"goto", "do", "finally", "__if_exists", "__if_not_exists"} - - def parse_compound_statement(self, toks: TokenRange) -> Statement: - nonWhitespace = lambda t: not t.is_whitespace - first = toks.first(nonWhitespace) - if first.value == "{": - inBraces = first.get_matching_range_in(toks) - if ( - not self.range(inBraces.end_pos, toks.end_pos) - .consume("}") - .all_match(lambda t: t.is_whitespace) - ): - raise ActorCompilerError( - inBraces.last().source_line, - "Unexpected tokens after compound statement", - ) - return self.parse_code_block(inBraces) - statements: List[Statement] = [] - self.parse_statement(toks.skip(1), statements) - return statements[0] - - def parse_code_block(self, toks: TokenRange) -> CodeBlock: - statements: List[Statement] = [] - while True: - delim = next( - ( - t - for t in toks - if t.paren_depth == toks.first().paren_depth - and t.brace_depth == toks.first().brace_depth - and t.value in (";", "}") - ), - None, - ) - if delim is None: - break - self.parse_statement(self.range(toks.begin_pos, delim.position + 1), statements) - toks = self.range(delim.position + 1, toks.end_pos) - if not toks.all_match(lambda t: t.is_whitespace): - raise ActorCompilerError( - toks.first(lambda t: not t.is_whitespace).source_line, - "Trailing unterminated statement in code block", - ) - return CodeBlock(statements=statements) - - def parse_statement(self, toks: TokenRange, statements: List[Statement]) -> None: - nonWhitespace = lambda t: not t.is_whitespace - toks = toks.skipWhile(lambda t: t.is_whitespace) - - def Add(stmt: Statement) -> None: - stmt.first_source_line = toks.first().source_line - statements.append(stmt) - - first_val = toks.first().value - if first_val == "loop": - Add(self.parse_loop_statement(toks)) - elif first_val == "while": - Add(self.parse_while_statement(toks)) - elif first_val == "for": - Add(self.parse_for_statement(toks)) - elif first_val == "break": - Add(BreakStatement()) - elif first_val == "continue": - Add(ContinueStatement()) - elif first_val == "return": - Add(self.parse_return_statement(toks)) - elif first_val == "{": - Add(self.parse_compound_statement(toks)) - elif first_val == "if": - Add(self.parse_if_statement(toks)) - elif first_val == "else": - self.parse_else_statement(toks, statements[-1]) - elif first_val == "choose": - Add(self.parse_choose_statement(toks)) - elif first_val == "when": - Add(self.parse_when_statement(toks)) - elif first_val == "try": - Add(self.parse_try_statement(toks)) - elif first_val == "catch": - self.parse_catch_statement(toks, statements[-1]) - elif first_val == "throw": - Add(self.parse_throw_statement(toks)) - else: - if first_val in self.illegal_keywords: - raise ActorCompilerError( - toks.first().source_line, - f"Statement '{first_val}' not supported in actors.", - ) - if any(t.value in ("wait", "waitNext") for t in toks): - Add(self.parse_wait_statement(toks)) - elif first_val == "state": - Add(self.parse_state_declaration(toks)) - elif first_val == "switch" and any(t.value == "return" for t in toks): - raise ActorCompilerError( - toks.first().source_line, - "Unsupported compound statement containing return.", - ) - elif first_val.startswith("#"): - raise ActorCompilerError( - toks.first().source_line, - f'Found "{first_val}". Preprocessor directives are not supported within ACTORs', - ) - else: - cleaned = toks.Revskip_while(lambda t: t.value == ";") - if any(nonWhitespace(t) for t in cleaned): - Add( - PlainOldCodeStatement( - code=self.str(self.normalize_whitespace(cleaned)) + ";" - ) - ) - - def parse_class_context(self, toks: TokenRange) -> Tuple[bool, str]: - name = "" - if toks.begin_pos == toks.end_pos: - return False, name - - nonWhitespace = lambda t: not t.is_whitespace - while True: - first = toks.first(nonWhitespace) - if first.value == "[": - contents = first.get_matching_range_in(toks) - toks = self.range(contents.end_pos + 1, toks.end_pos) - elif first.value == "alignas": - toks = self.range(first.position + 1, toks.end_pos) - first = toks.first(nonWhitespace) - first.ensure("Expected ( after alignas", lambda t: t.value == "(") - contents = first.get_matching_range_in(toks) - toks = self.range(contents.end_pos + 1, toks.end_pos) - else: - break - - first = toks.first(nonWhitespace) - if not self.identifier_pattern.match(first.value): - return False, name - - while True: - first.ensure( - "Expected identifier", lambda t: self.identifier_pattern.match(t.value) - ) - name += first.value - toks = self.range(first.position + 1, toks.end_pos) - next_token = toks.first(nonWhitespace) - if next_token.value == "::": - name += "::" - toks = toks.skipWhile(lambda t: t.is_whitespace).skip(1) - else: - break - first = toks.first(nonWhitespace) - - toks = toks.skipWhile( - lambda t: t.is_whitespace or t.value in ("final", "explicit") - ) - first = toks.first(nonWhitespace) - if first.value in (":", "{"): - return True, name - return False, "" - - def str(self, tokens: Iterable[Token]) -> str: - return "".join(tok.value for tok in tokens) diff --git a/flow/actorcompiler_py/compare_actor_output.py b/flow/actorcompiler_py/compare_actor_output.py deleted file mode 100644 index e4208ebb4c4..00000000000 --- a/flow/actorcompiler_py/compare_actor_output.py +++ /dev/null @@ -1,60 +0,0 @@ -import argparse -import sys -import re -import difflib -from pathlib import Path - -def compare_outputs(file1_path: Path, file2_path: Path) -> bool: - try: - file1_content = file1_path.read_text() - file2_content = file2_path.read_text() - - if file1_content == file2_content: - return True - - # Normalize #line directives - replace the filename part with a constant - # Pattern: #line "filename" - line_directive_pattern = re.compile(r'(#line\s+\d+\s+")[^"]*(")') - - file1_normalized = line_directive_pattern.sub( - r"\1NORMALIZED_PATH\2", file1_content - ) - file2_normalized = line_directive_pattern.sub( - r"\1NORMALIZED_PATH\2", file2_content - ) - - if file1_normalized == file2_normalized: - return True - - # Generate detailed diff of normalized content - file1_lines = file1_normalized.splitlines(keepends=True) - file2_lines = file2_normalized.splitlines(keepends=True) - - diff = difflib.unified_diff( - file1_lines, - file2_lines, - fromfile=f"{file1_path} (normalized)", - tofile=f"{file2_path} (normalized)", - lineterm="", - ) - - diff_text = "".join(diff) - print(diff_text, file=sys.stderr) - return False - - except Exception as e: - print(f"Error comparing files: {e}", file=sys.stderr) - return False - -def main(): - parser = argparse.ArgumentParser(description="Compare actor compiler outputs") - parser.add_argument("file1", type=Path, help="First file to compare") - parser.add_argument("file2", type=Path, help="Second file to compare") - args = parser.parse_args() - - if not compare_outputs(args.file1, args.file2): - sys.exit(1) - -if __name__ == "__main__": - main() - diff --git a/flow/actorcompiler_py/errors.py b/flow/actorcompiler_py/errors.py deleted file mode 100644 index 6ad783a7246..00000000000 --- a/flow/actorcompiler_py/errors.py +++ /dev/null @@ -1,15 +0,0 @@ -from __future__ import annotations - - -class ActorCompilerError(Exception): - """Exception raised for parser or compiler errors with source locations.""" - - def __init__(self, source_line: int, message: str, *args: object) -> None: - if args: - message = message.format(*args) - super().__init__(message) - self.source_line = source_line - - def __str__(self) -> str: - return f"{super().__str__()} (line {self.source_line})" - diff --git a/flow/actorcompiler_py/parse_tree.py b/flow/actorcompiler_py/parse_tree.py deleted file mode 100644 index 8aa697b7bae..00000000000 --- a/flow/actorcompiler_py/parse_tree.py +++ /dev/null @@ -1,221 +0,0 @@ -from __future__ import annotations - -from abc import ABC, abstractmethod -from dataclasses import dataclass, field -from typing import List, Optional, Sequence - - -@dataclass -class VarDeclaration: - type: str = "" - name: str = "" - initializer: Optional[str] = "" - initializer_constructor_syntax: bool = False - - -@dataclass -class Statement(ABC): - first_source_line: int = 0 - - @abstractmethod - def contains_wait(self) -> bool: - pass - - -@dataclass -class CodeBlock(Statement): - statements: Sequence[Statement] = field(default_factory=list) - - def __str__(self) -> str: - joined = "\n".join(str(stmt) for stmt in self.statements) - return f"CodeBlock\n{joined}\nEndCodeBlock" - - def contains_wait(self) -> bool: - return any(stmt.contains_wait() for stmt in self.statements) - - -@dataclass -class PlainOldCodeStatement(Statement): - code: str = "" - - def __str__(self) -> str: - return self.code - - def contains_wait(self) -> bool: - return False - - -@dataclass -class StateDeclarationStatement(Statement): - decl: VarDeclaration = field(default_factory=VarDeclaration) - - def __str__(self) -> str: - if self.decl.initializer_constructor_syntax: - return f"State {self.decl.type} {self.decl.name}({self.decl.initializer});" - return f"State {self.decl.type} {self.decl.name} = {self.decl.initializer};" - - def contains_wait(self) -> bool: - return False - - -@dataclass -class WhileStatement(Statement): - expression: str = "" - body: Statement = field(default_factory=CodeBlock) - - def contains_wait(self) -> bool: - return self.body.contains_wait() - - -@dataclass -class ForStatement(Statement): - init_expression: str = "" - cond_expression: str = "" - next_expression: str = "" - body: Statement = field(default_factory=CodeBlock) - - def contains_wait(self) -> bool: - return self.body.contains_wait() - - -@dataclass -class RangeForStatement(Statement): - range_expression: str = "" - range_decl: str = "" - body: Statement = field(default_factory=CodeBlock) - - def contains_wait(self) -> bool: - return self.body.contains_wait() - - -@dataclass -class LoopStatement(Statement): - body: Statement = field(default_factory=CodeBlock) - - def __str__(self) -> str: - return f"Loop {self.body}" - - def contains_wait(self) -> bool: - return self.body.contains_wait() - - -@dataclass -class BreakStatement(Statement): - def contains_wait(self) -> bool: - return False - - -@dataclass -class ContinueStatement(Statement): - def contains_wait(self) -> bool: - return False - - -@dataclass -class IfStatement(Statement): - expression: str = "" - constexpr: bool = False - if_body: Statement = field(default_factory=CodeBlock) - else_body: Optional[Statement] = None - - def contains_wait(self) -> bool: - return self.if_body.contains_wait() or ( - self.else_body is not None and self.else_body.contains_wait() - ) - - -@dataclass -class ReturnStatement(Statement): - expression: str = "" - - def __str__(self) -> str: - return f"Return {self.expression}" - - def contains_wait(self) -> bool: - return False - - -@dataclass -class WaitStatement(Statement): - result: VarDeclaration = field(default_factory=VarDeclaration) - future_expression: str = "" - result_is_state: bool = False - is_wait_next: bool = False - - def __str__(self) -> str: - return f"Wait {self.result.type} {self.result.name} <- {self.future_expression} ({'state' if self.result_is_state else 'local'})" - - def contains_wait(self) -> bool: - return True - - -@dataclass -class ChooseStatement(Statement): - body: Statement = field(default_factory=CodeBlock) - - def __str__(self) -> str: - return f"Choose {self.body}" - - def contains_wait(self) -> bool: - return self.body.contains_wait() - - -@dataclass -class WhenStatement(Statement): - wait: WaitStatement = field(default_factory=WaitStatement) - body: Optional[Statement] = None - - def __str__(self) -> str: - return f"When ({self.wait}) {self.body}" - - def contains_wait(self) -> bool: - return True - - -@dataclass -class TryStatement(Statement): - @dataclass - class Catch: - expression: str = "" - body: Statement = field(default_factory=CodeBlock) - first_source_line: int = 0 - - try_body: Statement = field(default_factory=CodeBlock) - catches: List["TryStatement.Catch"] = field(default_factory=list) - - def contains_wait(self) -> bool: - if self.try_body.contains_wait(): - return True - return any(c.body.contains_wait() for c in self.catches) - - -@dataclass -class ThrowStatement(Statement): - expression: str = "" - - def contains_wait(self) -> bool: - return False - - -@dataclass -class Actor: - attributes: List[str] = field(default_factory=list) - return_type: Optional[str] = None - name: str = "" - enclosing_class: Optional[str] = None - parameters: Sequence[VarDeclaration] = field(default_factory=list) - template_formals: Optional[Sequence[VarDeclaration]] = None - body: CodeBlock = field(default_factory=CodeBlock) - source_line: int = 0 - is_static: bool = False - _is_uncancellable: bool = False - test_case_parameters: Optional[str] = None - name_space: Optional[str] = None - is_forward_declaration: bool = False - is_test_case: bool = False - - def is_cancellable(self) -> bool: - return self.return_type is not None and not self._is_uncancellable - - def set_uncancellable(self) -> None: - self._is_uncancellable = True diff --git a/flow/bench/README.md b/flow/bench/README.md index f3c4b5aa52f..348ec9f6cfd 100644 --- a/flow/bench/README.md +++ b/flow/bench/README.md @@ -9,7 +9,7 @@ The benchmark suite is split into three executables: These binaries can be used to microbenchmark parts of the FoundationDB code without always pulling in higher-level dependencies. Specifically, they can be used to: -- Test the performance effects of changes to the actor compiler or to the `flow` and `fdbrpc` libraries +- Test the performance effects of changes to the `flow` and `fdbrpc` libraries - Test the performance of various uses of the `flow` and `fdbrpc` libraries - Find areas for improvement in the `flow` and `fdbrpc` libraries - Compare `flow`/`fdbrpc` primitives to alternatives provided by the standard library or other third-party libraries. diff --git a/flow/include/flow/CoroutinesImpl.h b/flow/include/flow/CoroutinesImpl.h index ed108ffc553..af689d85621 100644 --- a/flow/include/flow/CoroutinesImpl.h +++ b/flow/include/flow/CoroutinesImpl.h @@ -675,6 +675,9 @@ struct AwaitableFuture [[no_unique_address]] std::conditional_t, Empty> store; AwaitableFuture(const FutureType& f, PromiseType* pt) : future(f), pt(pt) {} + AwaitableFuture(FutureStream&& f, PromiseType* pt) + requires(IsStream) + : future(std::move(f)), pt(pt) {} void fire(FutureValue const& value) override { if constexpr (IsStream) { @@ -696,9 +699,9 @@ struct AwaitableFuture pt->resume(); } - auto getCallbackFuture() const { + auto getCallbackFuture() { if constexpr (IsStream) { - return future; + return std::move(future); } else { return StrictFuture(future); } @@ -983,6 +986,11 @@ struct CoroPromiseBase : CoroReturn { return coro::AwaitableFutureErrorOr{ std::move(future.future), self() }; } + template + auto await_transform(FutureStream&& futureStream) { + return coro::AwaitableFuture{ std::move(futureStream), self() }; + } + template auto await_transform(const FutureStream& futureStream) { return coro::AwaitableFuture{ futureStream, self() }; @@ -1190,6 +1198,11 @@ struct AsyncResultPromise return coro::AwaitableFutureErrorOr{ std::move(future.future), this }; } + template + auto await_transform(FutureStream&& futureStream) { + return coro::AwaitableFuture{ std::move(futureStream), this }; + } + template auto await_transform(const FutureStream& futureStream) { return coro::AwaitableFuture{ futureStream, this }; @@ -1300,6 +1313,11 @@ struct AsyncGeneratorPromise { return coro::AwaitableFuture{ future, this }; } + template + auto await_transform(FutureStream&& futureStream) { + return coro::AwaitableFuture{ std::move(futureStream), this }; + } + template auto await_transform(const FutureStream& futureStream) { return coro::AwaitableFuture{ futureStream, this }; diff --git a/flow/include/flow/FlowThread.h b/flow/include/flow/FlowThread.h index 444bc9c3160..62c91f72777 100644 --- a/flow/include/flow/FlowThread.h +++ b/flow/include/flow/FlowThread.h @@ -294,10 +294,4 @@ class ThreadReturnPromiseStream { ThreadNotifiedQueue* queue; }; -// Fixes IDE build. -#ifndef NO_INTELLISENSE -template -T waitNext(const ThreadFutureStream&); -#endif - #endif diff --git a/flow/include/flow/ObjectSerializer.h b/flow/include/flow/ObjectSerializer.h index f85176ebb2f..09465d28ac2 100644 --- a/flow/include/flow/ObjectSerializer.h +++ b/flow/include/flow/ObjectSerializer.h @@ -147,6 +147,12 @@ class ArenaObjectReader : public _ObjectReader { vo.read(*this); } + template + ArenaObjectReader(Arena&& arena, const StringRef& input, VersionOptions vo) + : _data(input.begin()), _arena(std::move(arena)) { + vo.read(*this); + } + const uint8_t* data() { return _data; } Arena& arena() { return _arena; } diff --git a/flow/include/flow/PriorityMultiLock.h b/flow/include/flow/PriorityMultiLock.h index 61172293588..bd8a2b7caac 100644 --- a/flow/include/flow/PriorityMultiLock.h +++ b/flow/include/flow/PriorityMultiLock.h @@ -22,10 +22,11 @@ #include "flow/flow.h" #include +#include #define PRIORITYMULTILOCK_DEBUG 0 -#if PRIORITYMULTILOCK_DEBUG || !defined(NO_INTELLISENSE) +#if PRIORITYMULTILOCK_DEBUG #define pml_debug_printf(...) \ if (now() > 0) { \ printf("pml line=%04d ", __LINE__); \ @@ -75,6 +76,35 @@ class PriorityMultiLock : public ReferenceCounted { Promise promise; }; + // Move-only holder for an immediately granted lock. It avoids the Promise and release callback needed by Lock. + class Releaser : NonCopyable { + Reference owner; + int priority = -1; + + friend class PriorityMultiLock; + Releaser(Reference owner, int priority) : owner(std::move(owner)), priority(priority) {} + + public: + Releaser() = default; + Releaser(Releaser&& rhs) noexcept : owner(std::move(rhs.owner)), priority(std::exchange(rhs.priority, -1)) {} + Releaser& operator=(Releaser&&) = delete; + + void release() { + if (!owner) { + return; + } + + // Waking the runner can synchronously resume waiters, so clear this holder first. + auto self = std::move(owner); + int index = std::exchange(priority, -1); + auto* p = &self->priorities[index]; + releaseRunner(std::move(self), p); + } + + bool isLocked() const { return owner.isValid(); } + ~Releaser() { release(); } + }; + PriorityMultiLock(int concurrency, std::string weights) : PriorityMultiLock(concurrency, parseStringToVector(weights, ',')) {} @@ -92,6 +122,32 @@ class PriorityMultiLock : public ReferenceCounted { ~PriorityMultiLock() { kill(); } + // Grants an immediately available lock synchronously without creating a Future; never enqueues a waiter. + Optional tryLock(int priority = 0) { + if (killed) + throw broken_promise(); + + Priority& p = priorities[priority]; + if (!p.queue.empty() || available <= 0) { + return {}; + } + + // With no waiters, available already proves this priority has capacity; skip the weight division. + if (totalPendingWeights != 0) { + totalPendingWeights += p.weight; + bool hasCapacity = p.runners < currentCapacity(p.weight); + totalPendingWeights -= p.weight; + if (!hasCapacity) { + return {}; + } + } + + ++p.runners; + --available; + pml_debug_printf("lock nowait priority %d %s\n", priority, toString().c_str()); + return Releaser(Reference::addRef(this), priority); + } + Future lock(int priority = 0) { if (killed) throw broken_promise(); diff --git a/flow/include/flow/TDMetric.h b/flow/include/flow/TDMetric.h index 709b9a5ff21..a32a6da6985 100644 --- a/flow/include/flow/TDMetric.h +++ b/flow/include/flow/TDMetric.h @@ -56,11 +56,12 @@ struct MetricNameRef { int expectedSize() const { return type.expectedSize() + name.expectedSize(); } inline int compare(MetricNameRef const& r) const { - int cmp; - if ((cmp = type.compare(r.type))) { + int cmp = type.compare(r.type); + if (cmp) { return cmp; } - if ((cmp = name.compare(r.name))) { + cmp = name.compare(r.name); + if (cmp) { return cmp; } return id.compare(r.id); @@ -396,12 +397,6 @@ template struct Descriptor { // Specialize Descriptor next to each metric payload struct, typically by inheriting from // DescribeType, ...>. -#ifndef NO_INTELLISENSE - using fields = std::tuple<>; - using field_indexes = tuple_indexes_t; - - static StringRef typeName() { return ""_sr; } -#endif }; // String literals need a wrapper type before they can be used as non-type template parameters. @@ -974,7 +969,6 @@ struct EventMetric final : E, ReferenceCounted>, MetricUtil void logFields(index_sequence, uint64_t t, int64_t l, bool& overflow, int64_t& bytes) { -#ifdef NO_INTELLISENSE auto _ = { (std::get(values).log( std::tuple_element::fields>::type::get(static_cast(*this)), t, @@ -983,23 +977,18 @@ struct EventMetric final : E, ReferenceCounted>, MetricUtil void initFields(index_sequence) { -#ifdef NO_INTELLISENSE auto _ = { (std::get(values).init(), Void())... }; (void)_; -#endif } template void nextKeys(index_sequence, uint64_t t, int64_t l) { -#ifdef NO_INTELLISENSE auto _ = { (std::get(values).nextKey(t, l), Void())... }; (void)_; -#endif } void flushData(MetricKeyRef const& mk, uint64_t rollTime, MetricBatch& batch) override { @@ -1013,10 +1002,8 @@ struct EventMetric final : E, ReferenceCounted>, MetricUtil void flushFields(index_sequence, MetricKeyRef const& mk, uint64_t rollTime, MetricBatch& batch) { -#ifdef NO_INTELLISENSE auto _ = { (std::get(values).flushField(mk, rollTime, batch), Void())... }; (void)_; -#endif } void rollMetric(uint64_t t) override { @@ -1026,10 +1013,8 @@ struct EventMetric final : E, ReferenceCounted>, MetricUtil void rollFields(index_sequence, uint64_t t) { -#ifdef NO_INTELLISENSE auto _ = { (std::get(values).rollMetric(t), Void())... }; (void)_; -#endif } void registerFields(MetricKeyRef const& mk, std::vector>& fieldKeys) override { @@ -1039,10 +1024,8 @@ struct EventMetric final : E, ReferenceCounted>, MetricUtil void registerFields(index_sequence, const MetricKeyRef& mk, std::vector>& fieldKeys) { -#ifdef NO_INTELLISENSE auto _ = { (std::get(values).registerField(mk, fieldKeys), Void())... }; (void)_; -#endif } private: diff --git a/flow/include/flow/TLSConfig.h b/flow/include/flow/TLSConfig.h index a6ee41072cf..1cdad46037e 100644 --- a/flow/include/flow/TLSConfig.h +++ b/flow/include/flow/TLSConfig.h @@ -84,8 +84,6 @@ struct Criteria { enum class TLSEndpointType { UNSET = 0, CLIENT, SERVER }; class TLSConfig; -template -class LoadAsyncActorState; class LoadedTLSConfig { public: @@ -124,8 +122,6 @@ class LoadedTLSConfig { TLSEndpointType endpointType = TLSEndpointType::UNSET; friend class TLSConfig; - template - friend class LoadAsyncActorState; }; class TLSConfig { @@ -212,8 +208,6 @@ class TLSConfig { private: #endif static Future loadAsync(const TLSConfig* self); // FIXME - template - friend class LoadAsyncActorState; std::string tlsCertPath, tlsKeyPath, tlsCAPath; std::string tlsCertBytes, tlsKeyBytes, tlsCABytes; diff --git a/flow/include/flow/TreeBenchmark.h b/flow/include/flow/TreeBenchmark.h index 2ea2c441df4..10b230ccb09 100644 --- a/flow/include/flow/TreeBenchmark.h +++ b/flow/include/flow/TreeBenchmark.h @@ -89,6 +89,7 @@ void treeBenchmark(T& tree, F generateKey) { int keyCount = 1000000; std::vector keys; + keys.reserve(keyCount); for (int i = 0; i < keyCount; i++) { keys.push_back(generateKey()); } diff --git a/flow/include/flow/UnitTest.h b/flow/include/flow/UnitTest.h index d78f75d4d44..e786f4fb5d2 100644 --- a/flow/include/flow/UnitTest.h +++ b/flow/include/flow/UnitTest.h @@ -46,8 +46,8 @@ * return Void(); * } * - * In an `.actor.cpp` file, the body of a TEST_CASE is an actor (may contain `wait`, `state`, etc) - * In a `.cpp` file, the body of a TEST_CASE is an ordinary function returning a Future + * The body of a TEST_CASE returns a Future. It may be an ordinary function + * or a C++ coroutine using `co_await` and `co_return`. * * Our tools for actually executing tests are external to flow (and use g_unittests to find test cases). * See the `UnitTestWorkload` class. @@ -121,7 +121,6 @@ extern bool noUnseed; #ifdef FLOW_DISABLE_UNIT_TESTS #define TEST_CASE(name) static Future FILE_UNIQUE_NAME(disabled_testcase_func)(const UnitTestParameters& params) -#define ACTOR_TEST_CASE(actorname, name) #else @@ -132,12 +131,6 @@ extern bool noUnseed; } \ static Future FILE_UNIQUE_NAME(testcase_func)(const UnitTestParameters& params) -// ACTOR_TEST_CASE generated by actorcompiler; don't use directly -#define ACTOR_TEST_CASE(actorname, name) \ - namespace { \ - UnitTest APPEND(testcase_, actorname)(name, __FILE__, __LINE__, &actorname); \ - } - #endif #endif diff --git a/flow/include/flow/actorcompiler.h b/flow/include/flow/actorcompiler.h deleted file mode 100644 index 3ceaae2828c..00000000000 --- a/flow/include/flow/actorcompiler.h +++ /dev/null @@ -1,83 +0,0 @@ -/* - * actorcompiler.h - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -#ifdef POST_ACTOR_COMPILER -#ifndef FLOW_DEFINED_WAIT_AND_WAIT_NEXT -#define FLOW_DEFINED_WAIT_AND_WAIT_NEXT - -// These should all be re-written by the actor compiler. We don't want to -// accidentally call them from something that's not an actor. `wait` is such a -// common identifier that `wait` calls outside ACTORs might accidentally -// compile. -template -T wait(const Future&) = delete; -void wait(const Never&) = delete; -template -T waitNext(const FutureStream&) = delete; - -#endif -#endif - -#ifndef POST_ACTOR_COMPILER - -template -class Future; -class Never; -template -class FutureStream; - -// These are for intellisense to do proper type inferring, etc. They are no included at build time. -#ifndef NO_INTELLISENSE -#define ACTOR -#define SWIFT_ACTOR -#define state -#define UNCANCELLABLE -#define choose if (1) -#define when(...) for (__VA_ARGS__;;) -template -T wait(const Future&); -void wait(const Never&); -template -T waitNext(const FutureStream&); -#endif - -#endif - -#define loop while (true) - -#ifdef NO_INTELLISENSE -#define THIS this -#define THIS_ADDR uintptr_t(this) -#else -#define THIS nullptr -#define THIS_ADDR uintptr_t(nullptr) -#endif - -#ifdef _MSC_VER -#pragma warning(disable : 4355) // 'this' : used in base member initializer list -#endif - -// Currently, #ifdef can't be used inside actors, so define no-op versions of these valgrind -// functions if valgrind is not defined -#ifndef VALGRIND -#define VALGRIND_MAKE_MEM_UNDEFINED(x, y) -#define VALGRIND_MAKE_MEM_DEFINED(x, y) -#define VALGRIND_CHECK_MEM_IS_DEFINED(x, y) 0 -#endif diff --git a/flow/include/flow/flat_buffers.h b/flow/include/flow/flat_buffers.h index 715a08123e6..e27deb998f4 100644 --- a/flow/include/flow/flat_buffers.h +++ b/flow/include/flow/flat_buffers.h @@ -1397,6 +1397,37 @@ struct EnsureTable T t; }; +template +struct EnsureTableRef + : std::conditional_t::value, detail::YesFileIdentifier, detail::NoFileIdentifier> { + EnsureTableRef() = default; + explicit EnsureTableRef(const T& t) : t(&t) {} + + template + void serialize(Archive& ar) { + if constexpr (is_fb_function) { + // Vtable collection walks default-constructed union alternatives, which have no referenced value. + T value{}; + if constexpr (detail::expect_serialize_member) { + if constexpr (serializable_traits::value) { + serializable_traits::serialize(ar, value); + } else { + value.serialize(ar); + } + } else { + serializer(ar, value); + } + } else { + serializer(ar, const_cast(*t)); + } + } + + const T& asUnderlyingType() const { return *t; } + +private: + const T* t = nullptr; +}; + namespace detail { // Ensure if there's a LoadSaveHelper specialization available for T it gets used. @@ -1428,4 +1459,23 @@ struct LoadSaveHelper, Context> : Context { LoadSaveHelper, Context> wrapInTable; }; +template +struct LoadSaveHelper, Context> : Context { + explicit LoadSaveHelper(const Context& context) : Context(context), alreadyATable(context), wrapInTable(context) {} + + template + RelativeOffset save(const EnsureTableRef& member, Writer& writer, const VTableSet* vtables) { + if constexpr (expect_serialize_member) { + return alreadyATable.save(member.asUnderlyingType(), writer, vtables); + } else { + FakeRoot t{ const_cast(member.asUnderlyingType()) }; + return wrapInTable.save(t, writer, vtables); + } + } + +private: + LoadSaveHelper alreadyATable; + LoadSaveHelper, Context> wrapInTable; +}; + } // namespace detail diff --git a/flow/include/flow/flow.h b/flow/include/flow/flow.h index 11bcfd3dfc7..86926a3c56d 100644 --- a/flow/include/flow/flow.h +++ b/flow/include/flow/flow.h @@ -1008,12 +1008,6 @@ SWIFT_CONFORMS_TO_PROTOCOL(flow_swift.FlowFutureOps) Future(Never) : sav(new SAV(1, 0)) { sav->send(Never()); } Future(const Error& error) : sav(new SAV(1, 0)) { sav->sendError(error); } -#ifndef NO_INTELLISENSE - template - requires(std::is_assignable_v) - Future(const U&) {} -#endif - ~Future() { if (sav) sav->delFutureRef(); diff --git a/flow/include/flow/unactorcompiler.h b/flow/include/flow/unactorcompiler.h deleted file mode 100644 index 088e18be82c..00000000000 --- a/flow/include/flow/unactorcompiler.h +++ /dev/null @@ -1,37 +0,0 @@ -/* - * unactorcompiler.h - * - * This source file is part of the FoundationDB open source project - * - * Copyright 2013-2026 Apple Inc. and the FoundationDB project authors - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -#ifndef POST_ACTOR_COMPILER - -#ifndef NO_INTELLISENSE -#undef ACTOR -#undef SWIFT_ACTOR -#undef state -#undef UNCANCELLABLE -#undef choose -#undef when -#endif - -#undef THIS -#undef THIS_ADDR - -#endif - -// loop is still defined diff --git a/flow/no_intellisense.opt b/flow/no_intellisense.opt deleted file mode 100644 index 74512516f65..00000000000 --- a/flow/no_intellisense.opt +++ /dev/null @@ -1 +0,0 @@ -/D "NO_INTELLISENSE=1" \ No newline at end of file diff --git a/packaging/docker/Dockerfile b/packaging/docker/Dockerfile index 9f2370aad09..d0b894f7165 100644 --- a/packaging/docker/Dockerfile +++ b/packaging/docker/Dockerfile @@ -192,6 +192,7 @@ ENV FDB_COORDINATOR="" ENV FDB_COORDINATOR_PORT=4500 ENV FDB_CLUSTER_FILE_CONTENTS="" ENV FDB_PROCESS_CLASS=unset +ENV FDB_IP_VERSION=4 ENTRYPOINT ["/usr/bin/tini", "-g", "--", "/var/fdb/scripts/fdb.bash"] FROM foundationdb-base AS mako diff --git a/packaging/docker/fdb.bash b/packaging/docker/fdb.bash index 086b08545dc..fbc7e6ed5fe 100755 --- a/packaging/docker/fdb.bash +++ b/packaging/docker/fdb.bash @@ -25,8 +25,12 @@ function create_cluster_file() { mkdir -p "$(dirname $FDB_CLUSTER_FILE)" if [[ -n $FDB_COORDINATOR ]]; then - coordinator_ip=$(dig +short "$FDB_COORDINATOR") - if [[ -z "$coordinator_ip" ]]; then + if [[ "$FDB_IP_VERSION" == '4' ]]; then + coordinator_ip="$(getent ahostsv4 "$FDB_COORDINATOR" | awk '{ print $1; exit }')" + elif [[ "$FDB_IP_VERSION" == '6' ]]; then + coordinator_ip="[$(getent ahostsv6 "$FDB_COORDINATOR" | awk '{ print $1; exit }')]" + fi + if [[ -z "$coordinator_ip" ]] || [[ "$coordinator_ip" == "[]" ]]; then echo "Failed to look up coordinator address for $FDB_COORDINATOR" 1>&2 exit 1 fi @@ -39,6 +43,10 @@ function create_cluster_file() { echo "Overwriting existing clusterfile." 1>&2 fi echo "$FDB_CLUSTER_FILE_CONTENTS" > "$FDB_CLUSTER_FILE" + if [[ $? != 0 ]]; then + echo "Unable to write to FDB_CLUSTER_FILE." 1>&2 + exit 1 + fi elif [[ ! -w "$FDB_CLUSTER_FILE" ]]; then # fdbserver requires write permissions to clusterfile, or it may *eventually* fail due to cluster migrations. # https://apple.github.io/foundationdb/administration.html#required-permissions @@ -47,26 +55,51 @@ function create_cluster_file() { else echo "Using existing clusterfile at \"$FDB_CLUSTER_FILE\"." 1>&2 fi +} - if (( $? != 0 )); then - echo "Unable to write to FDB_CLUSTER_FILE." 1>&2 - exit 1 - fi +function first_hostname_with_str() { + for addr in $(hostname -I); do + if [[ $addr == *"$1"* ]]; then + echo "$addr" + return 0 + fi + done + return 1 } function create_server_environment() { - env_file=/var/fdb/.fdbenv + FDB_IP_VERSION=${FDB_IP_VERSION:-4} + if [[ "$FDB_IP_VERSION" == '4' ]]; then + FDB_LISTEN_IP="${FDB_LISTEN_IP:-0.0.0.0}" + elif [[ "$FDB_IP_VERSION" == '6' ]]; then + FDB_LISTEN_IP="[${FDB_LISTEN_IP:-::}]" + else + echo "Unknown FDB_IP_VERSION \"$FDB_IP_VERSION\"" 1>&2 + exit 1 + fi if [[ "$FDB_NETWORKING_MODE" == "host" ]]; then - public_ip=127.0.0.1 + if [[ "$FDB_IP_VERSION" == '4' ]]; then + public_ip='127.0.0.1' + elif [[ "$FDB_IP_VERSION" == '6' ]]; then + public_ip='[::1]' + fi elif [[ "$FDB_NETWORKING_MODE" == "container" ]]; then - public_ip=$(hostname -i | awk '{print $1}') + if [[ "$FDB_IP_VERSION" == '4' ]]; then + public_ip="${FDB_PUBLIC_IP:-$(first_hostname_with_str '.')}" + elif [[ "$FDB_IP_VERSION" == '6' ]]; then + public_ip="[${FDB_PUBLIC_IP:-$(first_hostname_with_str ':')}]" + fi + if [[ $? != 0 ]]; then + echo "No valid IPv${FDB_IP_VERSION} address" 1>&2 + exit 1 + fi else echo "Unknown FDB Networking mode \"$FDB_NETWORKING_MODE\"" 1>&2 exit 1 fi + export PUBLIC_IP="$public_ip" - echo "export PUBLIC_IP=$public_ip" > $env_file # Set default cluster file contents only if no other configuration is specified. if [[ (! -s "$FDB_CLUSTER_FILE") && -z "$FDB_CLUSTER_FILE_CONTENTS" && -z "$FDB_COORDINATOR" ]]; then echo "Warning: No configuration available, falling back to self-coordinated." 1>&2 @@ -77,9 +110,8 @@ function create_server_environment() { } create_server_environment -source /var/fdb/.fdbenv -echo "Starting FDB server on $PUBLIC_IP:$FDB_PORT" -fdbserver --listen-address 0.0.0.0:"$FDB_PORT" --public-address "$PUBLIC_IP:$FDB_PORT" \ +echo "Starting FDB server on $PUBLIC_IP:$FDB_PORT, listening on $FDB_LISTEN_IP:$FDB_PORT" +fdbserver --listen-address "$FDB_LISTEN_IP:$FDB_PORT" --public-address "$PUBLIC_IP:$FDB_PORT" \ --datadir /var/fdb/data --logdir /var/fdb/logs \ --locality-zoneid="$(hostname)" --locality-machineid="$(hostname)" --class "$FDB_PROCESS_CLASS" --knob_disable_posix_kernel_aio=1 \ "$@" diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 72d5f3f5cd7..cee7ded19b1 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -157,7 +157,10 @@ if(WITH_PYTHON) add_fdb_test(TEST_FILES slow/BackupS3BlobBulkLoadRestore.toml) add_fdb_test(TEST_FILES slow/BackupS3BlobBulkLoadRestoreWithChaos.toml) add_fdb_test(TEST_FILES slow/BackupS3BlobBulkLoadRestoreMultiRange.toml) + add_fdb_test(TEST_FILES slow/BackupS3BlobBulkLoadRestoreNarrowFleet.toml) add_fdb_test(TEST_FILES fast/BulkLoading.toml) + add_fdb_test(TEST_FILES fast/BulkLoadingDestTeamFailure.toml) + add_fdb_test(TEST_FILES fast/BulkLoadingDestTeamFailureExhausted.toml) add_fdb_test(TEST_FILES slow/S3Client.toml) add_fdb_test(TEST_FILES slow/S3ClientWorkloadWithChaos.toml) add_fdb_test(TEST_FILES fast/CloggedSideband.toml) @@ -253,6 +256,7 @@ if(WITH_PYTHON) add_fdb_test(TEST_FILES slow/GcGenerations.toml) add_fdb_test(TEST_FILES fast/KillRegionCycle.toml) add_fdb_test(TEST_FILES rare/ClogRemoteTLog.toml) + add_fdb_test(TEST_FILES slow/DegradedMultiRegionStatus.toml) endif() if(WITH_ROCKSDB) diff --git a/tests/fast/BulkLoadingDestTeamFailure.toml b/tests/fast/BulkLoadingDestTeamFailure.toml new file mode 100644 index 00000000000..b1ebac68e67 --- /dev/null +++ b/tests/fast/BulkLoadingDestTeamFailure.toml @@ -0,0 +1,59 @@ +# BulkLoad retry when a data move loses its destination team +# +# An unhealthy destination team is recoverable -- another team can be chosen -- but a bulkload task's +# key-values exist only in the dump until an attempt ingests them, so abandoning the task removes that +# range from the database. Injects fewer failures than the re-dispatch budget allows, so retries win. +# +# BulkLoadingWorkload compares every restored key-value against what it wrote, and excludes ranges +# whose task ended in Error, so the assertion is: the load terminates and surviving ranges are correct. +# +# See BulkLoadingDestTeamFailureExhausted.toml for the case where the budget runs out. + +[configuration] +storageEngineExcludeTypes = ["ssd-sharded-rocksdb"] #FIXME: remove after allowing to do bulkloading with fetchKey and shardedrocksdb + +disableTss = true # TODO(BulkLoad): support TSS. finishMoveShard should wait for TSS if a data move is bulkload data move. + +# See BulkLoading.toml: 'triple' keeps the simulator from picking a topology where bulk-load team +# selection cannot find a disjoint candidate team, which would stall the load rather than test it. +config = 'triple' + +machineCount = 10 +extraStorageMachineCountPerDC = 5 +datacenters = 1 +generateFearless = false + +[[knobs]] +shard_encode_location_metadata = true + +# Rely on RangeLock +enable_read_lock_on_range = true + +# Do not support version vector +enable_version_vector = false +enable_version_vector_tlog_unicast = false +enable_version_vector_reply_recovery = false + +min_byte_sampling_probability = 0.5 + +cc_enforce_use_unfit_dd_in_sim = true + +disable_audit_storage_final_replica_check_in_sim = true + +max_trace_lines = 5000000 + +# Number of bulkload data moves told their destination team went unhealthy (0 elsewhere). +bulkload_sim_inject_dest_team_failures = 3 + +# Pin the budget above the injection count so retries genuinely win. Left at its default this knob +# buggifies to 1-3, which would silently turn this into a second copy of the Exhausted test. +dd_bulkload_max_retryable_redispatch = 20 + + +[[test]] +testTitle = 'BulkLoadingDestTeamFailureWorkload' +useDB = true +waitForQuiescence = false + + [[test.workload]] + testName = 'BulkLoadingWorkload' diff --git a/tests/fast/BulkLoadingDestTeamFailureExhausted.toml b/tests/fast/BulkLoadingDestTeamFailureExhausted.toml new file mode 100644 index 00000000000..b318aed76d4 --- /dev/null +++ b/tests/fast/BulkLoadingDestTeamFailureExhausted.toml @@ -0,0 +1,60 @@ +# BulkLoad give-up path when retries cannot clear a destination-team failure +# +# Complement of BulkLoadingDestTeamFailure.toml: injects more recoverable failures than the +# re-dispatch budget allows, so one task exhausts it and is marked Error. What must not happen either +# way is an endless retry loop, or a task reporting success without ingesting its data. +# +# The injection budget is spent on a single task, since the budget bounds one task's restartCount; a +# budget spread across tasks retries each a few times and never reaches the give-up path. +# +# Covers the data distribution half only. That a restore whose job ends in Error fails with +# restore_bulkload_failed rather than reporting success needs a restore-level test. + +[configuration] +storageEngineExcludeTypes = ["ssd-sharded-rocksdb"] #FIXME: remove after allowing to do bulkloading with fetchKey and shardedrocksdb + +disableTss = true # TODO(BulkLoad): support TSS. finishMoveShard should wait for TSS if a data move is bulkload data move. + +# See BulkLoading.toml: 'triple' keeps the simulator from picking a topology where bulk-load team +# selection cannot find a disjoint candidate team, which would stall the load rather than test it. +config = 'triple' + +machineCount = 10 +extraStorageMachineCountPerDC = 5 +datacenters = 1 +generateFearless = false + +[[knobs]] +shard_encode_location_metadata = true + +# Rely on RangeLock +enable_read_lock_on_range = true + +# Do not support version vector +enable_version_vector = false +enable_version_vector_tlog_unicast = false +enable_version_vector_reply_recovery = false + +min_byte_sampling_probability = 0.5 + +cc_enforce_use_unfit_dd_in_sim = true + +disable_audit_storage_final_replica_check_in_sim = true + +max_trace_lines = 5000000 + +# Exhaust the re-dispatch budget for one task. +bulkload_sim_inject_dest_team_failures = 40 + +# Pin the budget so the give-up path is reached deterministically rather than depending on whether +# buggify happened to lower it. +dd_bulkload_max_retryable_redispatch = 20 + + +[[test]] +testTitle = 'BulkLoadingDestTeamFailureExhaustedWorkload' +useDB = true +waitForQuiescence = false + + [[test.workload]] + testName = 'BulkLoadingWorkload' diff --git a/tests/fast/NativeCdcSharedTag.toml b/tests/fast/NativeCdcSharedTag.toml index 4c0d7991b05..5aae245d484 100644 --- a/tests/fast/NativeCdcSharedTag.toml +++ b/tests/fast/NativeCdcSharedTag.toml @@ -1,6 +1,10 @@ [configuration] -config = 'single' +config = 'single commit_proxies=1 grv_proxies=2' singleRegion = true +# Two proxies distinguish the persisted tag owner from the proxy receiving registration. +datacenters = 1 +machineCount = 12 +statelessProcessClassesPerDC = 3 buggify = false faultInjection = false @@ -22,6 +26,7 @@ connectionFailuresDisableDuration = 1000000 keyCount = 6 writesPerRound = 3 rounds = 6 + testTagOwnership = true drainProbability = 1.0 delayBetweenRounds = 0.1 operationTimeout = 120.0 diff --git a/tests/fast/RangeLocking.toml b/tests/fast/RangeLocking.toml index 7b83297f788..d66cb08b55a 100644 --- a/tests/fast/RangeLocking.toml +++ b/tests/fast/RangeLocking.toml @@ -3,6 +3,7 @@ [[knobs]] enable_read_lock_on_range = true transaction_lock_rejection_retriable = false +krm_get_range_limit = 4 enable_version_vector = false enable_version_vector_tlog_unicast = false enable_version_vector_reply_recovery = false diff --git a/tests/rare/RedwoodCorrectnessBTree.toml b/tests/rare/RedwoodCorrectnessBTree.toml index 1ed30ef14bc..feb1c993410 100644 --- a/tests/rare/RedwoodCorrectnessBTree.toml +++ b/tests/rare/RedwoodCorrectnessBTree.toml @@ -19,3 +19,15 @@ startDelay = 0 testName = 'UnitTests' maxTestCases = 1 testsMatching = '/redwood/correctness/btreeCloseWithQueuedCommits' + +[[test]] +testTitle = 'RedwoodSeekExactKey' +useDB = false +startDelay = 0 + + [[test.workload]] + testName = 'UnitTests' + maxTestCases = 1 + testsMatching = 'Lredwood/correctness/seekExactKey' + maxRunTimeWallTime = 250.0 + maxRunTimeSanitizerModeWallTime = 800.0 diff --git a/tests/slow/BackupS3BlobBulkLoadRestore.toml b/tests/slow/BackupS3BlobBulkLoadRestore.toml index 7951ffd656f..c8a648d1fcd 100644 --- a/tests/slow/BackupS3BlobBulkLoadRestore.toml +++ b/tests/slow/BackupS3BlobBulkLoadRestore.toml @@ -32,6 +32,13 @@ minimumRegions = 1 # Ensure enough storage servers for non-overlapping BulkLoad teams extraMachineCountDC = 3 +# Dedicated storage-class machines so bulk-load team selection can find a team disjoint from the +# source: it rejects any team sharing a server with the source (DDTeamCollection). Without these, +# runs fail with DDBulkLoadEngineTaskGetTeamFailedToFindValidTeam reporting ValidTeamSize 0 with +# every candidate team duplicated, the task is marked Error, and the restore is incomplete. +# extraMachineCountDC above adds ordinary machines, which do not supply that capacity. Per DC, so +# this holds for the multi-region variants too. See tests/fast/BulkLoading.toml for the same guard. +extraStorageMachineCountPerDC = 5 # Explicit simple config - single replication, single region, explicit process counts config = "triple usable_regions=1 storage_engine=ssd-2 perpetual_storage_wiggle=0 commit_proxies=3 grv_proxies=3 resolvers=3 logs=3" diff --git a/tests/slow/BackupS3BlobBulkLoadRestoreMultiRange.toml b/tests/slow/BackupS3BlobBulkLoadRestoreMultiRange.toml index cf7c7d5cbcf..0b966ce4872 100644 --- a/tests/slow/BackupS3BlobBulkLoadRestoreMultiRange.toml +++ b/tests/slow/BackupS3BlobBulkLoadRestoreMultiRange.toml @@ -33,6 +33,13 @@ minimumRegions = 2 # Ensure enough storage servers for non-overlapping BulkLoad teams # Triple replication needs 6+ servers (3 src + 3 different dest) to avoid overlapping extraMachineCountDC = 3 +# Dedicated storage-class machines so bulk-load team selection can find a team disjoint from the +# source: it rejects any team sharing a server with the source (DDTeamCollection). Without these, +# runs fail with DDBulkLoadEngineTaskGetTeamFailedToFindValidTeam reporting ValidTeamSize 0 with +# every candidate team duplicated, the task is marked Error, and the restore is incomplete. +# extraMachineCountDC above adds ordinary machines, which do not supply that capacity. Per DC, so +# this holds for the multi-region variants too. See tests/fast/BulkLoading.toml for the same guard. +extraStorageMachineCountPerDC = 5 config = "triple usable_regions=1 storage_engine=ssd-2 perpetual_storage_wiggle=0 commit_proxies=3 grv_proxies=3 resolvers=3 logs=3" diff --git a/tests/slow/BackupS3BlobBulkLoadRestoreNarrowFleet.toml b/tests/slow/BackupS3BlobBulkLoadRestoreNarrowFleet.toml new file mode 100644 index 00000000000..d3500e6d68c --- /dev/null +++ b/tests/slow/BackupS3BlobBulkLoadRestoreNarrowFleet.toml @@ -0,0 +1,157 @@ +# BulkLoad restore onto a fleet with barely room for a disjoint destination team +# +# Variant of BackupS3BlobBulkLoadRestore.toml that cuts the extra storage machines its sibling tests rely +# on down to the minimum that still permits a legal destination team, so bulk-load task placement +# genuinely fails and the restore has to recover by narrowing task ranges rather than by having spare +# capacity. See the comment on the configuration below for where that minimum comes from. +# +# Reproduces the condition measured on a 100M-key validation cluster of ~40 storage servers: every +# candidate destination team is rejected for overlapping the source, so the team selector reports +# ValidTeamSize 0 with no unhealthy or ineligible teams involved. On that cluster the recovery path ran +# end to end - GetTeamFailedToFindValidTeam, then a relocation declared stuck, then a task split - and the +# restore completed with the consistency check passing. +# +# A run that exercises the fix logs GetTeamFailedToFindValidTeam followed by BulkLoad task splits, and no +# SplitDeclined. Exact counts depend on the dataset knobs below, so they are deliberately not asserted +# here; a fleet large enough to place a disjoint team makes the test vacuous rather than failing it. +# +# The rest of this file is inherited from BackupS3BlobBulkLoadRestore.toml. +# +# BulkLoad Validation Test +# Tests that BulkLoad restore produces identical results to traditional restore +# +# This test validates BulkLoad produces the same data as traditional restore: +# - Backup creates BOTH range files AND SST files (snapshotMode=2) +# - Range files are used by traditional restore +# - SST files are used by BulkLoad restore +# - Validation: compare BulkLoad restore vs traditional restore +# 1. Restore with --add-prefix to system keyspace using TRADITIONAL (rangefile) mode +# This creates a "known good" baseline +# 2. Clear normalKeys (original data) +# 3. Restore to normalKeys using BULKLOAD mode (reads SST files) +# 4. Run audit_storage validate_restore to compare: +# - BulkLoad-restored data (in normalKeys) +# - Traditional-restored data (in system key prefix) +# 5. Clean up validation prefix data +# - This validates BulkLoad produces identical results to traditional restore +# +# Configuration aligned with working tests/slow/BulkDumpingS3.toml + +testClass = "Backup" + +[configuration] +storageEngineExcludeTypes = ["ssd-sharded-rocksdb"] # FIXME: remove after allowing bulkloading with fetchKey and shardedrocksdb +disableTss = true # TODO(BulkLoad): support TSS + +# HA (multi-region) configuration - randomly enabled for test coverage +generateFearless = false +simpleConfig = false +minimumRegions = 1 +# singleRegion not set - allows random HA configuration + +# Ensure enough storage servers for non-overlapping BulkLoad teams +extraMachineCountDC = 3 +# Far below the sibling tests' extraStorageMachineCountPerDC = 5, but deliberately not zero. +# +# Bulk-load team selection rejects any team sharing a server with the source, and source is the union of +# the owners of every shard the task range spans, so a wide enough task range has no legal destination +# team. That state is DDBulkLoadEngineTaskGetTeamFailedToFindValidTeam with ValidTeamSize 0 and every +# candidate team duplicated, which no number of attempts can clear because each one recomputes the same +# source set. Provoking it is the point of this test, and a small fleet is how it is provoked. +# +# It is escapable only by narrowing the range, so the task is split. Narrowing bottoms out at a range +# inside one shard, whose source is that shard's StorageTeamSize machines -- so the fleet must be able to +# field a team of StorageTeamSize machines that avoids those, i.e. at least 2 * StorageTeamSize machines +# with data. Teams are built per machine, not per server, so extra storage servers on existing machines do +# not help. Omitting this knob entirely leaves 5 healthy machines against StorageTeamSize 3: all +# C(5,3) = 10 machine teams exist, every one of them necessarily touches the 3 source machines, and the +# restore fails no matter how finely the task is split. Measured that way on seed 22222 -- 252 consecutive +# placement failures, ValidTeamSize 0 throughout, TotalHealthyMachines 5, CurrentMachineTeams 10 of 10 +# possible. That is the test asserting something no fix can deliver. +# +# 1 lands the fleet exactly on the threshold, which is where the test has to sit to be worth anything: a +# source of StorageTeamSize machines leaves precisely one legal machine team, so placement still fails +# often enough to force splitting, yet a destination exists. Raising it to 2 was measured to make the test +# vacuous -- 17 of 17 seeds completed without a single placement failure or split, so it stopped testing +# the thing it exists to test. +extraStorageMachineCountPerDC = 1 +# The knobs below still matter for keeping the test honest: manifests are written one per shard +# (FileBackupAgent.cpp), so min_shard_bytes controls how many exist and +# manifest_count_max_per_bulkload_task how many a task groups. Left to their defaults this test degrades +# into a single task holding a single manifest over the whole key space, which is not worth asserting on. + +# Explicit simple config - single replication, single region, explicit process counts +config = "triple usable_regions=1 storage_engine=ssd-2 perpetual_storage_wiggle=0 commit_proxies=3 grv_proxies=3 resolvers=3 logs=3" + +# Disable buggify and fault injection to avoid interference with MockS3/BulkLoad +buggify = false +faultInjection = false + +# Required knobs for BulkLoad functionality (from BulkDumpingS3.toml) +[[knobs]] +manifest_count_max_per_bulkload_task = 10 +min_shard_bytes = 10000 +bulkload_sim_failure_injection = false +shard_encode_location_metadata = true +enable_read_lock_on_range = true +enable_version_vector = false +enable_version_vector_tlog_unicast = false +enable_version_vector_reply_recovery = false +min_byte_sampling_probability = 0.5 +cc_enforce_use_unfit_dd_in_sim = true +disable_audit_storage_final_replica_check_in_sim = true +max_trace_lines = 5000000 +# Allow more time for BulkDump job to complete (Linux runs 4x slower than macOS) +bulkdump_job_timeout = 1200 +bulkload_job_timeout = 1200 + +# Disable buggified delays +[[flow_knobs]] +MAX_BUGGIFIED_DELAY = 0.0 + +# S3/Blobstore settings for stability/determinism +blobstore_max_connection_life = 300 +blobstore_request_timeout_min = 300 +blobstore_request_tries = 5 +blobstore_connect_tries = 5 +blobstore_connect_timeout = 30 +http_send_size = 1024 +http_read_size = 1024 +connection_monitor_loop_time = 0.1 +connection_monitor_timeout = 1.0 +connection_monitor_idle_timeout = 60.0 +dd_team_zero_server_left_log_delay = 0 +dd_rebalance_parallelism = 1 + +[[test]] +testTitle = 'BackupS3BlobBulkLoadRestoreNarrowFleet' +useDB = true +clearAfterTest = false +simBackupAgents = 'BackupToFile' +waitForQuiescence = false +connectionFailuresDisableDuration = 1000000 +runFailureWorkloads = false +timeout = 3600 + + [[test.workload]] + testName = 'Cycle' + nodeCount = 2000 + transactionsPerSecond = 100.0 + testDuration = 30.0 + + [[test.workload]] + testName = 'BackupS3BlobCorrectness' + backupAfter = 10.0 + restoreAfter = 600.0 + abortAndRestartAfter = 0.0 + stopDifferentialAfter = 0.0 + performRestore = true + backupRangesCount = -1 + skipDirtyRestore = false + backupURL = 'blobstore://mocks3:mocksecret:mocktoken@127.0.0.1:8080/backup_container?bucket=backup_bucket®ion=us-east-1&secure_connection=0&cwpf=1&cu=1' + # BulkDump/BulkLoad integration options + snapshotMode = 2 # 2 = BOTH (creates range files AND SST files for comparison) + useRangeFileRestore = false # false = use BulkLoad for restore + # Validation: Compare BulkLoad-restored vs traditional-restored using audit_storage validate_restore + # Compares BulkLoad-restored (normalKeys) vs traditional-restored (prefix) + performValidation = true diff --git a/tests/slow/BackupS3BlobBulkLoadRestoreWithChaos.toml b/tests/slow/BackupS3BlobBulkLoadRestoreWithChaos.toml index 475f04ca65f..dcfc3eb3911 100644 --- a/tests/slow/BackupS3BlobBulkLoadRestoreWithChaos.toml +++ b/tests/slow/BackupS3BlobBulkLoadRestoreWithChaos.toml @@ -31,6 +31,13 @@ minimumRegions = 2 # Ensure enough storage servers for non-overlapping BulkLoad teams # Triple replication needs 6+ servers (3 src + 3 different dest) to avoid overlapping extraMachineCountDC = 3 +# Dedicated storage-class machines so bulk-load team selection can find a team disjoint from the +# source: it rejects any team sharing a server with the source (DDTeamCollection). Without these, +# runs fail with DDBulkLoadEngineTaskGetTeamFailedToFindValidTeam reporting ValidTeamSize 0 with +# every candidate team duplicated, the task is marked Error, and the restore is incomplete. +# extraMachineCountDC above adds ordinary machines, which do not supply that capacity. Per DC, so +# this holds for the multi-region variants too. See tests/fast/BulkLoading.toml for the same guard. +extraStorageMachineCountPerDC = 5 config = "triple usable_regions=1 storage_engine=ssd-2 perpetual_storage_wiggle=0 commit_proxies=3 grv_proxies=3 resolvers=3 logs=3" diff --git a/tests/slow/DegradedMultiRegionStatus.toml b/tests/slow/DegradedMultiRegionStatus.toml new file mode 100644 index 00000000000..e8a283b9f76 --- /dev/null +++ b/tests/slow/DegradedMultiRegionStatus.toml @@ -0,0 +1,40 @@ +# Integration test for the "degraded multi-region" status signal. +# +# Simulated fearless topology (generateFearless=true) with two usable regions: +# - region 0 (primary): datacenter "0" (priority 2) + satellite "2" (priority 1) +# - region 1 (remote): datacenter "1" (priority 1) + satellite "3" (priority 1) +# - usable_regions=2, coordinators placed in non-primary DCs (minimumRegions=2) +# +# The workload kills ONLY datacenter "0" (the primary of the first region), leaving +# its satellite "2" and the remote region alive. The cluster fails over to region 1, +# recovers data from the surviving satellite "2", and recovery gets stuck at +# accepting_commits with the dead primary's log set not recruited +# (allLogs == false => RemoteRegionLogsMissing=1 => degraded_multi_region=true). + +[configuration] +generateFearless = true +minimumRegions = 2 +coordinators = 3 +datacenters = 4 +simHTTPServerEnabled = false +satelliteRedundancyMode = "one_satellite_single" +config = "single" +desiredTLogCount = 2 +buggify = false +extraMachineCountDC = 2 + +[[test]] +testTitle = 'DegradedMultiRegionStatus' +clearAfterTest = false +runFailureWorkloads = false +runConsistencyCheck = true + + [[test.workload]] + testName = 'Cycle' + nodeCount = 100 + transactionsPerSecond = 100.0 + testDuration = 3.0 + + [[test.workload]] + testName = 'DegradedMultiRegionStatus' + testDuration = 200.0